Coverage Report

Created: 2026-08-31 06:22

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/EncoderLib/EncCu.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or
4
other Intellectual Property Rights other than the copyrights concerning
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     EncCu.cpp
45
    \brief    Coding Unit (CU) encoder class
46
*/
47
48
#include "EncCu.h"
49
#include "EncLib.h"
50
#include "Analyze.h"
51
#include "EncPicture.h"
52
#include "EncModeCtrl.h"
53
#include "BitAllocation.h"
54
#include "EncStage.h"
55
56
#include "CommonLib/dtrace_codingstruct.h"
57
#include "CommonLib/Picture.h"
58
#include "CommonLib/UnitTools.h"
59
#include "CommonLib/dtrace_buffer.h"
60
#include "CommonLib/TimeProfiler.h"
61
#include "CommonLib/SearchSpaceCounter.h"
62
63
#include <mutex>
64
#include <cmath>
65
#include <algorithm>
66
67
//! \ingroup EncoderLib
68
//! \{
69
70
namespace vvenc {
71
72
const MergeIdxPair EncCu::m_GeoModeTest[GEO_MAX_NUM_CANDS] = { MergeIdxPair{0, 1}, MergeIdxPair{1, 0}, MergeIdxPair{0, 2}, MergeIdxPair{1, 2}, MergeIdxPair{2, 0},
73
                                                               MergeIdxPair{2, 1}, MergeIdxPair{0, 3}, MergeIdxPair{1, 3}, MergeIdxPair{2, 3}, MergeIdxPair{3, 0},
74
                                                               MergeIdxPair{3, 1}, MergeIdxPair{3, 2}, MergeIdxPair{0, 4}, MergeIdxPair{1, 4}, MergeIdxPair{2, 4},
75
                                                               MergeIdxPair{3, 4}, MergeIdxPair{4, 0}, MergeIdxPair{4, 1}, MergeIdxPair{4, 2}, MergeIdxPair{4, 3},
76
                                                               MergeIdxPair{0, 5}, MergeIdxPair{1, 5}, MergeIdxPair{2, 5}, MergeIdxPair{3, 5}, MergeIdxPair{4, 5},
77
                                                               MergeIdxPair{5, 0}, MergeIdxPair{5, 1}, MergeIdxPair{5, 2}, MergeIdxPair{5, 3}, MergeIdxPair{5, 4} };
78
79
80
// Shape coefSquareCUsFasterFastMedium (2 x 5 x 2 x 2 x 2): preset (faster and fast + medium) x cusize x nspred x sptype x numcoef
81
82
const double EncCu::coefSquareCUsFasterFastMedium[2][5][2][2][2] = {
83
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
84
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
85
{{{-1.00000000, -1.00000000, }, {0.07848505, 0.00225808, }, },  {{-1.00000000, -1.00000000, }, {0.07509575, 0.00204789, }, },  },
86
{{{-1.00000000, -1.00000000, }, {0.10833051, 0.00053144, }, },  {{-1.00000000, -1.00000000, }, {0.08304352, 0.00142876, }, },  },
87
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
88
},
89
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
90
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
91
{{{0.06852235, 0.00388054, }, {0.09236045, 0.00084528, }, },  {{0.06955832, 0.00289679, }, {0.09598522, 0.00096187, }, },  },
92
{{{0.07268085, 0.00302796, }, {0.09323753, 0.00050996, }, },  {{0.06123618, 0.00471601, }, {0.09253389, 0.00046826, }, },  },
93
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
94
},
95
};
96
97
// Shape coefSquareCUsSlowSlower (2 x 5 x 2 x 5 x 2 x 2 x 2): preset (Slow + Slower) x cusize x nspred x qtdepth x mtdepth x sptype x numcoef
98
99
const double EncCu::coefSquareCUsSlowSlower[2][5][2][5][2][2][2] = {
100
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
101
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
102
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
103
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
104
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
105
},
106
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
107
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
108
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
109
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
110
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
111
},
112
},
113
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
114
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
115
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
116
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
117
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
118
},
119
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
120
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
121
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
122
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
123
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
124
},
125
},
126
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
127
{{{0.06057349, 0.00447803, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
128
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06948736, 0.00327774, }, {0.06396861, 0.00789923, }, }, },
129
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
130
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
131
},
132
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
133
{{{0.05526461, 0.00436703, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
134
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05661641, 0.00546358, }, {0.06680048, 0.00670604, }, }, },
135
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
136
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
137
},
138
},
139
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
140
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06556926, 0.00455697, }, {0.06803347, 0.00365396, }, }, },
141
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
142
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
143
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
144
},
145
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
146
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06394328, 0.00403533, }, {0.06688063, 0.00272684, }, }, },
147
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
148
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
149
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
150
},
151
},
152
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05533525, 0.00235007, }, {0.07184475, 0.00086154, }, }, },
153
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
154
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
155
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
156
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
157
},
158
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
159
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
160
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
161
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
162
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
163
},
164
},
165
},
166
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
167
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
168
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
169
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
170
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
171
},
172
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
173
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
174
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
175
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
176
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
177
},
178
},
179
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
180
{{{0.04770815, 0.00774597, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
181
{{{0.05509813, 0.00812349, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
182
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04719064, 0.01023887, }, {0.04641434, 0.01279769, }, }, },
183
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
184
},
185
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
186
{{{0.02777251, 0.00811989, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
187
{{{0.04848973, 0.00628111, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
188
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04924294, 0.00769725, }, {0.04491680, 0.01120645, }, }, },
189
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
190
},
191
},
192
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
193
{{{0.06856533, 0.00418949, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
194
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.07059085, 0.00348835, }, {0.06051544, 0.00821365, }, }, },
195
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
196
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
197
},
198
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
199
{{{0.07461087, 0.00229937, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
200
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05893714, 0.00513168, }, {0.05809189, 0.00772994, }, }, },
201
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
202
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
203
},
204
},
205
{{{{{0.06922080, 0.00274816, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
206
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06176300, 0.00374356, }, {0.06085891, 0.00361274, }, }, },
207
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
208
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
209
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
210
},
211
{{{{0.06611224, 0.00256748, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
212
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06364765, 0.00282819, }, {0.05888407, 0.00320947, }, }, },
213
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
214
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
215
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
216
},
217
},
218
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05637817, 0.00184637, }, {0.06937475, 0.00091860, }, }, },
219
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
220
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
221
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
222
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
223
},
224
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
225
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
226
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
227
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
228
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
229
},
230
},
231
},
232
};
233
234
// ====================================================================================================================
235
EncCu::EncCu()
236
19.2k
  : m_CtxCache          ( nullptr )
237
19.2k
  , m_globalCtuQpVector ( nullptr )
238
19.2k
  , m_wppMutex          ( nullptr )
239
19.2k
  , m_CABACEstimator    ( nullptr )
240
19.2k
{
241
19.2k
}
242
243
void EncCu::initPic( Picture* pic )
244
4.82k
{
245
4.82k
  m_cInterSearch.setSearchRange( pic->cs->slice, *m_pcEncCfg );
246
247
4.82k
  m_wppMutex = (m_pcEncCfg->m_numThreads > 0 ) ? &pic->wppMutex : nullptr;
248
4.82k
}
249
250
void EncCu::initSlice( const Slice* slice )
251
4.82k
{
252
4.82k
  m_cTrQuant.setLambdas( slice->getLambdas() );
253
4.82k
  m_cRdCost.setLambda( slice->getLambdas()[0], slice->sps->bitDepths );
254
4.82k
}
255
256
void EncCu::setCtuEncRsrc( CABACWriter* cabacEstimator, CtxCache* ctxCache, ReuseUniMv* pReuseUniMv, BlkUniMvInfoBuffer* pBlkUniMvInfoBuffer, AffineProfList* pAffineProfList, IbcBvCand* pCachedBvs )
257
3.76k
{
258
3.76k
  m_CABACEstimator = cabacEstimator;
259
3.76k
  m_CtxCache       = ctxCache;
260
3.76k
  m_cIntraSearch.setCtuEncRsrc( cabacEstimator, ctxCache );
261
3.76k
  m_cInterSearch.setCtuEncRsrc( cabacEstimator, ctxCache, pReuseUniMv, pBlkUniMvInfoBuffer, pAffineProfList, pCachedBvs );
262
3.76k
}
263
264
void EncCu::setUpLambda (Slice& slice, const double dLambda, const int iQP, const bool setSliceLambda, const bool saveUnadjusted)
265
8.58k
{
266
  // store lambda
267
8.58k
  m_cRdCost.setLambda( dLambda, slice.sps->bitDepths );
268
269
  // for RDO
270
  // in RdCost there is only one lambda because the luma and chroma bits are not separated, instead we weight the distortion of chroma.
271
8.58k
  double dLambdas[MAX_NUM_COMP] = { dLambda };
272
25.7k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
273
17.1k
  {
274
17.1k
    const ComponentID compID = ComponentID( compIdx );
275
17.1k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
276
17.1k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, iQP) + chromaQPOffset;
277
17.1k
    double tmpWeight         = pow( 2.0, ( iQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
278
17.1k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
279
17.1k
    {
280
17.1k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
281
17.1k
    }
282
17.1k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
283
17.1k
    dLambdas[compIdx] = dLambda / tmpWeight;
284
17.1k
  }
285
286
  // for RDOQ
287
8.58k
  m_cTrQuant.setLambdas( dLambdas );
288
289
  // for SAO, ALF
290
8.58k
  if (setSliceLambda)
291
4.82k
  {
292
4.82k
    slice.setLambdas( dLambdas );
293
4.82k
  }
294
8.58k
  if( saveUnadjusted )
295
8.58k
  {
296
8.58k
    m_cRdCost.saveUnadjustedLambda();
297
8.58k
  }
298
8.58k
}
299
300
void EncCu::updateLambda(const Slice& slice, const double ctuLambda, const int ctuQP, const int newQP, const bool saveUnadjusted)
301
3.41k
{
302
3.41k
  const double  corrFactor = pow (2.0, double (newQP - ctuQP) / 3.0);
303
3.41k
  const double  newLambda  = ctuLambda * corrFactor;
304
  
305
3.41k
  double newLambdas[MAX_NUM_COMP] = { newLambda };
306
10.2k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
307
6.82k
  {
308
6.82k
    const ComponentID compID = ComponentID( compIdx );
309
6.82k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
310
6.82k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, newQP) + chromaQPOffset;
311
6.82k
    double tmpWeight         = pow( 2.0, ( newQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
312
6.82k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
313
6.82k
    {
314
6.82k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
315
6.82k
    }
316
6.82k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
317
6.82k
    newLambdas[compIdx] = newLambda / tmpWeight;
318
6.82k
  }
319
320
3.41k
  m_cTrQuant.setLambdas ( newLambdas);
321
3.41k
  m_cRdCost.setLambda   ( newLambda, slice.sps->bitDepths);
322
323
3.41k
  if (saveUnadjusted)
324
3.41k
  {
325
3.41k
    m_cRdCost.saveUnadjustedLambda(); // TODO hlm: check if this actually improves the overall quality
326
3.41k
  }
327
3.41k
}
328
329
void EncCu::init( const VVEncCfg& encCfg, const SPS& sps, std::vector<int>* const globalCtuQpVector, Ctx* syncPicCtx, RateCtrl* pRateCtrl )
330
19.2k
{
331
19.2k
  DecCu::init( &m_cTrQuant, &m_cIntraSearch, &m_cInterSearch, encCfg.m_internChromaFormat );
332
19.2k
  m_cRdCost.create     ();
333
19.2k
  m_cRdCost.setCostMode( encCfg.m_costMode );
334
19.2k
  if ( encCfg.m_lumaLevelToDeltaQPEnabled == 1 )
335
0
  {
336
0
    m_cRdCost.setChromaFormat( encCfg.m_internChromaFormat );
337
0
    m_cRdCost.initLumaLevelToWeightTable( encCfg.m_internalBitDepth[ CH_L ] );
338
0
    xInitLumaDeltaQpLUT();
339
0
  }
340
341
19.2k
  m_modeCtrl.init     ( encCfg, &m_cRdCost );
342
19.2k
  m_cIntraSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_SortedPelUnitBufs, m_unitCache );
343
19.2k
  m_cInterSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_modeCtrl, m_cIntraSearch.getSaveCSBuf() );
344
19.2k
  m_cTrQuant.init     ( nullptr, encCfg.m_RDOQ, encCfg.m_useRDOQTS, false, true, encCfg.m_quantThresholdVal );
345
346
19.2k
  m_syncPicCtx = syncPicCtx;                         ///< context storage for state of contexts at the wavefront/WPP/entropy-coding-sync second CTU of tile-row used for estimation
347
19.2k
  m_pcRateCtrl = pRateCtrl;
348
349
  // Initialise scaling lists: The encoder will only use the SPS scaling lists. The PPS will never be marked present.
350
19.2k
  const int maxLog2TrDynamicRange[ MAX_NUM_CH ] = { sps.getMaxLog2TrDynamicRange(), sps.getMaxLog2TrDynamicRange() };
351
19.2k
  m_cTrQuant.getQuant()->setFlatScalingList( maxLog2TrDynamicRange, sps.bitDepths );
352
353
19.2k
  m_pcEncCfg       = &encCfg;
354
355
19.2k
  m_GeoCostList.init( encCfg.m_maxNumGeoCand );
356
357
19.2k
  unsigned      uiMaxSize    = encCfg.m_CTUSize;
358
19.2k
  ChromaFormat  chromaFormat = encCfg.m_internChromaFormat;
359
360
19.2k
  Area ctuArea = Area( 0, 0, uiMaxSize, uiMaxSize );
361
362
19.2k
  m_mergeItemList.init( encCfg.m_maxMergeRdCandNumTotal, m_pcEncCfg->m_Geo > 1 ? 3 : 1, chromaFormat, uiMaxSize, uiMaxSize );
363
364
250k
  for( int i = 0; i < maxCuDepth; i++ )
365
231k
  {
366
231k
    Area area = Area( 0, 0, uiMaxSize >> ( i >> 1 ), uiMaxSize >> ( ( i + 1 ) >> 1 ) );
367
368
231k
    if( area.width < (1 << MIN_CU_LOG2) || area.height < (1 << MIN_CU_LOG2) )
369
19.2k
    {
370
19.2k
      m_pTempCS[i] = m_pBestCS[i] = nullptr;
371
19.2k
      continue;
372
19.2k
    }
373
374
212k
    m_pTempCS[i] = new CodingStructure( m_unitCache, nullptr );
375
212k
    m_pBestCS[i] = new CodingStructure( m_unitCache, nullptr );
376
377
212k
    m_pTempCS[i]->createForSearch( chromaFormat, area );
378
212k
    m_pBestCS[i]->createForSearch( chromaFormat, area );
379
380
212k
    m_pOrgBuffer[i].create( chromaFormat, area );
381
212k
    m_pRspBuffer[i].create( CHROMA_400, area );
382
212k
  }
383
384
19.2k
  m_pTempCS2 = new CodingStructure( m_unitCache, nullptr );
385
19.2k
  m_pBestCS2 = new CodingStructure( m_unitCache, nullptr );
386
387
19.2k
  m_pTempCS2->createForSearch( chromaFormat, ctuArea );
388
19.2k
  m_pBestCS2->createForSearch( chromaFormat, ctuArea );
389
390
19.2k
  m_cuChromaQpOffsetIdxPlus1 = 0;
391
19.2k
  m_tempQpDiff = 0;
392
19.2k
  m_globalCtuQpVector = globalCtuQpVector;
393
394
19.2k
  m_SortedPelUnitBufs.create( chromaFormat, uiMaxSize, uiMaxSize );
395
396
289k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
397
269k
  {
398
269k
    m_aTmpStorageLCU[i].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
399
269k
  }
400
134k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
401
115k
  {
402
115k
    m_acMergeTmpBuffer[ui].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
403
115k
  }
404
405
19.2k
  const unsigned maxDepth = 2 * MAX_CU_SIZE_IDX;
406
19.2k
  m_CtxBuffer.resize( maxDepth );
407
19.2k
  m_CurrCtx = 0;
408
19.2k
  if( encCfg.m_EDO )
409
19.2k
    m_dbBuffer.create( chromaFormat, Area( 0, 0, uiMaxSize, uiMaxSize ), 0, 8 );
410
411
19.2k
  m_MergeSimpleFlag = 0;
412
19.2k
  m_tileIdx = 0;
413
19.2k
}
414
415
416
void EncCu::destroy()
417
19.2k
{
418
250k
  for( int i = 0; i < maxCuDepth; i++ )
419
231k
  {
420
231k
    if( m_pTempCS[i] )
421
212k
    {
422
212k
      m_pTempCS[i]->destroy();
423
212k
      delete m_pTempCS[i]; m_pTempCS[i] = nullptr;
424
212k
    }
425
426
231k
    if( m_pBestCS[i] )
427
212k
    {
428
212k
      m_pBestCS[i]->destroy();
429
212k
      delete m_pBestCS[i]; m_pBestCS[i] = nullptr;
430
212k
    }
431
432
231k
    m_pOrgBuffer[i].destroy();
433
231k
    m_pRspBuffer[i].destroy();
434
231k
  }
435
436
19.2k
  m_pTempCS2->destroy();
437
19.2k
  m_pBestCS2->destroy();
438
439
19.2k
  delete m_pTempCS2; m_pTempCS2 = nullptr;
440
19.2k
  delete m_pBestCS2; m_pBestCS2 = nullptr;
441
442
19.2k
  m_SortedPelUnitBufs.destroy();
443
444
289k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
445
269k
  {
446
269k
    m_aTmpStorageLCU[i].destroy();
447
269k
  }
448
134k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
449
115k
  {
450
115k
    m_acMergeTmpBuffer[ui].destroy();
451
115k
  }
452
453
454
19.2k
  m_dbBuffer.destroy();
455
19.2k
}
456
457
458
EncCu::~EncCu()
459
19.2k
{
460
19.2k
  destroy();
461
19.2k
}
462
463
// ====================================================================================================================
464
// Public member functions
465
// ====================================================================================================================
466
467
void EncCu::encodeCtu( Picture* pic, int (&prevQP)[MAX_NUM_CH], uint32_t ctuXPosInCtus, uint32_t ctuYPosInCtus )
468
3.76k
{
469
3.76k
  CodingStructure&     cs          = *pic->cs;
470
3.76k
  Slice*               slice       = cs.slice;
471
3.76k
  const PreCalcValues& pcv         = *cs.pcv;
472
473
#if ENABLE_MEASURE_SEARCH_SPACE
474
  if( ctuXPosInCtus == 0 && ctuYPosInCtus == 0 )
475
  {
476
    g_searchSpaceAcc.picW = pic->lwidth();
477
    g_searchSpaceAcc.picH = pic->lheight();
478
    g_searchSpaceAcc.addSlice( slice->isIntra(), slice->depth );
479
  }
480
481
#endif
482
3.76k
  const int ctuRsAddr                 = ctuYPosInCtus * pcv.widthInCtus + ctuXPosInCtus;
483
484
3.76k
  const Position pos (ctuXPosInCtus * pcv.maxCUSize, ctuYPosInCtus * pcv.maxCUSize);
485
3.76k
  const UnitArea ctuArea( cs.area.chromaFormat, Area( pos.x, pos.y, pcv.maxCUSize, pcv.maxCUSize ) );
486
3.76k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "ctu", ctuRsAddr ) );
487
488
3.76k
  const int tileXPosInCtus = cs.pps->tileColBd[cs.pps->ctuToTileCol[ctuXPosInCtus]];
489
3.76k
  const int tileYPosInCtus = cs.pps->tileRowBd[cs.pps->ctuToTileRow[ctuYPosInCtus]];
490
491
3.76k
  if( ( cs.slice->sliceType != VVENC_I_SLICE || cs.sps->IBC ) && ctuXPosInCtus == tileXPosInCtus )
492
2.14k
  {
493
2.14k
    const int tileRowId = cs.pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus );
494
2.14k
    cs.motionLutBuf[tileRowId].lut.resize( 0 );
495
2.14k
    cs.motionLutBuf[tileRowId].lutIbc.resize( 0 );
496
2.14k
  }
497
498
3.76k
  if( ( m_pcEncCfg->m_ensureWppBitEqual || m_pcEncCfg->m_entropyCodingSyncEnabled ) && ctuXPosInCtus == tileXPosInCtus )
499
2.14k
  {
500
2.14k
    m_CABACEstimator->initCtxModels( *slice );
501
502
2.14k
    if( m_pcEncCfg->m_entropyCodingSyncEnabled && ( ctuYPosInCtus > tileYPosInCtus ) )
503
0
    {
504
0
      m_CABACEstimator->getCtx() = m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus - 1 )];
505
0
    }
506
507
2.14k
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
508
2.14k
  }
509
1.62k
  else if( ctuXPosInCtus == tileXPosInCtus && ctuYPosInCtus == tileYPosInCtus )
510
0
  {
511
0
    m_CABACEstimator->initCtxModels( *slice );
512
0
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
513
0
  }
514
515
3.76k
  xCompressCtu( cs, ctuArea, ctuRsAddr, prevQP );
516
517
3.76k
  m_CABACEstimator->resetBits();
518
3.76k
  m_CABACEstimator->coding_tree_unit( cs, ctuArea, prevQP, ctuRsAddr, true, true );
519
520
  // Store probabilities of second CTU in line into buffer - used only if wavefront-parallel-processing is enabled.
521
3.76k
  if( ctuXPosInCtus == tileXPosInCtus && m_pcEncCfg->m_entropyCodingSyncEnabled )
522
0
  {
523
0
    m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus )] = m_CABACEstimator->getCtx();
524
0
  }
525
526
3.76k
  DTRACE_AREA_CRC( g_trace_ctx, D_CRC, cs, ctuArea );
527
3.76k
}
528
529
// ====================================================================================================================
530
// Protected member functions
531
// ====================================================================================================================
532
533
void EncCu::xCompressCtu( CodingStructure& cs, const UnitArea& area, const unsigned ctuRsAddr, const int prevQP[] )
534
3.76k
{
535
3.76k
  m_tileIdx = cs.pps->getTileIdx( area.lumaPos() );
536
537
3.76k
  m_modeCtrl.initCTUEncoding( *cs.slice, m_tileIdx );
538
539
  // init the partitioning manager
540
3.76k
  Partitioner *partitioner = &m_partitioner;
541
3.76k
  partitioner->initCtu( area, CH_L, *cs.slice );
542
  
543
3.76k
  const Position& lumaPos = area.lumaPos();
544
3.76k
  const bool leftSameTile  = lumaPos.x == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset(-1, 0) );
545
3.76k
  const bool aboveSameTile = lumaPos.y == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset( 0,-1) );
546
3.76k
  m_EDO = (!m_pcEncCfg->m_tileParallelCtuEnc || (leftSameTile && aboveSameTile)) ? m_pcEncCfg->m_EDO : 0;
547
  
548
3.76k
  if( m_pcEncCfg->m_IBCMode )
549
3.76k
  {
550
3.76k
    m_cInterSearch.resetCtuRecordIBC();
551
3.76k
  }
552
553
  // init current context pointer
554
3.76k
  m_CurrCtx = m_CtxBuffer.data();
555
556
3.76k
  PelStorage* orgBuffer = &m_pOrgBuffer[0];
557
3.76k
  PelStorage* rspBuffer = &m_pRspBuffer[0];
558
3.76k
  CodingStructure *tempCS =  m_pTempCS [0];
559
3.76k
  CodingStructure *bestCS =  m_pBestCS [0];
560
3.76k
  cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
561
3.76k
  cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
562
3.76k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_COMPRESS_CU, tempCS, CH_L );
563
564
  // copy the relevant area
565
3.76k
  UnitArea clippedArea = clipArea( partitioner->currArea(), cs.area );
566
3.76k
  CPelUnitBuf org = cs.picture->getFilteredOrigBuffer().valid() ? cs.picture->getFiltOrigBuf( clippedArea ) : cs.picture->getOrigBuf( clippedArea );
567
3.76k
  tempCS->getOrgBuf( clippedArea ).copyFrom( org );
568
3.76k
  tempCS->currQP[CH_L] = bestCS->currQP[CH_L] =
569
3.76k
  tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
570
3.76k
  tempCS->prevQP[CH_L] = bestCS->prevQP[CH_L] = prevQP[CH_L];
571
572
3.76k
  xCompressCU( tempCS, bestCS, *partitioner );
573
  // all signals were already copied during compression if the CTU was split - at this point only the structures are copied to the top level CS
574
  
575
  // Ensure that a coding was found
576
  // Selected mode's RD-cost must be not MAX_DOUBLE.
577
3.76k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
578
3.76k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
579
3.76k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
580
581
3.76k
  if ( m_wppMutex ) m_wppMutex->lock();
582
583
3.76k
  cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
584
585
3.76k
  if ( m_wppMutex ) m_wppMutex->unlock();
586
587
3.76k
  if( CS::isDualITree( cs ) && isChromaEnabled( cs.pcv->chrFormat ) )
588
3.76k
  {
589
3.76k
    m_CABACEstimator->getCtx() = m_CurrCtx->start;
590
591
3.76k
    partitioner->initCtu( area, CH_C, *cs.slice );
592
593
3.76k
    cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
594
3.76k
    cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
595
3.76k
    tempCS->currQP[CH_C] = bestCS->currQP[CH_C] =
596
3.76k
    tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
597
3.76k
    tempCS->prevQP[CH_C] = bestCS->prevQP[CH_C] = prevQP[CH_C];
598
599
3.76k
    xCompressCU( tempCS, bestCS, *partitioner );
600
    
601
    // Ensure that a coding was found
602
    // Selected mode's RD-cost must be not MAX_DOUBLE.
603
3.76k
    CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
604
3.76k
    CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
605
3.76k
    CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
606
607
3.76k
    if ( m_wppMutex ) m_wppMutex->lock();
608
609
3.76k
    cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
610
611
3.76k
    if ( m_wppMutex ) m_wppMutex->unlock();
612
3.76k
  }
613
614
  // reset context states and uninit context pointer
615
3.76k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
616
3.76k
  m_CurrCtx                  = 0;
617
3.76k
}
618
619
620
621
bool EncCu::xCheckBestMode( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const bool useEDO )
622
283k
{
623
283k
  bool bestCSUpdated = false;
624
625
283k
  if( !tempCS->cus.empty() )
626
172k
  {
627
172k
    if( tempCS->cus.size() == 1 )
628
129k
    {
629
129k
      const CodingUnit& cu = *tempCS->cus.front();
630
129k
      CHECK( cu.skip && !cu.mergeFlag, "Skip flag without a merge flag is not allowed!" );
631
129k
    }
632
633
172k
    DTRACE_BEST_MODE( tempCS, bestCS, m_cRdCost.getLambda(true), useEDO );
634
635
172k
    if( m_modeCtrl.useModeResult( encTestMode, tempCS, partitioner, useEDO ) )
636
155k
    {
637
155k
      std::swap( tempCS, bestCS );
638
      // store temp best CI for next CU coding
639
155k
      m_CurrCtx->best = m_CABACEstimator->getCtx();
640
155k
      bestCSUpdated = true;
641
155k
    }
642
172k
  }
643
644
  // reset context states
645
283k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
646
283k
  return bestCSUpdated;
647
648
283k
}
649
650
void xCheckFastCuChromaSplitting( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner&  partitioner, ComprCUCtx& cuECtx )
651
0
{
652
0
  const uint32_t uiLPelX = tempCS->area.Cb().lumaPos().x;
653
0
  const uint32_t uiTPelY = tempCS->area.Cb().lumaPos().y;
654
655
0
  int lumaw = 0, lumah = 0;
656
0
  bool splitver      = true;
657
0
  bool splithor      = true;
658
0
  bool qtSplitChroma = true;
659
660
0
  if( partitioner.isSepTree( *tempCS ) && isChroma( partitioner.chType ) )
661
0
  {
662
0
    Position lumaRefPos( uiLPelX, uiTPelY );
663
0
    CodingUnit* colLumaCu = bestCS->lumaCS->getCU( lumaRefPos, CH_L, TREE_D );
664
665
0
    if( colLumaCu )
666
0
    {
667
0
      lumah = colLumaCu->Y().height;
668
0
      lumaw = colLumaCu->Y().width;
669
0
    }
670
0
  }
671
0
  else
672
0
  {
673
0
    return;
674
0
  }
675
676
0
  if( partitioner.getImplicitSplit( *tempCS ) != CU_DONT_SPLIT ) return;
677
678
0
  const CPelBuf orgCb = tempCS->getOrgBuf( COMP_Cb );
679
0
  const CPelBuf orgCr = tempCS->getOrgBuf( COMP_Cr );
680
681
0
  int th1 = FCBP_TH1;
682
683
0
  if( ( lumaw >> getChannelTypeScaleX( CH_C, tempCS->area.chromaFormat ) ) == orgCb.width )
684
0
  {
685
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
686
0
    {
687
0
      splitver      = false;
688
0
      qtSplitChroma = false;
689
0
    }
690
0
  }
691
692
0
  if( ( lumah >> getChannelTypeScaleY( CH_C, tempCS->area.chromaFormat ) ) == orgCb.height )
693
0
  {
694
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
695
0
    {
696
0
      splithor      = false;
697
0
      qtSplitChroma = false;
698
0
    }
699
0
  }
700
701
0
  cuECtx.doHorChromaSplit = splithor;
702
0
  cuECtx.doVerChromaSplit = splitver;
703
0
  cuECtx.doQtChromaSplit  = qtSplitChroma;
704
705
0
  if( orgCb.width == orgCb.height )
706
0
  {
707
0
    int varh_cb, varv_cb;
708
0
    int varh_cr, varv_cr;
709
710
0
    orgCb.calcVarianceSplit( orgCb, orgCb.width, varh_cb, varv_cb );
711
0
    orgCr.calcVarianceSplit( orgCr, orgCr.width, varh_cr, varv_cr );
712
713
0
    if( ( varh_cr*FCBP_TH2 < varv_cr * 100 ) && ( varh_cb*FCBP_TH2 < varv_cb * 100 ) )
714
0
    {
715
0
      cuECtx.doVerChromaSplit = false;
716
0
    }
717
0
    else if( ( varv_cr*FCBP_TH2 < varh_cr * 100 ) && ( varv_cb*FCBP_TH2 < varh_cb * 100 ) )
718
0
    {
719
0
      cuECtx.doHorChromaSplit = false;
720
0
    }
721
0
  }
722
0
}
723
724
void EncCu::xCompressCU( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner )
725
130k
{
726
130k
  const Area& lumaArea = tempCS->area.Y();
727
728
130k
  Slice&   slice      = *tempCS->slice;
729
130k
  const PPS &pps      = *tempCS->pps;
730
130k
  const SPS &sps      = *tempCS->sps;
731
130k
  const uint32_t uiLPelX  = tempCS->area.Y().lumaPos().x;
732
130k
  const uint32_t uiTPelY  = tempCS->area.Y().lumaPos().y;
733
130k
  const bool isBimEnabled = (m_pcEncCfg->m_blockImportanceMapping && !bestCS->picture->m_picShared->m_ctuBimQpOffset.empty());
734
735
130k
  m_modeCtrl.initBlk( tempCS->area, slice.pic->poc );
736
130k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
737
738
130k
  if ((m_pcEncCfg->m_usePerceptQPA || isBimEnabled || m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1) && ( pps.useDQP || m_pcEncCfg->m_maxDeltaQP == 0 ) && isLuma (partitioner.chType) && partitioner.currQgEnable())
739
7.17k
  {
740
7.17k
    const PreCalcValues &pcv = *pps.pcv;
741
7.17k
    Picture* const pic = bestCS->picture;
742
7.17k
    const uint32_t ctuRsAddr = getCtuAddr (partitioner.currQgPos, pcv);
743
744
7.17k
    if (partitioner.currSubdiv == 0) // CTU-level QP adaptation
745
3.76k
    {
746
3.76k
      if (m_pcEncCfg->m_usePerceptQPA)
747
3.76k
      {
748
3.76k
        if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
749
0
        {
750
0
          m_tempQpDiff = pic->ctuAdaptedQP[ctuRsAddr] - BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
751
0
        }
752
753
3.76k
        int clippedMaxDQP = 0;
754
755
3.76k
        if ((!slice.isIntra()) && (pcv.maxCUSize > 64) && // sub-CTU QPA behavior - Museum fix
756
0
            (uiLPelX + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceWidth)) &&
757
0
            (uiTPelY + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceHeight)))
758
0
        {
759
0
          const uint32_t h = lumaArea.height >> 1;
760
0
          const uint32_t w = lumaArea.width  >> 1;
761
0
          const int adQPTL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
762
0
          const int adQPTR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
763
0
          const int adQPBL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
764
0
          const int adQPBR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
765
766
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
767
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = std::min (std::min (adQPTL, adQPTR), std::min (adQPBL, adQPBR));
768
769
0
          if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
770
0
          {
771
0
            if ((m_globalCtuQpVector->size() > ctuRsAddr) && (slice.TLayer == 0) && // last CTU row of non-Intra key-frame
772
0
                (m_pcEncCfg->m_IntraPeriod == 2 * m_pcEncCfg->m_GOPSize) && (ctuRsAddr >= pcv.widthInCtus) && (uiTPelY + pcv.maxCUSize > m_pcEncCfg->m_PadSourceHeight))
773
0
            {
774
0
              m_globalCtuQpVector->at (ctuRsAddr) = m_globalCtuQpVector->at (ctuRsAddr - pcv.widthInCtus); // copy the pumping reducing QP offset from the top CTU neighbor
775
0
              tempCS->currQP[partitioner.chType] = tempCS->baseQP =
776
0
              bestCS->currQP[partitioner.chType] = bestCS->baseQP = tempCS->baseQP - m_globalCtuQpVector->at (ctuRsAddr);
777
0
            }
778
0
            tempCS->currQP[partitioner.chType] = tempCS->baseQP =
779
0
            bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
780
0
          }
781
0
        }
782
3.76k
        else
783
3.76k
        {
784
3.76k
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
785
3.76k
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = pic->ctuAdaptedQP[ctuRsAddr];
786
3.76k
        }
787
788
3.76k
        int deltaQP       = slice.sliceQp - tempCS->baseQP;
789
3.76k
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
790
0
        {
791
0
          clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
792
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
793
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
794
0
        }
795
796
        //the ctuAdaptedQP correspones with the ctuQpaLambda
797
3.76k
        setUpLambda (slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], false, true);
798
3.76k
      }
799
0
      else // isBimEnabled without QPA
800
0
      {
801
0
        const int baseQp  = tempCS->baseQP;
802
0
        int adaptedLumaQP = 0;
803
0
        int lumaQPoffset  = 0;
804
        
805
0
        if( m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1 )
806
0
        {
807
0
          uint32_t meanLuma = 0;
808
0
          if( ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples ) || ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) )
809
0
          {
810
0
            int wdth = ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples )   ? pps.picWidthInLumaSamples - uiLPelX  : lumaArea.width;
811
0
            int hght = ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) ? pps.picHeightInLumaSamples - uiTPelY : lumaArea.height;
812
            
813
0
            int sum = 0;
814
0
            for( int y = 0; y < hght; y++ )
815
0
            {
816
0
              for( int x = 0; x < wdth; x++ )
817
0
              {
818
0
                sum += tempCS->getOrgBuf().Y().at( x, y );
819
0
              }
820
0
            }
821
0
            double avg = (double) sum / ( wdth*hght );
822
0
            meanLuma = (uint32_t)( avg + 0.5 );
823
0
          }
824
0
          else
825
0
          {
826
0
            meanLuma = tempCS->getOrgBuf().Y().getAvg();
827
0
          }
828
829
0
          int lumaBD     = m_pcEncCfg->m_internalBitDepth[ CH_L ];
830
0
          int lumaIdxOrg = Clip3<int>( 0, int( 1 << lumaBD ) - 1, meanLuma );
831
0
          int lumaIdx    = lumaBD < 10 ? lumaIdxOrg << (10 - lumaBD) : lumaBD > 10 ? lumaIdxOrg >> (lumaBD - 10) : lumaIdxOrg;
832
0
          lumaQPoffset   = m_lumaLevelToDeltaQPLUT[lumaIdx];
833
0
          adaptedLumaQP  = Clip3( -sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset );
834
0
        }
835
        
836
0
        if( isBimEnabled )
837
0
        {
838
0
          const unsigned bimQpSize = (unsigned) bestCS->picture->m_picShared->m_ctuBimQpOffset.size();
839
0
          uint32_t ctuAddr         = ctuRsAddr;
840
841
0
          double avgBimOffset = 0.0;
842
843
0
          if (bimQpSize != pcv.sizeInCtus) // re-calculate correct address of BIM CTU QP offset
844
0
          {
845
0
            const unsigned bimCtuSize  = m_pcEncCfg->m_bimCtuSize;
846
0
            const unsigned bimCtuWidth = (pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize;
847
848
0
            ctuAddr = getCtuAddrFromCtuSize (partitioner.currQgPos, Log2 (bimCtuSize), bimCtuWidth);
849
0
            CHECK (ctuAddr >= bimQpSize, "ctuAddr exceeds size of m_ctuBimQpOffset");
850
851
0
            int maxY = m_pcEncCfg->m_CTUSize / bimCtuSize;
852
0
            int maxX = m_pcEncCfg->m_CTUSize / bimCtuSize;
853
0
            if( ( partitioner.currQgPos.y + maxY * bimCtuSize ) > pcv.lumaHeight )
854
0
            {
855
0
              int reduceHeight = ( partitioner.currQgPos.y + ( maxY - 1 ) * bimCtuSize ) - pcv.lumaHeight;
856
0
              int reduceY = ( reduceHeight + bimCtuSize ) / bimCtuSize;
857
0
              maxY = maxY - reduceY;
858
0
            }
859
0
            if( ( partitioner.currQgPos.x + maxX * bimCtuSize ) > pcv.lumaWidth )
860
0
            {
861
0
              int reduceWidth = ( partitioner.currQgPos.x + ( maxX - 1 ) * bimCtuSize ) - pcv.lumaWidth;
862
0
              int reduceX = ( reduceWidth + bimCtuSize ) / bimCtuSize;
863
0
              maxX = maxX - reduceX;
864
0
            }
865
0
            for (int y = 0; y < maxY * bimCtuSize; y = y + bimCtuSize)
866
0
            {
867
0
              for (int x = 0; x < maxX * bimCtuSize; x = x + bimCtuSize)
868
0
              {
869
0
                int theBlockId = ctuAddr + (y / bimCtuSize) * ((pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize) + (x / bimCtuSize);
870
0
                avgBimOffset += pic->m_picShared->m_ctuBimQpOffset[theBlockId];
871
0
              }
872
0
            }
873
0
            if (maxY * maxX > 0)
874
0
            {
875
0
              avgBimOffset = avgBimOffset / (double)(maxY * maxX);
876
0
            }
877
0
            else
878
0
            {
879
0
              avgBimOffset = 0.0;
880
0
            }
881
0
          }
882
0
          else
883
0
          {
884
0
            avgBimOffset = pic->m_picShared->m_ctuBimQpOffset[ctuAddr];
885
0
          }
886
0
          int intAvgBimOffset = avgBimOffset >= 0 ? (int)(0.5 + avgBimOffset) : (int)(-0.5 + avgBimOffset);
887
0
          adaptedLumaQP       = Clip3 (-sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset + intAvgBimOffset );
888
0
        }
889
        
890
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP =
891
0
        bestCS->currQP[partitioner.chType] = bestCS->baseQP = adaptedLumaQP;
892
        
893
0
        int deltaQP         = slice.sliceQp - adaptedLumaQP;
894
0
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
895
0
        {
896
0
          int clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
897
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
898
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
899
0
        }
900
901
0
        updateLambda (slice, slice.getLambdas()[0], baseQp, adaptedLumaQP, true);
902
0
      }
903
3.76k
    }
904
3.41k
    else if (m_pcEncCfg->m_usePerceptQPA && slice.isIntra()) // currSubdiv 2 - use sub-CTU QPA
905
3.41k
    {
906
3.41k
      CHECK ((partitioner.currArea().lwidth() >= pcv.maxCUSize) || (partitioner.currArea().lheight() >= pcv.maxCUSize), "sub-CTU delta-QP error");
907
3.41k
      tempCS->currQP[partitioner.chType] = tempCS->baseQP = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
908
909
3.41k
      if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
910
0
      {
911
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
912
0
      }
913
914
3.41k
      int backupBaseQP  = tempCS->baseQP;
915
3.41k
      int deltaQP       = slice.sliceQp - tempCS->baseQP;
916
3.41k
      int clippedMaxDQP = 0;
917
3.41k
      if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
918
0
      {
919
0
        clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
920
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = clippedMaxDQP;
921
0
      }
922
3.41k
      updateLambda( slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], backupBaseQP, true );
923
3.41k
    }
924
7.17k
  }
925
926
130k
  if (partitioner.currQtDepth == 0)
927
7.53k
  {
928
7.53k
    m_MergeSimpleFlag = 0;
929
7.53k
  }
930
130k
  m_modeCtrl.initCULevel( partitioner, *tempCS, m_MergeSimpleFlag );
931
130k
  m_sbtCostSave[0] = m_sbtCostSave[1] = MAX_DOUBLE;
932
933
130k
  m_CurrCtx->start = m_CABACEstimator->getCtx();
934
935
130k
  m_cuChromaQpOffsetIdxPlus1 = 0;
936
937
130k
  if( slice.chromaQpAdjEnabled && partitioner.currQgChromaEnable() )
938
0
  {
939
    // TODO M0133 : double check encoder decisions with respect to chroma QG detection and actual encode
940
0
    int cuChromaQpOffsetSubdiv = slice.isIntra() ? slice.picHeader->cuChromaQpOffsetSubdivIntra : slice.picHeader->cuChromaQpOffsetSubdivInter;
941
0
    int lgMinCuSize = sps.log2MinCodingBlockSize +
942
0
      std::max<int>(0, floorLog2(sps.CTUSize) - sps.log2MinCodingBlockSize - int((cuChromaQpOffsetSubdiv + 1) / 2));
943
0
    m_cuChromaQpOffsetIdxPlus1 = ( ( uiLPelX >> lgMinCuSize ) + ( uiTPelY >> lgMinCuSize ) ) % ( pps.chromaQpOffsetListLen + 1 );
944
0
  }
945
946
130k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cux", uiLPelX ) );
947
130k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuy", uiTPelY ) );
948
130k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuw", tempCS->area.lwidth() ) );
949
130k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuh", tempCS->area.lheight() ) );
950
130k
  DTRACE( g_trace_ctx, D_COMMON, "@(%4d,%4d) [%2dx%2d]\n", tempCS->area.lx(), tempCS->area.ly(), tempCS->area.lwidth(), tempCS->area.lheight() );
951
952
130k
  if( tempCS->slice->checkLDC )
953
130k
  {
954
130k
    m_bestBcwCost[0] = m_bestBcwCost[1] = std::numeric_limits<double>::max();
955
130k
    m_bestBcwIdx[0] = m_bestBcwIdx[1] = -1;
956
130k
  }
957
958
130k
  m_cInterSearch.resetSavedAffineMotion();
959
130k
  {
960
130k
    const ComprCUCtx &cuECtx      = *m_modeCtrl.comprCUCtx;
961
130k
    const CodingStructure& cs     = *tempCS;
962
130k
    const PartSplit implicitSplit = partitioner.getImplicitSplit( cs );
963
130k
    const bool isBoundary         = implicitSplit != CU_DONT_SPLIT;
964
130k
    const bool lossless           = false;
965
130k
    int qp                        = cs.baseQP;
966
967
#if ENABLE_MEASURE_SEARCH_SPACE
968
    if( !isBoundary )
969
    {
970
      g_searchSpaceAcc.addPartition( partitioner.currArea(), partitioner.isSepTree( *tempCS ) ? partitioner.chType : MAX_NUM_CH );
971
    }
972
973
#endif
974
130k
    if( ! isBoundary )
975
82.5k
    {
976
82.5k
      if (pps.useDQP && partitioner.isSepTree (*tempCS) && isChroma (partitioner.chType))
977
57.2k
      {
978
57.2k
        const ChromaFormat chromaFm = tempCS->area.chromaFormat;
979
57.2k
        const Position chromaCentral (tempCS->area.Cb().chromaPos().offset (tempCS->area.Cb().chromaSize().width >> 1, tempCS->area.Cb().chromaSize().height >> 1));
980
57.2k
        const Position lumaRefPos (chromaCentral.x << getChannelTypeScaleX (CH_C, chromaFm), chromaCentral.y << getChannelTypeScaleY (CH_C, chromaFm));
981
57.2k
        const CodingUnit* colLumaCu = bestCS->lumaCS->getCU (lumaRefPos, CH_L, TREE_D);
982
        // update qp
983
57.2k
        qp = colLumaCu->qp;
984
57.2k
      }
985
986
82.5k
      m_cIntraSearch.reset();
987
988
82.5k
      bool isReuseCU = m_modeCtrl.isReusingCuValid( cs, partitioner, qp );
989
990
82.5k
      bool checkIbc = m_pcEncCfg->m_IBCMode && bestCS->picture->useIBC && (partitioner.chType == CH_L);
991
82.5k
      if ((m_pcEncCfg->m_IBCFastMethod>3) && (cs.area.lwidth() * cs.area.lheight()) > (16 * 16))
992
0
      {
993
0
        checkIbc = false;
994
0
      }
995
82.5k
      if( isReuseCU )
996
0
      {
997
0
        xReuseCachedResult( tempCS, bestCS, partitioner );
998
0
      }
999
82.5k
      else
1000
82.5k
      {
1001
        // add first pass modes
1002
82.5k
        if ( !slice.isIntra() && !slice.isIRAP() && !( cs.area.lwidth() == 4 && cs.area.lheight() == 4 ) && !partitioner.isConsIntra() )
1003
0
        {
1004
          // add inter modes
1005
0
          EncTestMode encTestModeSkip = { ETM_MERGE_SKIP, ETO_STANDARD, qp, lossless };
1006
0
          if (m_modeCtrl.tryMode(encTestModeSkip, cs, partitioner))
1007
0
          {
1008
0
            xCheckRDCostUnifiedMerge(tempCS, bestCS, partitioner, encTestModeSkip);
1009
1010
0
            CodingUnit* cu = bestCS->getCU(partitioner.chType, partitioner.treeType);
1011
0
            if (cu)
1012
0
              cu->mmvdSkip = cu->skip == false ? false : cu->mmvdSkip;
1013
0
          }
1014
0
          EncTestMode encTestMode = { ETM_INTER_ME, ETO_STANDARD, qp, lossless };
1015
0
          if (m_modeCtrl.tryMode(encTestMode, cs, partitioner))
1016
0
          {
1017
0
            xCheckRDCostInter(tempCS, bestCS, partitioner, encTestMode);
1018
0
          }
1019
1020
0
          if (m_pcEncCfg->m_AMVRspeed)
1021
0
          {
1022
0
            double bestIntPelCost = MAX_DOUBLE;
1023
1024
0
            EncTestMode encTestMode = {ETM_INTER_IMV, ETO_STANDARD, qp, lossless};
1025
0
            if( m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1026
0
            {
1027
0
              const bool skipAltHpelIF = ( int( ( encTestMode.opts & ETO_IMV ) >> ETO_IMV_SHIFT ) == 4 ) && ( bestIntPelCost > 1.25 * bestCS->cost );
1028
0
              if (!skipAltHpelIF)
1029
0
              {
1030
0
                xCheckRDCostInterIMV(tempCS, bestCS, partitioner, encTestMode );
1031
0
              }
1032
0
            }
1033
0
          }
1034
0
        }
1035
1036
82.5k
        if (checkIbc && !partitioner.isConsInter())
1037
25.2k
        {
1038
25.2k
          EncTestMode encTestModeIBCMerge = { ETM_IBC_MERGE, ETO_STANDARD, qp, lossless };
1039
25.2k
          if ((m_pcEncCfg->m_IBCFastMethod < 4) && (partitioner.chType == CH_L) && m_modeCtrl.tryMode(encTestModeIBCMerge, cs, partitioner))
1040
25.2k
          {
1041
25.2k
            xCheckRDCostIBCModeMerge2Nx2N(tempCS, bestCS, partitioner, encTestModeIBCMerge);
1042
25.2k
          }
1043
1044
25.2k
          EncTestMode encTestModeIBC = { ETM_IBC, ETO_STANDARD, qp, lossless };
1045
25.2k
          if (m_modeCtrl.tryMode(encTestModeIBC, cs, partitioner))
1046
25.2k
          {
1047
25.2k
            xCheckRDCostIBCMode(tempCS, bestCS, partitioner, encTestModeIBC);
1048
25.2k
          }
1049
25.2k
        }
1050
82.5k
        if( m_EDO && bestCS->cost != MAX_DOUBLE )
1051
22.5k
        {
1052
22.5k
          xCalDebCost(*bestCS, partitioner);
1053
22.5k
        }
1054
1055
        // add intra modes
1056
82.5k
        EncTestMode encTestMode( {ETM_INTRA, ETO_STANDARD, qp, lossless} );
1057
82.5k
        if( !partitioner.isConsInter() && m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1058
82.5k
        {
1059
82.5k
          xCheckRDCostIntra( tempCS, bestCS, partitioner, encTestMode );
1060
82.5k
        }
1061
82.5k
      } // reusing cu
1062
1063
82.5k
      m_modeCtrl.beforeSplit( partitioner );
1064
1065
82.5k
      if (cuECtx.bestCS && ((cuECtx.bestCostNoImv == (MAX_DOUBLE * .5) || cuECtx.isReusingCu) && !slice.isIntra()) )
1066
0
      {
1067
0
        m_cInterSearch.loadGlobalUniMvs( lumaArea, *pps.pcv );
1068
0
      }
1069
1070
82.5k
      if (!cs.slice->isIntra() && (partitioner.chType == CH_L) && ( m_pcEncCfg->m_qtbttSpeedUpMode & 2) && (partitioner.currQtDepth < 3) && bestCS->cus.size())
1071
0
      {
1072
0
        int flagDbefore = (bestCS->cus[0]->mergeFlag && !bestCS->cus[0]->mmvdMergeFlag && !bestCS->cus[0]->ispMode && !bestCS->cus[0]->geo) ? 1 : 0;
1073
0
        if (partitioner.currQtDepth == 0)
1074
0
        {
1075
0
          m_MergeSimpleFlag = flagDbefore;
1076
0
        }
1077
0
        else
1078
0
        {
1079
0
          int markFlag = (partitioner.currQtDepth == 1) ? 1 : 3;
1080
0
          m_MergeSimpleFlag = (flagDbefore << partitioner.currQtDepth) | (m_MergeSimpleFlag & markFlag);
1081
0
        }
1082
0
      }
1083
82.5k
    } //boundary
1084
1085
130k
    if( ( m_pcEncCfg->m_IntraPeriod == 1 ) && ( partitioner.chType == CH_C ) )
1086
0
    {
1087
0
      xCheckFastCuChromaSplitting( tempCS, bestCS, partitioner, *m_modeCtrl.comprCUCtx );
1088
0
    }
1089
    //////////////////////////////////////////////////////////////////////////
1090
    // split modes
1091
130k
    EncTestMode lastTestMode;
1092
1093
130k
    if( cuECtx.qtBeforeBt )
1094
20.7k
    {
1095
20.7k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1096
20.7k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1097
13.4k
      {
1098
13.4k
        lastTestMode = encTestMode;
1099
13.4k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1100
13.4k
      }
1101
20.7k
    }
1102
1103
130k
    if( partitioner.canSplit( CU_HORZ_SPLIT, cs ) )
1104
69.1k
    {
1105
      // add split modes
1106
69.1k
      EncTestMode encTestMode( { ETM_SPLIT_BT_H, ETO_STANDARD, qp, false } );
1107
69.1k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1108
69.1k
      {
1109
69.1k
        lastTestMode = encTestMode;
1110
69.1k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1111
69.1k
      }
1112
69.1k
    }
1113
1114
130k
    if( partitioner.canSplit( CU_VERT_SPLIT, cs ) )
1115
65.9k
    {
1116
      // add split modes
1117
65.9k
      EncTestMode encTestMode( { ETM_SPLIT_BT_V, ETO_STANDARD, qp, false } );
1118
65.9k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1119
65.9k
      {
1120
65.9k
        lastTestMode = encTestMode;
1121
65.9k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1122
65.9k
      }
1123
65.9k
    }
1124
1125
130k
    if( partitioner.canSplit( CU_TRIH_SPLIT, cs ) )
1126
29.1k
    {
1127
      // add split modes
1128
29.1k
      EncTestMode encTestMode( { ETM_SPLIT_TT_H, ETO_STANDARD, qp, false } );
1129
29.1k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1130
1.58k
      {
1131
1.58k
        lastTestMode = encTestMode;
1132
1.58k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1133
1.58k
      }
1134
29.1k
    }
1135
1136
130k
    if( partitioner.canSplit( CU_TRIV_SPLIT, cs ) )
1137
22.8k
    {
1138
      // add split modes
1139
22.8k
      EncTestMode encTestMode( { ETM_SPLIT_TT_V, ETO_STANDARD, qp, false } );
1140
22.8k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1141
1.17k
      {
1142
1.17k
        lastTestMode = encTestMode;
1143
1.17k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1144
1.17k
      }
1145
22.8k
    }
1146
1147
130k
    if( !cuECtx.qtBeforeBt )
1148
109k
    {
1149
109k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1150
109k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1151
27.0k
      {
1152
27.0k
        lastTestMode = encTestMode;
1153
27.0k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1154
27.0k
      }
1155
109k
    }
1156
130k
  }
1157
1158
130k
  if( bestCS->cus.empty() )
1159
0
  {
1160
0
    m_modeCtrl.finishCULevel( partitioner );
1161
0
    return;
1162
0
  }
1163
1164
  //////////////////////////////////////////////////////////////////////////
1165
  // Finishing CU
1166
  // set context states
1167
130k
  m_CABACEstimator->getCtx() = m_CurrCtx->best;
1168
1169
  // QP from last processed CU for further processing
1170
  //copy the qp of the last non-chroma CU
1171
130k
  int numCUInThisNode = (int)bestCS->cus.size();
1172
130k
  if( numCUInThisNode > 1 && bestCS->cus.back()->chType == CH_C && !CS::isDualITree( *bestCS ) )
1173
0
  {
1174
0
    CHECK( bestCS->cus[numCUInThisNode-2]->chType != CH_L, "wrong chType" );
1175
0
    bestCS->prevQP[partitioner.chType] = bestCS->cus[numCUInThisNode-2]->qp;
1176
0
  }
1177
130k
  else
1178
130k
  {
1179
130k
    bestCS->prevQP[partitioner.chType] = bestCS->cus.back()->qp;
1180
130k
  }
1181
130k
  if( ( !slice.isIntra() || slice.sps->IBC )
1182
130k
    && partitioner.chType == CH_L
1183
51.8k
    && bestCS->cus.size() == 1 && ( bestCS->cus.back()->predMode == MODE_INTER || bestCS->cus.back()->predMode == MODE_IBC )
1184
0
    && bestCS->area.Y() == (*bestCS->cus.back()).Y() )
1185
0
  {
1186
0
    const CodingUnit& cu = *bestCS->cus.front();
1187
0
    bool isIbcSmallBlk = CU::isIBC(cu) && (cu.lwidth() * cu.lheight() <= 16);
1188
0
    if (!cu.affine && !cu.geo && !isIbcSmallBlk)
1189
0
    {
1190
0
      const MotionInfo &mi = cu.getMotionInfo();
1191
0
      HPMVInfo hMi( mi, ( mi.interDir() == 3 ) ? cu.BcwIdx : BCW_DEFAULT, cu.imv == IMV_HPEL, CU::isIBC( cu ) );
1192
0
      cu.cs->addMiToLut( CU::isIBC( cu ) ? cu.cs->motionLut.lutIbc : cu.cs->motionLut.lut, hMi );
1193
0
    }
1194
0
  }
1195
1196
130k
  m_modeCtrl.finishCULevel( partitioner );
1197
130k
  if( m_cIntraSearch.getSaveCuCostInSCIPU() && bestCS->cus.size() == 1 )
1198
0
  {
1199
0
    m_cIntraSearch.saveCuAreaCostInSCIPU( Area( partitioner.currArea().lumaPos(), partitioner.currArea().lumaSize() ), bestCS->cost );
1200
0
  }
1201
1202
  // Assert if Best prediction mode is NONE
1203
  // Selected mode's RD-cost must be not MAX_DOUBLE.
1204
130k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
1205
130k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
1206
130k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
1207
130k
}
1208
1209
1210
void EncCu::xCheckModeSplit(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1211
178k
{
1212
178k
  const ModeType modeTypeParent  = partitioner.modeType;
1213
178k
  const TreeType treeTypeParent  = partitioner.treeType;
1214
178k
  const ChannelType chTypeParent = partitioner.chType;
1215
1216
178k
  int signalModeConsVal = CS::signalModeCons( *tempCS, partitioner.currArea(), getPartSplit(encTestMode), modeTypeParent);
1217
178k
  int numRoundRdo = signalModeConsVal == LDT_MODE_TYPE_SIGNAL ? 2 : 1;
1218
178k
  bool skipInterPass = false;
1219
357k
  for( int i = 0; i < numRoundRdo; i++ )
1220
178k
  {
1221
    //change cons modes
1222
178k
    if( signalModeConsVal == LDT_MODE_TYPE_SIGNAL )
1223
0
    {
1224
0
      CHECK( numRoundRdo != 2, "numRoundRdo shall be 2 - [LDT_MODE_TYPE_SIGNAL]" );
1225
0
      partitioner.modeType = (i == 0) ? MODE_TYPE_INTER : MODE_TYPE_INTRA;
1226
0
    }
1227
178k
    else if( signalModeConsVal == LDT_MODE_TYPE_INFER )
1228
0
    {
1229
0
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INFER]" );
1230
0
      partitioner.modeType = MODE_TYPE_INTRA;
1231
0
    }
1232
178k
    else if( signalModeConsVal == LDT_MODE_TYPE_INHERIT )
1233
178k
    {
1234
178k
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INHERIT]" );
1235
178k
      partitioner.modeType = modeTypeParent;
1236
178k
    }
1237
1238
    //for lite intra encoding fast algorithm, set the status to save inter coding info
1239
178k
    if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType == MODE_TYPE_INTER )
1240
0
    {
1241
0
      m_cIntraSearch.setSaveCuCostInSCIPU( true );
1242
0
      m_cIntraSearch.setNumCuInSCIPU( 0 );
1243
0
    }
1244
178k
    else if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType != MODE_TYPE_INTER )
1245
178k
    {
1246
178k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1247
178k
      if( partitioner.modeType == MODE_TYPE_ALL )
1248
178k
      {
1249
178k
        m_cIntraSearch.setNumCuInSCIPU( 0 );
1250
178k
      }
1251
178k
    }
1252
1253
178k
    xCheckModeSplitInternal( tempCS, bestCS, partitioner, encTestMode, modeTypeParent, skipInterPass );
1254
    //recover cons modes
1255
178k
    partitioner.modeType = modeTypeParent;
1256
178k
    partitioner.treeType = treeTypeParent;
1257
178k
    partitioner.chType = chTypeParent;
1258
178k
    if( modeTypeParent == MODE_TYPE_ALL )
1259
178k
    {
1260
178k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1261
178k
      if( numRoundRdo == 2 && partitioner.modeType == MODE_TYPE_INTRA )
1262
0
      {
1263
0
        m_cIntraSearch.initCuAreaCostInSCIPU();
1264
0
      }
1265
178k
    }
1266
178k
    if( skipInterPass )
1267
0
    {
1268
0
      break;
1269
0
    }
1270
178k
  }
1271
178k
}
1272
1273
void EncCu::xCheckModeSplitInternal(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const ModeType modeTypeParent, bool& skipInterPass )
1274
178k
{
1275
178k
  const int qp                     = encTestMode.qp;
1276
178k
  const int oldPrevQp              = tempCS->prevQP[partitioner.chType];
1277
178k
  const auto oldMotionLut          = tempCS->motionLut;
1278
                                   
1279
178k
  const PartSplit split            = getPartSplit( encTestMode );
1280
178k
  const ModeType  modeTypeChild    = partitioner.modeType;
1281
1282
178k
  CHECK( !( split == CU_QUAD_SPLIT || split == CU_HORZ_SPLIT || split == CU_VERT_SPLIT
1283
178k
         || split == CU_TRIH_SPLIT || split == CU_TRIV_SPLIT ), "invalid split type" );
1284
1285
178k
  tempCS->initStructData( qp );
1286
1287
178k
  m_CABACEstimator->getCtx()       = m_CurrCtx->start;
1288
1289
178k
  const uint16_t split_ctx_size    = Ctx::SplitFlag.size() + Ctx::SplitQtFlag.size() + Ctx::SplitHvFlag.size() + Ctx::Split12Flag.size() + Ctx::ModeConsFlag.size();
1290
178k
  const TempCtx  ctxSplitFlags     ( m_CtxCache, SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), m_CABACEstimator->getCtx() ) );
1291
1292
178k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
1293
178k
  m_CABACEstimator->resetBits           ();
1294
178k
  m_CABACEstimator->split_cu_mode       ( split, *tempCS, partitioner );
1295
178k
  partitioner     . modeType            = modeTypeParent;
1296
178k
  m_CABACEstimator->mode_constraint     ( split, *tempCS, partitioner, modeTypeChild );
1297
178k
  partitioner     . modeType            = modeTypeChild;
1298
1299
178k
  const int64_t splitBits   = m_CABACEstimator->getEstFracBits();
1300
1301
178k
  const bool chromaNotSplit = modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTRA;
1302
178k
  const bool isChromaTooBig = isChromaEnabled( tempCS->pps->pcv->chrFormat ) && tempCS->area.Y().maxDim() > tempCS->sps->getMaxTbSize();
1303
178k
  bool       skipSplitTest  = chromaNotSplit && isChromaTooBig;
1304
1305
178k
  if( !skipSplitTest )
1306
178k
  {
1307
178k
    double         a = -1, b = -1;
1308
178k
    const unsigned w       = partitioner.currArea().lwidth();
1309
178k
    const unsigned h       = partitioner.currArea().lheight();
1310
178k
    const bool contextCond = w == h && tempCS->slice->sliceType == VVENC_B_SLICE && isLuma( partitioner.chType ) && m_pcEncCfg->m_splitCostThrParamId >= 0 && m_pcEncCfg->m_splitCostThrParamId <= 3;
1311
1312
178k
    if( contextCond )
1313
0
    {
1314
0
      uint8_t nsPredInd = m_modeCtrl.comprCUCtx->bestNsPredMode.type == ETM_INTRA;
1315
0
      uint8_t szInd     = getLog2( w ) - 3;
1316
0
      uint8_t splitInd  = split == CU_QUAD_SPLIT ? 1 : 0;
1317
0
      if ( m_pcEncCfg->m_splitCostThrParamId <= 1 )
1318
0
      {
1319
0
        a = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][0];
1320
0
        b = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][1];
1321
0
      }
1322
0
      else
1323
0
      {
1324
0
        uint8_t mtInd = (partitioner.currMtDepth == 0);
1325
0
        a = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][0];
1326
0
        b = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][1];
1327
0
      }
1328
0
    }
1329
1330
178k
    if( a > -1 && b > -1 )
1331
0
    {
1332
0
      const double bestNsCost    = m_modeCtrl.comprCUCtx->bestCostBeforeSplit == MAX_DOUBLE ? -1 : m_modeCtrl.comprCUCtx->bestCostBeforeSplit;
1333
0
      const double factor        = 1.0 + b * exp( a * qp );
1334
0
      const double predSplitCost = bestNsCost / factor + splitBits;
1335
0
      skipSplitTest              = bestNsCost >= 0 && predSplitCost >= bestNsCost;
1336
0
    }
1337
178k
    else
1338
178k
    {
1339
178k
      int numChild = 3;
1340
178k
      if( split == CU_VERT_SPLIT || split == CU_HORZ_SPLIT ) numChild--;
1341
43.3k
      else if( split == CU_QUAD_SPLIT ) numChild++;
1342
1343
178k
      int64_t approxBits = m_pcEncCfg->m_qtbttSpeedUp > 0 ? numChild << SCALE_BITS : 0;
1344
1345
178k
      const double factor     = ( tempCS->currQP[partitioner.chType] > 30                              ? 1.1  : 1.075 ) +
1346
178k
                                (   m_pcEncCfg->m_qtbttSpeedUp > 0                                     ? 0.01 : 0.0   ) +
1347
178k
                                ( ( m_pcEncCfg->m_qtbttSpeedUp > 0 && isChroma( partitioner.chType ) ) ? 0.2  : 0.0   );
1348
       
1349
178k
      const double baseCost   = bestCS->cost + bestCS->costDbOffset;
1350
178k
      const double predCost   = baseCost / factor + splitBits + approxBits;
1351
178k
      skipSplitTest           = predCost >= baseCost;
1352
178k
    }
1353
178k
  }
1354
1355
178k
  if( skipSplitTest )
1356
111k
  {
1357
111k
    m_CABACEstimator->getCtx() = SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), ctxSplitFlags );
1358
111k
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1359
111k
    return;
1360
111k
  }
1361
1362
67.4k
  if( partitioner.treeType == TREE_D )
1363
67.4k
  {
1364
67.4k
    if( chromaNotSplit )
1365
0
    {
1366
0
      CHECK( partitioner.chType != CH_L, "chType must be luma" );
1367
0
      partitioner.treeType = TREE_L;
1368
0
    }
1369
67.4k
    else
1370
67.4k
    {
1371
67.4k
      partitioner.treeType = TREE_D;
1372
67.4k
    }
1373
67.4k
  }
1374
1375
67.4k
  partitioner.splitCurrArea( split, *tempCS );
1376
67.4k
  bool qgEnableChildren = partitioner.currQgEnable(); // QG possible at children level
1377
1378
67.4k
  m_CurrCtx++;
1379
1380
67.4k
  AffineMVInfo tmpMVInfo;
1381
67.4k
  bool isAffMVInfoSaved = m_cInterSearch.m_AffineProfList->savePrevAffMVInfo( 0, tmpMVInfo );
1382
1383
67.4k
  BlkUniMvInfo tmpUniMvInfo;
1384
67.4k
  bool         isUniMvInfoSaved = false;
1385
67.4k
  if( !tempCS->slice->isIntra() )
1386
0
  {
1387
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->savePrevUniMvInfo( tempCS->area.Y(), tmpUniMvInfo, isUniMvInfoSaved );
1388
0
  }
1389
1390
67.4k
  DeriveCtx deriveCtx = m_CABACEstimator->getDeriveCtx();
1391
1392
67.4k
  do
1393
172k
  {
1394
172k
    const auto &subCUArea  = partitioner.currArea();
1395
1396
172k
    if( tempCS->picture->Y().contains( subCUArea.lumaPos() ) )
1397
122k
    {
1398
122k
      PelStorage* orgBuffer =  &m_pOrgBuffer[partitioner.currDepth];
1399
122k
      PelStorage* rspBuffer =  &m_pRspBuffer[partitioner.currDepth];
1400
122k
      CodingStructure *tempSubCS = m_pTempCS[partitioner.currDepth];
1401
122k
      CodingStructure *bestSubCS = m_pBestCS[partitioner.currDepth];
1402
1403
122k
      tempCS->initSubStructure( *tempSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1404
122k
      tempCS->initSubStructure( *bestSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1405
1406
      // copy org buffer, need to be done after initSubStructure because of reshaping!
1407
122k
      orgBuffer->copyFrom( tempCS->getOrgBuf( subCUArea ) );
1408
1409
122k
      tempSubCS->bestParent = bestSubCS->bestParent = bestCS;
1410
1411
122k
      xCompressCU(tempSubCS, bestSubCS, partitioner );
1412
1413
122k
      tempSubCS->bestParent = bestSubCS->bestParent = nullptr;
1414
1415
122k
      if( bestSubCS->cost == MAX_DOUBLE )
1416
0
      {
1417
0
        CHECK( split == CU_QUAD_SPLIT, "Split decision reusing cannot skip quad split" );
1418
0
        tempCS->cost = MAX_DOUBLE;
1419
0
        tempCS->costDbOffset = 0;
1420
0
        m_CurrCtx--;
1421
0
        partitioner.exitCurrSplit();
1422
0
        xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1423
0
        if( partitioner.chType == CH_L )
1424
0
        {
1425
0
          tempCS->motionLut = oldMotionLut;
1426
0
        }
1427
1428
0
        m_CABACEstimator->getDeriveCtx() = deriveCtx;
1429
0
        return;
1430
0
      }
1431
1432
122k
      tempCS->useSubStructure( *bestSubCS, partitioner.chType, TREE_D, CS::getArea( *tempCS, subCUArea, partitioner.chType, partitioner.treeType ), partitioner.hasNextPart() || chromaNotSplit );
1433
1434
122k
      if( partitioner.currQgEnable() )
1435
6.82k
      {
1436
6.82k
        tempCS->prevQP[partitioner.chType] = bestSubCS->prevQP[partitioner.chType];
1437
6.82k
      }
1438
122k
      if( partitioner.isConsInter() )
1439
0
      {
1440
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1441
0
        {
1442
0
          CHECK( bestSubCS->cus[i]->predMode != MODE_INTER, "all CUs must be inter mode in an Inter coding region (SCIPU)" );
1443
0
        }
1444
0
      }
1445
122k
      else if( partitioner.isConsIntra() )
1446
0
      {
1447
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1448
0
        {
1449
0
          CHECK( bestSubCS->cus[i]->predMode == MODE_INTER, "all CUs must not be inter mode in an Intra coding region (SCIPU)" );
1450
0
        }
1451
0
      }
1452
1453
122k
      tempSubCS->releaseIntermediateData();
1454
122k
      bestSubCS->releaseIntermediateData();
1455
122k
      if( !tempCS->slice->isIntra() && partitioner.isConsIntra() )
1456
0
      {
1457
0
        tempCS->cost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1458
0
        if( tempCS->cost > bestCS->cost )
1459
0
        {
1460
0
          tempCS->cost = MAX_DOUBLE;
1461
0
          tempCS->costDbOffset = 0;
1462
0
          m_CurrCtx--;
1463
0
          partitioner.exitCurrSplit();
1464
0
          if( partitioner.chType == CH_L )
1465
0
          {
1466
0
            tempCS->motionLut = oldMotionLut;
1467
0
          }
1468
1469
0
          m_CABACEstimator->getDeriveCtx() = deriveCtx;
1470
0
          return;
1471
0
        }
1472
0
      }
1473
122k
    }
1474
172k
  } while( partitioner.nextPart( *tempCS ) );
1475
1476
67.4k
  partitioner.exitCurrSplit();
1477
1478
67.4k
  m_CurrCtx--;
1479
1480
67.4k
  m_CABACEstimator->getDeriveCtx() = deriveCtx;
1481
1482
67.4k
  if( chromaNotSplit )
1483
0
  {
1484
    //Note: In local dual tree region, the chroma CU refers to the central luma CU's QP.
1485
    //If the luma CU QP shall be predQP (no residual in it and before it in the QG), it must be revised to predQP before encoding the chroma CU
1486
    //Otherwise, the chroma CU uses predQP+deltaQP in encoding but is decoded as using predQP, thus causing encoder-decoded mismatch on chroma qp.
1487
0
    if( tempCS->pps->useDQP )
1488
0
    {
1489
      //find parent CS that including all coded CUs in the QG before this node
1490
0
      CodingStructure* qgCS = tempCS;
1491
0
      bool deltaQpCodedBeforeThisNode = false;
1492
0
      if( partitioner.currArea().lumaPos() != partitioner.currQgPos )
1493
0
      {
1494
0
        int numParentNodeToQgCS = 0;
1495
0
        while( qgCS->area.lumaPos() != partitioner.currQgPos )
1496
0
        {
1497
0
          CHECK( qgCS->parent == nullptr, "parent of qgCS shall exsit" );
1498
0
          qgCS = qgCS->parent;
1499
0
          numParentNodeToQgCS++;
1500
0
        }
1501
1502
        //check whether deltaQP has been coded (in luma CU or luma&chroma CU) before this node
1503
0
        CodingStructure* parentCS = tempCS->parent;
1504
0
        for( int i = 0; i < numParentNodeToQgCS; i++ )
1505
0
        {
1506
          //checking each parent
1507
0
          CHECK( parentCS == nullptr, "parentCS shall exsit" );
1508
0
          for( const auto &cu : parentCS->cus )
1509
0
          {
1510
0
            if( cu->rootCbf && !isChroma( cu->chType ) )
1511
0
            {
1512
0
              deltaQpCodedBeforeThisNode = true;
1513
0
              break;
1514
0
            }
1515
0
          }
1516
0
          parentCS = parentCS->parent;
1517
0
        }
1518
0
      }
1519
1520
      //revise luma CU qp before the first luma CU with residual in the SCIPU to predQP
1521
0
      if( !deltaQpCodedBeforeThisNode )
1522
0
      {
1523
        //get pred QP of the QG
1524
0
        const CodingUnit* cuFirst = qgCS->getCU( CH_L, TREE_D );
1525
0
        CHECK( cuFirst->lumaPos() != partitioner.currQgPos, "First cu of the Qg is wrong" );
1526
0
        int predQp = CU::predictQP( *cuFirst, qgCS->prevQP[CH_L] );
1527
1528
        //revise to predQP
1529
0
        int firstCuHasResidual = (int)tempCS->cus.size();
1530
0
        for( int i = 0; i < tempCS->cus.size(); i++ )
1531
0
        {
1532
0
          if( tempCS->cus[i]->rootCbf )
1533
0
          {
1534
0
            firstCuHasResidual = i;
1535
0
            break;
1536
0
          }
1537
0
        }
1538
1539
0
        for( int i = 0; i < firstCuHasResidual; i++ )
1540
0
        {
1541
0
          tempCS->cus[i]->qp = predQp;
1542
0
        }
1543
0
      }
1544
0
    }
1545
0
    partitioner.chType   = CH_C;
1546
0
    partitioner.treeType = TREE_C;
1547
1548
0
    m_CurrCtx++;
1549
1550
0
    CodingStructure *tempCSChroma = m_pTempCS2;
1551
0
    CodingStructure *bestCSChroma = m_pBestCS2;
1552
1553
0
    tempCS->initSubStructure( *tempCSChroma, partitioner.chType, partitioner.currArea(), false );
1554
0
    tempCS->initSubStructure( *bestCSChroma, partitioner.chType, partitioner.currArea(), false );
1555
0
    tempCSChroma->lumaCS = tempCS;
1556
0
    bestCSChroma->lumaCS = tempCS;
1557
0
    xCompressCU( tempCSChroma, bestCSChroma, partitioner );
1558
1559
    //attach chromaCS to luma CS and update cost
1560
0
    tempCS->useSubStructure( *bestCSChroma, partitioner.chType, TREE_D, CS::getArea( *bestCSChroma, partitioner.currArea(), partitioner.chType, partitioner.treeType ), false );
1561
1562
    //release tmp resource
1563
0
    tempCSChroma->releaseIntermediateData();
1564
0
    bestCSChroma->releaseIntermediateData();
1565
1566
0
    m_CurrCtx--;
1567
    //recover luma tree status
1568
0
    partitioner.chType = CH_L;
1569
0
    partitioner.treeType = TREE_D;
1570
0
    partitioner.modeType = MODE_TYPE_ALL;
1571
0
  }
1572
1573
  // Finally, add split-signaling bits for RD-cost check
1574
67.4k
  tempCS->fracBits += splitBits; // split bits
1575
67.4k
  tempCS->cost      = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1576
67.4k
  partitioner.modeType = modeTypeParent;
1577
1578
  // Check Delta QP bits for splitted structure
1579
67.4k
  if( !qgEnableChildren ) // check at deepest QG level only
1580
64.8k
    xCheckDQP( *tempCS, partitioner, true );
1581
1582
  // If the configuration being tested exceeds the maximum number of bytes for a slice / slice-segment, then
1583
  // a proper RD evaluation cannot be performed. Therefore, termination of the
1584
  // slice/slice-segment must be made prior to this CTU.
1585
  // This can be achieved by forcing the decision to be that of the rpcTempCU.
1586
  // The exception is each slice / slice-segment must have at least one CTU.
1587
67.4k
  if( bestCS->cost == MAX_DOUBLE )
1588
48.0k
  {
1589
48.0k
    bestCS->costDbOffset = 0;
1590
48.0k
  }
1591
1592
67.4k
  if( tempCS->cus.size() > 0 && modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTER )
1593
0
  {
1594
0
    int areaSizeNoResiCu = 0;
1595
0
    for( int k = 0; k < tempCS->cus.size(); k++ )
1596
0
    {
1597
0
      areaSizeNoResiCu += (tempCS->cus[k]->rootCbf == false) ? tempCS->cus[k]->lumaSize().area() : 0;
1598
0
    }
1599
0
    if( areaSizeNoResiCu >= (tempCS->area.lumaSize().area() >> 1) )
1600
0
    {
1601
0
      skipInterPass = true;
1602
0
    }
1603
0
  }
1604
1605
  // RD check for sub partitioned coding structure.
1606
67.4k
  xCheckBestMode( tempCS, bestCS, partitioner, encTestMode, m_EDO );
1607
1608
67.4k
  if( isAffMVInfoSaved )
1609
0
  {
1610
0
    m_cInterSearch.m_AffineProfList->addAffMVInfo(tmpMVInfo);
1611
0
  }
1612
1613
67.4k
  if( !tempCS->slice->isIntra() && isUniMvInfoSaved )
1614
0
  {
1615
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->addUniMvInfo(tmpUniMvInfo);
1616
0
  }
1617
1618
67.4k
  tempCS->motionLut = oldMotionLut;
1619
67.4k
  tempCS->releaseIntermediateData();
1620
67.4k
  tempCS->prevQP[partitioner.chType] = oldPrevQp;
1621
67.4k
}
1622
1623
1624
void EncCu::xCheckRDCostIntra( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1625
82.5k
{
1626
82.5k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTRA, tempCS, partitioner.chType );
1627
1628
82.5k
  tempCS->initStructData( encTestMode.qp, false ); // clear motion buffer
1629
1630
82.5k
  CodingUnit &cu      = tempCS->addCU( CS::getArea( *tempCS, tempCS->area, partitioner.chType, partitioner.treeType ), partitioner.chType );
1631
1632
82.5k
  partitioner.setCUData( cu );
1633
82.5k
  cu.slice            = tempCS->slice;
1634
82.5k
  cu.tileIdx          = m_tileIdx;
1635
82.5k
  cu.skip             = false;
1636
82.5k
  cu.mmvdSkip         = false;
1637
82.5k
  cu.predMode         = MODE_INTRA;
1638
82.5k
  cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
1639
82.5k
  cu.qp               = encTestMode.qp;
1640
82.5k
  cu.ispMode          = NOT_INTRA_SUBPARTITIONS;
1641
82.5k
  cu.initPuData();
1642
1643
82.5k
  m_cIntraSearch.m_ispTestedModes[0].init(0, 0, 1);
1644
82.5k
  if (m_pcEncCfg->m_FastIntraTools)
1645
82.5k
  {
1646
82.5k
    m_modeCtrl.comprCUCtx->intraWasTested = false;
1647
82.5k
    m_cIntraSearch.m_ispTestedModes[0].relatedCuIsValid = m_modeCtrl.comprCUCtx->relatedCuIsValid;
1648
82.5k
    if (!bestCS->cus.empty())
1649
22.5k
    {
1650
22.5k
      if ((bestCS->cus[0]->mergeFlag || bestCS->cus[0]->imv || bestCS->cus[0]->affine) && (!bestCS->cus[0]->ciip))
1651
22.5k
      {
1652
22.5k
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[0] = -1;
1653
22.5k
      }
1654
22.5k
    }
1655
82.5k
    if (!bestCS->slice->isIntra())
1656
0
    {
1657
0
      const Position posBL = cu.Y().bottomLeft();
1658
0
      const Position posTR = cu.Y().topRight();
1659
0
      for (int i = 0; i < 2; i++)
1660
0
      {
1661
0
        const CodingUnit* neigh = i ? cu.cs->getCURestricted(posTR.offset(0, -1), cu, CH_L) :cu.cs->getCURestricted(posBL.offset(-1, 0), cu, CH_L);
1662
0
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = -1;
1663
0
        if (neigh != nullptr)
1664
0
        {
1665
0
          int bestMode = neigh->firstTU->mtsIdx[0] ? 4 : 0;
1666
0
          bestMode |= neigh->lfnstIdx ? 2 : 0;
1667
0
          bestMode |= neigh->ispMode ? 1 : 0;
1668
0
          m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = bestMode;
1669
0
        }
1670
0
      }
1671
0
    }
1672
82.5k
  }
1673
1674
82.5k
  tempCS->interHad    = m_modeCtrl.comprCUCtx->interHad;
1675
82.5k
  double maxCostAllowedForChroma = MAX_DOUBLE;
1676
82.5k
  if( isLuma( partitioner.chType ) )
1677
25.2k
  {
1678
25.2k
    if (!tempCS->slice->isIntra() && bestCS)
1679
0
    {
1680
0
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner, bestCS->cost);
1681
0
    }
1682
25.2k
    else
1683
25.2k
    {
1684
25.2k
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner);
1685
25.2k
    }
1686
25.2k
    if (m_pcEncCfg->m_FastIntraTools)
1687
25.2k
    {
1688
25.2k
      if (m_cIntraSearch.m_ispTestedModes[0].intraWasTested)
1689
25.2k
      {
1690
25.2k
        m_modeCtrl.comprCUCtx->intraWasTested = m_cIntraSearch.m_ispTestedModes[0].intraWasTested;
1691
25.2k
      }
1692
25.2k
    }
1693
1694
25.2k
    if( !partitioner.isSepTree( *tempCS ) )
1695
0
    {
1696
0
      tempCS->lumaCost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1697
0
    }
1698
25.2k
    if (m_pcEncCfg->m_usePbIntraFast && tempCS->dist == MAX_DISTORTION && tempCS->interHad == 0)
1699
0
    {
1700
      // JEM assumes only perfect reconstructions can from now on beat the inter mode
1701
0
      m_modeCtrl.comprCUCtx->interHad = 0;
1702
0
      return;
1703
0
    }
1704
25.2k
  }
1705
1706
82.5k
  if( tempCS->area.chromaFormat != CHROMA_400 && ( partitioner.chType == CH_C || !CU::isSepTree(cu) ) )
1707
57.2k
  {
1708
57.2k
    bool useIntraSubPartitions = cu.ispMode != NOT_INTRA_SUBPARTITIONS;
1709
57.2k
    Partitioner subTuPartitioner = partitioner;
1710
57.2k
    if ((m_pcEncCfg->m_ISP >= 3) && (!partitioner.isSepTree(*tempCS) && useIntraSubPartitions))
1711
0
    {
1712
0
      maxCostAllowedForChroma = bestCS->cost < MAX_DOUBLE ? bestCS->cost - tempCS->lumaCost : MAX_DOUBLE;
1713
0
    }
1714
57.2k
    m_cIntraSearch.estIntraPredChromaQT(
1715
57.2k
      cu, (!useIntraSubPartitions || (CU::isSepTree(cu) && !isLuma(CH_C))) ? partitioner : subTuPartitioner,
1716
57.2k
      maxCostAllowedForChroma);
1717
57.2k
    if ((m_pcEncCfg->m_ISP >= 3) && useIntraSubPartitions && !cu.ispMode)
1718
0
    {
1719
0
      return;
1720
0
    }
1721
57.2k
  }
1722
1723
82.5k
  cu.rootCbf = false;
1724
1725
330k
  for (uint32_t t = 0; t < getNumberValidTBlocks(*cu.cs->pcv); t++)
1726
247k
  {
1727
247k
    cu.rootCbf |= cu.firstTU->cbf[t] != 0;
1728
247k
  }
1729
1730
  // Get total bits for current mode: encode CU
1731
82.5k
  m_CABACEstimator->resetBits();
1732
1733
82.5k
  if ((!cu.cs->slice->isIntra() || cu.cs->slice->sps->IBC) && cu.Y().valid())
1734
25.2k
  {
1735
25.2k
    m_CABACEstimator->cu_skip_flag(cu);
1736
25.2k
  }
1737
82.5k
  m_CABACEstimator->pred_mode(cu);
1738
82.5k
  m_CABACEstimator->cu_pred_data(cu);
1739
1740
  // Encode Coefficients
1741
82.5k
  CUCtx cuCtx;
1742
82.5k
  cuCtx.isDQPCoded = true;
1743
82.5k
  cuCtx.isChromaQpAdjCoded = true;
1744
82.5k
  m_CABACEstimator->cu_residual(cu, partitioner, cuCtx);
1745
1746
82.5k
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
1747
82.5k
  tempCS->cost = m_cRdCost.calcRdCost(tempCS->fracBits, tempCS->dist);
1748
1749
82.5k
  xEncodeDontSplit(*tempCS, partitioner);
1750
1751
82.5k
  xCheckDQP(*tempCS, partitioner);
1752
1753
82.5k
  if( m_EDO )
1754
82.5k
  {
1755
82.5k
    xCalDebCost(*tempCS, partitioner);
1756
82.5k
  }
1757
1758
82.5k
  DTRACE_MODE_COST(*tempCS, m_cRdCost.getLambda(true));
1759
82.5k
  xCheckBestMode(tempCS, bestCS, partitioner, encTestMode, m_EDO);
1760
1761
82.5k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
1762
82.5k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
1763
82.5k
}
1764
1765
void EncCu::xCheckDQP( CodingStructure& cs, Partitioner& partitioner, bool bKeepCtx )
1766
169k
{
1767
169k
  if( !cs.pps->useDQP )
1768
0
  {
1769
0
    return;
1770
0
  }
1771
1772
169k
  if (partitioner.isSepTree(cs) && isChroma(partitioner.chType))
1773
96.2k
  {
1774
96.2k
    return;
1775
96.2k
  }
1776
1777
73.6k
  if( !partitioner.currQgEnable() ) // do not consider split or leaf/not leaf QG condition (checked by caller)
1778
66.4k
  {
1779
66.4k
    return;
1780
66.4k
  }
1781
1782
7.17k
  CodingUnit* cuFirst = cs.getCU( partitioner.chType, partitioner.treeType );
1783
1784
7.17k
  CHECK( bKeepCtx && cs.cus.size() <= 1 && partitioner.getImplicitSplit( cs ) == CU_DONT_SPLIT, "bKeepCtx should only be set in split case" );
1785
7.17k
  CHECK( !bKeepCtx && cs.cus.size() > 1, "bKeepCtx should never be set for non-split case" );
1786
7.17k
  CHECK( !cuFirst, "No CU available" );
1787
1788
7.17k
  bool hasResidual = false;
1789
7.17k
  for( const auto &cu : cs.cus )
1790
20.1k
  {
1791
    //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1792
20.1k
    if( cu->rootCbf && !isChroma( cu->chType ))
1793
1.21k
    {
1794
1.21k
      hasResidual = true;
1795
1.21k
      break;
1796
1.21k
    }
1797
20.1k
  }
1798
1799
7.17k
  int predQP = CU::predictQP( *cuFirst, cs.prevQP[partitioner.chType] );
1800
1801
7.17k
  if( hasResidual )
1802
1.21k
  {
1803
1.21k
    TempCtx ctxTemp( m_CtxCache );
1804
1.21k
    if( !bKeepCtx ) ctxTemp = SubCtx( Ctx::DeltaQP, m_CABACEstimator->getCtx() );
1805
1806
1.21k
    m_CABACEstimator->resetBits();
1807
1.21k
    m_CABACEstimator->cu_qp_delta( *cuFirst, predQP, cuFirst->qp );
1808
1809
1.21k
    cs.fracBits += m_CABACEstimator->getEstFracBits(); // dQP bits
1810
1.21k
    cs.cost      = m_cRdCost.calcRdCost(cs.fracBits, cs.dist);
1811
1812
1813
1.21k
    if( !bKeepCtx ) m_CABACEstimator->getCtx() = SubCtx( Ctx::DeltaQP, ctxTemp );
1814
1815
    // NOTE: reset QPs for CUs without residuals up to first coded CU
1816
1.21k
    for( const auto &cu : cs.cus )
1817
1.21k
    {
1818
      //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1819
1.21k
      if( cu->rootCbf && !isChroma( cu->chType ))
1820
1.21k
      {
1821
1.21k
        break;
1822
1.21k
      }
1823
7
      cu->qp = predQP;
1824
7
    }
1825
1.21k
  }
1826
5.96k
  else
1827
5.96k
  {
1828
    // No residuals: reset CU QP to predicted value
1829
5.96k
    for( const auto &cu : cs.cus )
1830
18.9k
    {
1831
18.9k
      cu->qp = predQP;
1832
18.9k
    }
1833
5.96k
  }
1834
7.17k
}
1835
1836
CodingUnit *EncCu::getCuForInterPrediction( CodingStructure *cs, const EncTestMode& encTestMode )
1837
0
{
1838
0
  CodingUnit *cu = cs->getCU( CH_L, TREE_D );
1839
1840
0
  if( cu == nullptr )
1841
0
  {
1842
0
    CHECK( cs->getCU( CH_L, TREE_D ) != nullptr, "Wrong CU/PU setting in CS" );
1843
0
    cu = &cs->addCU( cs->area, CH_L );
1844
0
  }
1845
1846
0
  cu->slice       = cs->slice;
1847
0
  cu->tileIdx     = m_tileIdx;
1848
0
  cu->skip        = false;
1849
0
  cu->mmvdSkip    = false;
1850
0
  cu->mmvdMergeFlag
1851
0
                  = false;
1852
0
  cu->geo         = false;
1853
0
  cu->predMode    = MODE_INTER;
1854
0
  cu->chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
1855
0
  cu->qp          = encTestMode.qp;
1856
0
  cu->affine      = false;
1857
0
  cu->multiRefIdx = 0;
1858
0
  cu->mipFlag     = false;
1859
0
  cu->ciip        = false;
1860
1861
0
  return cu;
1862
0
}
1863
1864
int getDmvrMvdNum( const CodingUnit &cu )
1865
0
{
1866
0
  const int dx = std::max<int>( cu.lwidth()  >> DMVR_SUBCU_SIZE_LOG2, 1 );
1867
0
  const int dy = std::max<int>( cu.lheight() >> DMVR_SUBCU_SIZE_LOG2, 1 );
1868
0
  return dx * dy;
1869
0
}
1870
1871
void EncCu::xCheckRDCostUnifiedMerge( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, EncTestMode &encTestMode )
1872
0
{
1873
0
  const Slice &slice = *tempCS->slice;
1874
1875
0
  CHECK( slice.sliceType == VVENC_I_SLICE, "Merge modes not available for I-slices" );
1876
1877
0
  tempCS->initStructData( encTestMode.qp );
1878
1879
0
  MergeCtx          mergeCtx, gpmMergeCtx;
1880
0
  AffineMergeCtx    affineMergeCtx;
1881
0
  GeoComboCostList &comboList = m_comboList;
1882
0
  const SPS        &sps       = *tempCS->sps;
1883
1884
0
  if( sps.SbtMvp )
1885
0
  {
1886
0
    const Size bufSize           = g_miScaling.scale( tempCS->area.lumaSize() );
1887
0
    affineMergeCtx.subPuMvpMiBuf = MotionBuf        ( m_subPuMiBuf, bufSize );
1888
0
  }
1889
1890
0
  m_mergeBestSATDCost = MAX_DOUBLE;
1891
1892
0
  CodingUnit *cu = getCuForInterPrediction( tempCS, encTestMode );
1893
0
  partitioner.setCUData            ( *cu );
1894
0
  CU::getInterMergeCandidates      ( *cu, mergeCtx, 0 );
1895
0
  if( sps.MMVD )
1896
0
    CU::getInterMMVDMergeCandidates( *cu, mergeCtx );
1897
1898
0
  bool sameMV[MRG_MAX_NUM_CANDS] = { false, };
1899
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
1900
0
  {
1901
0
    for( int m = 0; m < mergeCtx.numValidMergeCand - 1; m++ )
1902
0
    {
1903
0
      if( !sameMV[m] )
1904
0
      {
1905
0
        for( int n = m + 1; n < mergeCtx.numValidMergeCand; n++ )
1906
0
        {
1907
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
1908
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
1909
0
        }
1910
0
      }
1911
0
    }
1912
0
  }
1913
1914
0
  MergeBufVector mrgPredBufNoCiip;
1915
0
  MergeBufVector geoBuffer;
1916
0
  const double  sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda() * FRAC_BITS_SCALE;
1917
1918
0
  const UnitArea localUnitArea( cu->chromaFormat, Area( 0, 0, cu->Y().width, cu->Y().height ) );
1919
0
  for( int i = 0; i < mergeCtx.numValidMergeCand; i++ )
1920
0
  {
1921
0
    mrgPredBufNoCiip.push_back( m_acMergeTmpBuffer[i].getCompactBuf( localUnitArea ) );
1922
0
  }
1923
1924
0
  int numMergeSatdCand = std::min( bestCS->area.lumaSize().area() >= 64 ? m_pcEncCfg->m_mergeRdCandQuotaRegular : m_pcEncCfg->m_mergeRdCandQuotaRegularSmallBlk, mergeCtx.numValidMergeCand );
1925
1926
0
  bool isCiipEnabled  = sps.CIIP && bestCS->area.lumaSize().area() >= 64 && bestCS->area.lumaSize().maxDim() < MAX_CU_SIZE;
1927
0
       isCiipEnabled &= m_pcEncCfg->m_CIIP <= 1 || !m_modeCtrl.getBlkInfo( tempCS->area ).isSkip; //5
1928
1929
0
  if( isCiipEnabled )
1930
0
  {
1931
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaCiip, mergeCtx.numValidMergeCand );
1932
0
  }
1933
1934
0
  const bool affineMrgAvail = ( m_pcEncCfg->m_Affine <= 2 || slice.TLayer <= 3 || m_pcEncCfg->m_SbTMVP )
1935
0
                           && ( m_pcEncCfg->m_Affine || sps.SbtMvp ) && m_pcEncCfg->m_maxNumAffineMergeCand && bestCS->area.Y().minDim() >= 8;
1936
1937
0
  if( affineMrgAvail )
1938
0
  {
1939
0
    CU::getAffineMergeCand( *cu, affineMergeCtx );
1940
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaSubBlk, affineMergeCtx.numValidMergeCand );
1941
0
  }
1942
1943
0
  int numSatdCandPreGeo = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1944
0
  bool toAddGpmCand     = false;
1945
0
  if( sps.GEO && slice.isInterB() // base checks
1946
0
      && cu->lumaSize().minDim() >= GEO_MIN_CU_SIZE  && cu->lumaSize().maxDim() <= GEO_MAX_CU_SIZE && cu->lumaSize().maxDim() < 8 * cu->lumaSize().minDim() // size checks
1947
0
      && !( m_pcEncCfg->m_Geo > 2 && slice.TLayer <= 1 ) ) // speedups
1948
0
  {
1949
0
    cu->mergeFlag            = true;
1950
0
    cu->geo                  = true;
1951
0
    CU::getGeoMergeCandidates( *cu, gpmMergeCtx );
1952
0
    toAddGpmCand             = prepareGpmComboList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, comboList, geoBuffer, *cu );
1953
0
    numMergeSatdCand        += toAddGpmCand ? std::min( m_pcEncCfg->m_mergeRdCandQuotaGpm, ( int ) comboList.list.size() ) : 0;
1954
0
  }
1955
1956
0
  numMergeSatdCand  = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1957
1958
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
1959
0
  m_mergeItemList.resetList( numMergeSatdCand );
1960
0
  const TempCtx ctxStart   ( m_CtxCache, m_CABACEstimator->getCtx() );
1961
0
  const DFunc   dfunc      = encTestMode.lossless ? DF_SAD : ( m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
1962
0
  DistParam     distParam  = m_cRdCost.setDistParam( tempCS->getOrgBuf().Y(), tempCS->getOrgBuf().Y(), sps.bitDepths[CH_L], dfunc );
1963
0
  m_uiSadBestForQPA        = MAX_DISTORTION;
1964
1965
0
  addRegularCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV, mrgPredBufNoCiip );
1966
1967
  // add CIIP candidates directly after adding regular cands
1968
0
  if( isCiipEnabled )
1969
0
  {
1970
0
    addCiipCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV );
1971
0
  }
1972
1973
0
  if( sps.MMVD && !!m_mergeItemList.size() && !( m_pcEncCfg->m_useFastMrg >= 2 && m_mergeItemList.size() <= 1 ) )
1974
0
  {
1975
0
    addMmvdCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1976
0
  }
1977
1978
0
  if( affineMergeCtx.numValidMergeCand > 0 )
1979
0
  {
1980
0
    addAffineCandsToPruningList( affineMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1981
0
  }
1982
1983
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
1984
0
  {
1985
0
    m_mergeBestSATDCost    = m_mergeItemList.getMergeItemInList( 0 )->cost;
1986
0
    const double threshold = m_mergeBestSATDCost * MRG_FAST_RATIO[tempCS->picture->useFastMrg];
1987
0
    const   int shrinkSize = std::min( numSatdCandPreGeo, ( int ) updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand ) );
1988
0
    m_mergeItemList        . shrinkList( shrinkSize );
1989
0
  }
1990
0
  else
1991
0
  {
1992
0
    m_mergeItemList        . shrinkList( numSatdCandPreGeo );
1993
0
  }
1994
1995
0
  if( toAddGpmCand )
1996
0
  {
1997
0
    addGpmCandsToPruningList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, comboList, geoBuffer, distParam, *cu );
1998
0
  }
1999
2000
0
  if(    m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2 && m_uiSadBestForQPA < MAX_DISTORTION && slice.TLayer == 0 // non-Intra key-frame
2001
0
      && m_pcEncCfg->m_salienceBasedOpt
2002
0
      && m_pcEncCfg->m_usePerceptQPA && partitioner.currQgEnable() && partitioner.currSubdiv == 0 ) // CTU-level luma quantization group
2003
0
  {
2004
0
    CHECK( bestCS->cost < MAX_DOUBLE, "This has to be the first test performed!" );
2005
2006
0
    const Picture *pic         = slice.pic;
2007
0
    const bool     isBIM       = m_pcEncCfg->m_RCNumPasses != 2 && m_pcEncCfg->m_blockImportanceMapping && !pic->m_picShared->m_ctuBimQpOffset.empty();
2008
0
    const uint32_t rsAddr      = getCtuAddr( partitioner.currQgPos, *pic->cs->pcv );
2009
0
    const int      pumpReducQP = BitAllocation::getCtuPumpingReducingQP( &slice, tempCS->getOrgBuf( COMP_Y ), m_uiSadBestForQPA, *m_globalCtuQpVector, rsAddr,
2010
0
                                                                         m_pcEncCfg->m_QP, isBIM );
2011
2012
0
    if( pumpReducQP != 0 ) // subtract QP offset, reduces Intra-period pumping or overcoding
2013
0
    {
2014
0
      encTestMode.qp = Clip3( 0, MAX_QP, encTestMode.qp - pumpReducQP );
2015
0
      tempCS->currQP[partitioner.chType] = tempCS->baseQP =
2016
0
      bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3( 0, MAX_QP, tempCS->baseQP - pumpReducQP );
2017
2018
0
      updateLambda( slice, pic->ctuQpaLambda[rsAddr], pic->ctuAdaptedQP[rsAddr], tempCS->baseQP, true );
2019
0
    }
2020
0
  }
2021
2022
  // Try to limit number of candidates using SATD-costs
2023
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
2024
0
  {
2025
    // shrink GEO list as well
2026
0
    const double threshold = m_mergeItemList.getMergeItemInList( 0 )->cost * MRG_FAST_RATIO[0];
2027
0
    numMergeSatdCand       = updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand );
2028
0
    m_mergeBestSATDCost    = m_mergeItemList.size() != 0 ? m_mergeItemList.getMergeItemInList( 0 )->cost : MAX_DOUBLE;
2029
0
  }
2030
0
  else
2031
0
  {
2032
0
    numMergeSatdCand       = std::min<int>( numMergeSatdCand, ( int ) m_mergeItemList.size() );
2033
0
  }
2034
2035
  // 2. Pass: RD checking 
2036
0
  tempCS->initStructData( encTestMode.qp );
2037
0
  m_CABACEstimator->getCtx() = ctxStart;
2038
2039
0
  double bestEndCost                            =   MAX_DOUBLE;
2040
0
  bool bestIsSkip                               =   false;
2041
0
  PelUnitBuf ciipBuf                            =   m_aTmpStorageLCU[1].getCompactBuf( *cu );
2042
0
  bool ciipChromaDone                           =   false;
2043
0
  bool isRegularTestedAsSkip[MRG_MAX_NUM_CANDS] = { false, };
2044
0
  bool geoWasTested                             =   false;
2045
0
  int  stopCand                                 =   numMergeSatdCand;
2046
2047
0
  CHECK( numMergeSatdCand > 0 && m_mergeItemList.size() == 0, "Empty merge item list is not expected" );
2048
2049
0
  for( uint32_t noResidualPass = 0; noResidualPass < 2; noResidualPass++ )
2050
0
  {
2051
0
    const bool forceNoResidual = noResidualPass == 1;
2052
0
    for( uint32_t mrgHadIdx = 0; mrgHadIdx < stopCand; mrgHadIdx++ )
2053
0
    {
2054
0
      auto mergeItem = m_mergeItemList.getMergeItemInList( mrgHadIdx );
2055
0
      CHECK( mergeItem == nullptr, "Wrong merge item" );
2056
2057
0
      const bool isCiip = mergeItem->mergeItemType == MergeItem::MergeItemType::CIIP;
2058
0
      const bool isGeo  = mergeItem->mergeItemType == MergeItem::MergeItemType::GPM;
2059
0
      const bool isRglr = mergeItem->mergeItemType == MergeItem::MergeItemType::REGULAR;
2060
0
      const bool isMmvd = mergeItem->mergeItemType == MergeItem::MergeItemType::MMVD;
2061
2062
0
      if( noResidualPass != 0 && isCiip && isRegularTestedAsSkip[mergeItem->mergeIdx] )
2063
0
      {
2064
0
        continue;
2065
0
      }
2066
2067
0
      if( noResidualPass ? mergeItem->noResidual : bestIsSkip )
2068
0
      {
2069
0
        continue;
2070
0
      }
2071
2072
0
      if( isGeo )
2073
0
      {
2074
0
        if( m_pcEncCfg->m_Geo > 2 && geoWasTested && !bestCS->cus.empty() && !bestCS->getCU( partitioner.chType, partitioner.treeType )->geo )
2075
0
        {
2076
0
          continue;
2077
0
        }
2078
2079
0
        geoWasTested = true;
2080
0
      }
2081
2082
0
      cu = getCuForInterPrediction( tempCS, encTestMode );
2083
0
      partitioner.setCUData( *cu );
2084
0
      const bool resetCiip2Regular = mergeItem->exportMergeInfo( *cu, forceNoResidual );
2085
2086
0
      if( isRglr || resetCiip2Regular )
2087
0
      {
2088
0
        if( CU::checkDMVRCondition( *cu ) ) std::copy_n( m_subPuMvOffset[mergeItem->mergeIdx].data(), getDmvrMvdNum( *cu ), cu->mvdL0SubPu );
2089
0
      }
2090
2091
0
      if( isMmvd && mergeItem->noBdofRefine )
2092
0
      {
2093
        // no BDOF refinement was made for the luma prediction, need to have luma prediction again
2094
0
        mergeItem->lumaPredReady = false;
2095
0
      }
2096
2097
0
      PelUnitBuf *predBuf1   = nullptr, *predBuf2 = isCiip ? &ciipBuf : nullptr;
2098
0
      PelUnitBuf  dstPredBuf = tempCS->getPredBuf( *cu );
2099
2100
0
      if( isGeo )
2101
0
      {
2102
0
        predBuf1 = &geoBuffer[cu->geoMergeIdx[0]];
2103
0
        predBuf2 = &geoBuffer[cu->geoMergeIdx[1]];
2104
0
      }
2105
2106
0
      if( resetCiip2Regular )
2107
0
      {
2108
0
        dstPredBuf.copyFrom( mrgPredBufNoCiip[mergeItem->mergeIdx] );
2109
0
      }
2110
0
      else
2111
0
      {
2112
0
        if( isCiip && !resetCiip2Regular && isChromaEnabled( cu->chromaFormat ) && cu->chromaSize().width > 2 )
2113
0
        {
2114
0
          if( !ciipChromaDone )
2115
0
          {
2116
0
            cu->intraDir[0] = PLANAR_IDX;
2117
0
            cu->intraDir[1] = DM_CHROMA_IDX;
2118
2119
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cb() );
2120
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cb, ciipBuf.Cb(), *cu );
2121
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cr() );
2122
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cr, ciipBuf.Cr(), *cu );
2123
2124
0
            ciipChromaDone  = true;
2125
0
          }
2126
0
        }
2127
2128
0
        if(  mergeItem->lumaPredReady ||  mergeItem->chromaPredReady )
2129
0
          dstPredBuf.copyFrom( mergeItem->getPredBuf( localUnitArea ), mergeItem->lumaPredReady, mergeItem->chromaPredReady );
2130
0
        if( !mergeItem->lumaPredReady || !mergeItem->chromaPredReady )
2131
0
          generateMergePrediction( localUnitArea, mergeItem, *cu, !mergeItem->lumaPredReady, !mergeItem->chromaPredReady, dstPredBuf, true, forceNoResidual, predBuf1, predBuf2 );
2132
0
      }
2133
2134
0
      if( !cu->mmvdSkip && !cu->ciip && !cu->affine && !cu->geo && noResidualPass != 0 )
2135
0
      {
2136
0
        CHECK( mergeItem->mergeIdx >= mergeCtx.numValidMergeCand, "out of normal merge" );
2137
0
        isRegularTestedAsSkip[mergeItem->mergeIdx] = true;
2138
0
      }
2139
2140
0
      xEncodeInterResidual( tempCS, bestCS, partitioner, encTestMode, noResidualPass, noResidualPass == 0 ? &mergeItem->noResidual : nullptr );
2141
2142
0
      if( m_pcEncCfg->m_useFastMrg >= 2 )
2143
0
      {
2144
0
        if( cu->ciip && bestCS->cost == MAX_DOUBLE && mrgHadIdx + 1 == numMergeSatdCand )
2145
0
        {
2146
0
          numMergeSatdCand = ( unsigned ) m_mergeItemList.size();
2147
0
        }
2148
      
2149
0
        if( mrgHadIdx > 0 && tempCS->cost >= bestEndCost && !cu->ciip && !isGeo )
2150
0
        {
2151
0
          stopCand = mrgHadIdx + 1;
2152
0
        }
2153
      
2154
0
        if( noResidualPass == 0 )
2155
0
        {
2156
0
          bestEndCost = std::min( bestEndCost, tempCS->cost );
2157
0
        }
2158
0
      }
2159
2160
0
      if( m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip && !cu->ciip )
2161
0
      {
2162
0
        bestIsSkip = !bestCS->cus.empty() && bestCS->getCU( partitioner.chType, partitioner.treeType )->rootCbf == 0;
2163
0
      }
2164
2165
0
      tempCS->initStructData( encTestMode.qp );
2166
0
    }   // end loop mrgHadIdx
2167
0
  }
2168
0
}
2169
2170
unsigned int EncCu::updateRdCheckingNum( MergeItemList &mergeItemList, double threshold, unsigned int numMergeSatdCand )
2171
0
{
2172
0
  for( uint32_t i = 0; i < mergeItemList.size(); i++ )
2173
0
  {
2174
0
    const auto mergeItem = mergeItemList.getMergeItemInList( i );
2175
0
    if( mergeItem == nullptr || mergeItem->cost > threshold )
2176
0
    {
2177
0
      numMergeSatdCand = i;
2178
0
      break;
2179
0
    }
2180
0
  }
2181
0
  return std::min( numMergeSatdCand, ( unsigned ) mergeItemList.size() );
2182
0
}
2183
2184
void EncCu::generateMergePrediction( const UnitArea &unitArea, MergeItem *mergeItem, CodingUnit &pu, bool luma, bool chroma,
2185
                                     PelUnitBuf &dstBuf, bool finalRd, bool forceNoResidual, PelUnitBuf *predBuf1, PelUnitBuf *predBuf2 )
2186
0
{
2187
0
  CHECK( ( luma && mergeItem->lumaPredReady ) || ( chroma && mergeItem->chromaPredReady ), "Prediction has been avaiable" );
2188
2189
0
  pu.mcControl = ( !luma ? 4 : 0 ) | ( !chroma ? 2 : 0 );
2190
2191
0
  switch( mergeItem->mergeItemType )
2192
0
  {
2193
0
  case MergeItem::MergeItemType::REGULAR:
2194
    // here predBuf1 is predBufNoCiip
2195
0
    pu.mvRefine = true;
2196
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2197
0
    pu.mvRefine = false;
2198
0
    if( predBuf1 != nullptr )
2199
0
    {
2200
0
      predBuf1->copyFrom( dstBuf, luma, chroma );
2201
0
    }
2202
0
    break;
2203
2204
0
  case MergeItem::MergeItemType::CIIP:
2205
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2206
2207
0
    if( luma )
2208
0
    {
2209
      // generate intrainter Y prediction
2210
0
      dstBuf.Y().weightCiip( predBuf2->Y(), mergeItem->numCiipIntra );
2211
0
    }
2212
2213
0
    if( chroma )
2214
0
    {
2215
0
      if( pu.chromaSize().width > 2 )
2216
0
      {
2217
0
        dstBuf.Cb().weightCiip( predBuf2->Cb(), mergeItem->numCiipIntra );
2218
0
        dstBuf.Cr().weightCiip( predBuf2->Cr(), mergeItem->numCiipIntra );
2219
0
      }
2220
0
    }
2221
2222
0
    break;
2223
2224
0
  case MergeItem::MergeItemType::MMVD:
2225
0
    pu.mcControl           |= finalRd ? 0 : ( pu.mmvdMergeIdx.pos.step > 2 || m_pcEncCfg->m_MMVD > 1 ) ? 1 : 0;
2226
0
    mergeItem->noBdofRefine = pu.mccNoBdof() && pu.cs->sps->BDOF && !pu.cs->picHeader->disBdofFlag;
2227
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2228
0
    break;
2229
2230
0
  case MergeItem::MergeItemType::SBTMVP:
2231
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2232
0
    break;
2233
2234
0
  case MergeItem::MergeItemType::AFFINE:
2235
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2236
0
    break;
2237
2238
0
  case MergeItem::MergeItemType::GPM:
2239
    // here predBuf1 and predBuf2 point to geoBuffer[mergeCand0] and geoBuffer[mergeCand1], respectively
2240
0
    CHECK( predBuf1 == nullptr || predBuf2 == nullptr, "Invalid input buffer to GPM" );
2241
0
    m_cInterSearch.weightedGeoBlk( pu.slice->clpRngs, pu, pu.geoSplitDir, luma && chroma ? MAX_NUM_CH : luma ? CH_L : CH_C, dstBuf, *predBuf1, *predBuf2 );
2242
0
    break;
2243
2244
0
  default:
2245
0
    THROW("Wrong merge item type");
2246
0
  }
2247
2248
0
  auto mergeItemPredBuf = mergeItem->getPredBuf( unitArea );
2249
2250
0
  if( dstBuf.Y().buf == mergeItemPredBuf.Y().buf )
2251
0
  {
2252
    // dst is the internal buffer
2253
0
    mergeItem->lumaPredReady   |= luma;
2254
0
    mergeItem->chromaPredReady |= chroma;
2255
0
  }
2256
0
  else if( finalRd && !forceNoResidual )
2257
0
  {
2258
    // at final RD stage, with and without residuals are both checked
2259
    // it makes sense to buffer the prediction
2260
0
    mergeItemPredBuf.copyFrom( dstBuf, luma, chroma );
2261
0
    mergeItem->lumaPredReady   |= luma;
2262
0
    mergeItem->chromaPredReady |= chroma;
2263
0
  }
2264
0
}
2265
2266
void EncCu::addRegularCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart,
2267
                                          DistParam& distParam, CodingUnit& pu, bool* sameMv, MergeBufVector& regularPred )
2268
0
{
2269
0
  pu.geo = pu.affine
2270
0
         = pu.mmvdMergeFlag = pu.mmvdSkip
2271
0
         = pu.ciip
2272
0
         = false;
2273
2274
0
  for( uint32_t uiMergeCand = 0; uiMergeCand < mergeCtx.numValidMergeCand; uiMergeCand++ )
2275
0
  {
2276
0
    if( sameMv[uiMergeCand] ) continue;
2277
2278
0
    mergeCtx.setMergeInfo   ( pu, uiMergeCand );
2279
2280
0
    if( m_pcEncCfg->m_ifpLines && // what about DMVR?
2281
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2282
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2283
0
    {
2284
0
      continue;
2285
0
    }
2286
2287
0
    pu.interDir             = mergeCtx.interDirNeighbours[uiMergeCand];
2288
0
    pu.BcwIdx               = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2289
0
    pu.imv                  = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2290
0
    CU::spanMotionInfo      ( pu );
2291
2292
0
    MergeItem *regularMerge = m_mergeItemList.allocateNewMergeItem();
2293
0
    regularMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::REGULAR, pu );
2294
0
    auto dstBuf             = regularMerge->getPredBuf( localUnitArea );
2295
0
    generateMergePrediction ( localUnitArea, regularMerge, pu, true, true, dstBuf, false, false, &regularPred[uiMergeCand], nullptr );
2296
0
    regularMerge->cost      = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2297
0
    if( CU::checkDMVRCondition( pu ) ) std::copy_n( pu.mvdL0SubPu, getDmvrMvdNum( pu ), m_subPuMvOffset[uiMergeCand].data() );
2298
0
    m_mergeItemList         . insertMergeItemToList( regularMerge );
2299
0
  }
2300
0
}
2301
2302
void EncCu::addCiipCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart, DistParam &distParam, CodingUnit &pu, bool* sameMv )
2303
0
{
2304
0
  int                numCiipIntra = -1;
2305
0
  PelUnitBuf         ciipBuf      = m_aTmpStorageLCU[1].getCompactBuf( pu );
2306
2307
0
  pu.ciip        = true;
2308
0
  pu.intraDir[0] = PLANAR_IDX;
2309
0
  pu.geo         = pu.affine
2310
0
                 = pu.mmvdMergeFlag = pu.mmvdSkip
2311
0
                 = false;
2312
0
  m_cIntraSearch . initIntraPatternChType        ( pu, pu.Y() );
2313
0
  m_cIntraSearch . predIntraAng                  ( COMP_Y, ciipBuf.Y(), pu );
2314
0
  numCiipIntra   = m_cIntraSearch.getNumIntraCiip( pu );
2315
2316
0
  int nonCiipMrgCnds[MRG_MAX_NUM_CANDS] = { 0, };
2317
0
  int numNonCiipCnds                    =   0;
2318
0
  for( ; numNonCiipCnds < m_mergeItemList.size(); numNonCiipCnds++ ) nonCiipMrgCnds[numNonCiipCnds] = m_mergeItemList.getMergeItemInList( numNonCiipCnds )->mergeIdx;
2319
2320
0
  for( int i = 0; i < numNonCiipCnds; i++ )
2321
0
  {
2322
0
    const unsigned int uiMergeCand = nonCiipMrgCnds[i];
2323
2324
0
    if( sameMv[uiMergeCand] ) continue;
2325
2326
0
    mergeCtx.setMergeInfo     ( pu, uiMergeCand );
2327
2328
0
    if( m_pcEncCfg->m_ifpLines && 
2329
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2330
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2331
0
    {
2332
0
      continue;
2333
0
    }
2334
2335
0
    pu.interDir               = mergeCtx.interDirNeighbours[uiMergeCand];
2336
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2337
0
    pu.imv                    = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2338
0
    CU::spanMotionInfo        ( pu );
2339
2340
0
    MergeItem* ciipMerge      = m_mergeItemList.allocateNewMergeItem();
2341
0
    ciipMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::CIIP, pu );
2342
0
    ciipMerge->numCiipIntra   = numCiipIntra;
2343
0
    auto dstBuf               = ciipMerge->getPredBuf( localUnitArea );
2344
0
    generateMergePrediction   ( localUnitArea, ciipMerge, pu, true, false, dstBuf, false, false, nullptr, &ciipBuf );
2345
0
      ciipMerge->cost         = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2346
0
    if( !m_mergeItemList      . insertMergeItemToList( ciipMerge ) && m_pcEncCfg->m_CIIP > 1 )
2347
0
    {
2348
0
      break;
2349
0
    }
2350
0
  }
2351
0
}
2352
2353
void EncCu::addMmvdCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx& ctxStart,
2354
                                       DistParam& distParam, CodingUnit& pu )
2355
0
{
2356
0
  pu.mmvdSkip              = true;
2357
0
  pu.affine                = pu.geo
2358
0
                           = pu.ciip
2359
0
                           = false;
2360
2361
0
  int       mmvdTestNum    = mergeCtx.numValidMergeCand > 1 ? MmvdIdx::ADD_NUM : MmvdIdx::ADD_NUM >> 1;
2362
0
  int       bestDir        = 0;
2363
0
  size_t    curListSize    = m_mergeItemList.size();
2364
0
  double    bestCostMerge  = m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost;
2365
0
  double    bestCostOffset = MAX_DOUBLE;
2366
0
  int       shiftCandStart = 0;
2367
2368
0
  if( m_pcEncCfg->m_MMVD == 4 )
2369
0
  {
2370
0
    const int cnd1idx = m_mergeItemList.size() == 1 ? 0 : 1;
2371
0
    const int mrgCnd0 = m_mergeItemList.getMergeItemInList(       0 )->mergeIdx;
2372
0
    const int mrgCnd1 = m_mergeItemList.getMergeItemInList( cnd1idx )->mergeIdx;
2373
2374
0
    if( mrgCnd0 > 1 && mrgCnd1 > 1 )
2375
0
    {
2376
0
      mmvdTestNum = 0;
2377
0
    }
2378
0
    else if( mrgCnd0 > 1 || mrgCnd1 > 1 )
2379
0
    {
2380
0
      int shiftCand = mrgCnd0 < 2 ? mrgCnd0 : mrgCnd1;
2381
2382
0
      if( shiftCand )
2383
0
      {
2384
0
        shiftCandStart = MMVD_MAX_REFINE_NUM;
2385
0
      }
2386
0
      else
2387
0
      {
2388
0
        mmvdTestNum    = MMVD_MAX_REFINE_NUM;
2389
0
      }
2390
0
    }
2391
0
  }
2392
2393
0
  for( int mmvdMergeCand = shiftCandStart; mmvdMergeCand < mmvdTestNum; mmvdMergeCand++ )
2394
0
  {
2395
0
    MmvdIdx mmvdIdx;
2396
0
    mmvdIdx.val = mmvdMergeCand;
2397
2398
0
    if( mmvdIdx.pos.step >= m_pcEncCfg->m_MmvdDisNum )
2399
0
    {
2400
0
      continue;
2401
0
    }
2402
2403
0
    if( m_pcEncCfg->m_MMVD > 1 )
2404
0
    {
2405
0
      int checkMMVD = xCheckMMVDCand( mmvdIdx, bestDir, mmvdTestNum, bestCostOffset, bestCostMerge, m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost );
2406
0
      mmvdMergeCand = mmvdIdx.val;
2407
2408
0
      if( checkMMVD )
2409
0
      {
2410
0
        if( checkMMVD == 2 )
2411
0
        {
2412
0
          break;
2413
0
        }
2414
0
        continue;
2415
0
      }
2416
0
    }
2417
2418
0
    mergeCtx.setMmvdMergeCandiInfo( pu, mmvdIdx );
2419
2420
0
    if( m_pcEncCfg->m_ifpLines &&
2421
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2422
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2423
0
    {
2424
      // skip candidate
2425
0
      continue;
2426
0
    }
2427
2428
0
    pu.interDir               = mergeCtx.interDirNeighbours[mmvdIdx.pos.baseIdx];
2429
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[mmvdIdx.pos.baseIdx] : BCW_DEFAULT;
2430
0
    pu.imv                    = mergeCtx.useAltHpelIf[mmvdIdx.pos.baseIdx] ? IMV_HPEL : IMV_OFF;
2431
0
    CU::spanMotionInfo        ( pu );
2432
2433
0
    MergeItem *mmvdMerge      = m_mergeItemList.allocateNewMergeItem();
2434
0
    mmvdMerge->importMergeInfo( mergeCtx, mmvdIdx.val, MergeItem::MergeItemType::MMVD, pu );
2435
0
    auto dstBuf               = mmvdMerge->getPredBuf( localUnitArea );
2436
0
    generateMergePrediction   ( localUnitArea, mmvdMerge, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2437
0
    mmvdMerge->cost           = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2438
0
    m_mergeItemList           . insertMergeItemToList( mmvdMerge );
2439
2440
0
    if( m_pcEncCfg->m_MMVD > 1 && mmvdMerge->cost < bestCostOffset )
2441
0
    {
2442
0
      bestCostOffset          = mmvdMerge->cost;
2443
0
      int CandCur             = mmvdIdx.val - MMVD_MAX_REFINE_NUM * mmvdIdx.pos.baseIdx;
2444
0
      if( CandCur < 4 )
2445
0
        bestDir               = CandCur;
2446
0
    }
2447
0
  }
2448
2449
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2450
0
  {
2451
0
    m_mergeItemList           . shrinkList( curListSize );
2452
0
  }
2453
0
}
2454
2455
void EncCu::addAffineCandsToPruningList( AffineMergeCtx &affineMergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2456
                                         const TempCtx& ctxStart, DistParam& distParam, CodingUnit& pu)
2457
0
{
2458
0
  bool sameMV[AFFINE_MRG_MAX_NUM_CANDS + 1]
2459
0
                      = { false, };
2460
0
  size_t curListSize  = m_mergeItemList.size();
2461
2462
0
  pu.mergeFlag = true;
2463
0
  pu.affine    = true;
2464
0
  pu.imv       = IMV_OFF;
2465
0
  pu.geo       = pu.mmvdMergeFlag = pu.mmvdSkip
2466
0
               = pu.ciip
2467
0
               = false;
2468
2469
0
  if( m_pcEncCfg->m_Affine > 1 )
2470
0
  {
2471
0
    for( int m = 0; m < affineMergeCtx.numValidMergeCand; m++ )
2472
0
    {
2473
0
      if( pu.cs->slice->TLayer > 3 && affineMergeCtx.mergeType[m] != MRG_TYPE_SUBPU_ATMVP )
2474
0
      {
2475
0
        sameMV[m] = m != 0;
2476
0
      }
2477
0
      else if( !sameMV[m + 1] )
2478
0
      {
2479
0
        for( int n = m + 1; n < affineMergeCtx.numValidMergeCand; n++ )
2480
0
        {
2481
0
          sameMV[n] |= affineMergeCtx.mvFieldNeighbours[m][0][0] == affineMergeCtx.mvFieldNeighbours[n][0][0]
2482
0
                    && affineMergeCtx.mvFieldNeighbours[m][1][0] == affineMergeCtx.mvFieldNeighbours[n][1][0];
2483
0
        }
2484
0
      }
2485
0
    }
2486
0
  }
2487
2488
0
  for( uint32_t mergeIdx = 0; mergeIdx < affineMergeCtx.numValidMergeCand; mergeIdx++ )
2489
0
  {
2490
0
    if( ( affineMergeCtx.mergeType[mergeIdx] != MRG_TYPE_SUBPU_ATMVP && m_pcEncCfg->m_Affine == 0 ) || sameMV[mergeIdx] )
2491
0
    {
2492
0
      continue;
2493
0
    }
2494
2495
0
    pu.mergeType              = affineMergeCtx.mergeType[mergeIdx];
2496
0
    pu.affineType             = affineMergeCtx.affineType[mergeIdx];
2497
0
    pu.interDir               = affineMergeCtx.interDirNeighbours[mergeIdx];
2498
0
    pu.BcwIdx                 = pu.interDir == 3 ? affineMergeCtx.BcwIdx[mergeIdx] : BCW_DEFAULT;
2499
2500
    // generate motion buf for IFP
2501
0
    if( affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP )
2502
0
    {
2503
0
      pu.refIdx[L0]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].refIdx;
2504
0
      pu.refIdx[L1]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].refIdx;
2505
0
      pu.mv    [L0][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].mv;
2506
0
      pu.mv    [L1][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].mv;
2507
0
      CU::spanMotionInfo      ( pu, &affineMergeCtx );
2508
0
    }
2509
0
    else
2510
0
    {
2511
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L0], L0 );
2512
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L1], L1 );
2513
0
      CU::spanMotionInfo      ( pu );
2514
0
    }
2515
2516
0
    if( m_pcEncCfg->m_ifpLines && !CU::isMotionBufInRangeFPP( pu, m_pcEncCfg->m_ifpLines ) )
2517
0
    {
2518
0
      continue;
2519
0
    }
2520
2521
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2522
0
    mergeItem->importMergeInfo( affineMergeCtx, mergeIdx, affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP ? MergeItem::MergeItemType::SBTMVP : MergeItem::MergeItemType::AFFINE, pu );
2523
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2524
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2525
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2526
0
    m_mergeItemList        . insertMergeItemToList( mergeItem );
2527
0
  }
2528
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2529
0
  {
2530
0
    m_mergeItemList        . shrinkList( curListSize );
2531
0
  }
2532
0
}
2533
2534
void EncCu::addGpmCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2535
                                      const TempCtx& ctxStart, const GeoComboCostList& comboList, MergeBufVector& geoBuffer, DistParam& distParam, CodingUnit& pu)
2536
0
{
2537
0
  int geoNumMrgSadCand    = std::min( GEO_MAX_TRY_WEIGHTED_SAD, ( int ) comboList.list.size() );
2538
0
  geoNumMrgSadCand        = std::min( geoNumMrgSadCand, m_pcEncCfg->m_Geo > 2 ? 10 : GEO_MAX_TRY_WEIGHTED_SAD );
2539
0
  double bestGeoCost      = MAX_DOUBLE / 2.0;
2540
0
  MergeItem* best2geo[2]  = { nullptr, nullptr };
2541
2542
0
  pu.mergeFlag = true;
2543
0
  pu.geo       = true;
2544
0
  pu.mergeType = MRG_TYPE_DEFAULT_N;
2545
0
  pu.BcwIdx    = BCW_DEFAULT;
2546
0
  pu.interDir  = 3;
2547
0
  pu.imv       = IMV_OFF;
2548
0
  pu.affine    = pu.mmvdMergeFlag = pu.mmvdSkip
2549
0
               = pu.ciip
2550
0
               = false;
2551
2552
0
  for( int candidateIdx = 0; candidateIdx < geoNumMrgSadCand; candidateIdx++ )
2553
0
  {
2554
0
    const int          splitDir     = comboList.list[candidateIdx].splitDir;
2555
0
    const MergeIdxPair mergeIdxPair { comboList.list[candidateIdx].mergeIdx0, comboList.list[candidateIdx].mergeIdx1 };
2556
0
    const int          gpmIndex     = MergeItem::getGpmUnfiedIndex( splitDir, mergeIdxPair );
2557
2558
0
    pu.mergeIdx            = gpmIndex;
2559
0
    pu.geoMergeIdx         = mergeIdxPair;
2560
0
    pu.geoSplitDir         = splitDir;
2561
0
    CU::spanGeoMotionInfo  ( pu, mergeCtx, pu.geoSplitDir, pu.geoMergeIdx[0], pu.geoMergeIdx[1] );
2562
2563
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2564
0
    mergeItem->importMergeInfo( mergeCtx, gpmIndex, MergeItem::MergeItemType::GPM, pu );
2565
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2566
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, &geoBuffer[mergeIdxPair[0]], &geoBuffer[mergeIdxPair[1]] );
2567
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2568
0
    bestGeoCost            = std::min( mergeItem->cost, bestGeoCost );
2569
2570
0
    if( mergeItem->cost > MRG_FAST_RATIO[0] * bestGeoCost || mergeItem->cost > m_mergeBestSATDCost )
2571
0
    {
2572
0
      m_mergeItemList      . giveBackMergeItem( mergeItem );
2573
2574
0
      if( m_pcEncCfg->m_Geo > 2 ) break;
2575
0
    }
2576
0
    else if( m_pcEncCfg->m_Geo < 2 )
2577
0
    {
2578
0
      m_mergeItemList      . insertMergeItemToList( mergeItem );
2579
0
    }
2580
0
    else
2581
0
    {
2582
0
      if( ( m_mergeItemList.size() > 0 && m_mergeItemList.getMergeItemInList( m_mergeItemList.size() - 1 )->cost <= mergeItem->cost ) ||
2583
0
        ( best2geo[1] && best2geo[1]->cost <= mergeItem->cost ) )
2584
0
      {
2585
0
        m_mergeItemList    . giveBackMergeItem( mergeItem );
2586
0
      }
2587
0
      else
2588
0
      {
2589
0
        if( !best2geo[0] || mergeItem->cost < best2geo[0]->cost )
2590
0
        {
2591
0
          if( best2geo[1] )
2592
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2593
2594
0
          best2geo[1] = best2geo[0]; best2geo[0] = mergeItem;
2595
0
        }
2596
0
        else
2597
0
        {
2598
0
          if( best2geo[1] ) 
2599
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2600
2601
0
          best2geo[1] = mergeItem;
2602
0
        }
2603
0
      }
2604
0
    }
2605
0
  }
2606
2607
0
  if( best2geo[0] )
2608
0
    m_mergeItemList        . insertMergeItemToList( best2geo[0] );
2609
0
  if( best2geo[1] )
2610
0
    m_mergeItemList        . insertMergeItemToList( best2geo[1] );
2611
0
}
2612
2613
bool EncCu::prepareGpmComboList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2614
                                 GeoComboCostList& comboList, MergeBufVector& geoBuffer, CodingUnit& pu )
2615
0
{
2616
0
          sqrtLambdaForFirstPass /= FRAC_BITS_SCALE;
2617
0
  const int bitsForPartitionIdx   = floorLog2(GEO_NUM_PARTITION_MODE);
2618
0
  const int maxNumMergeCandidates = std::min( ( int ) pu.cs->sps->maxNumGeoCand, MRG_MAX_NUM_CANDS );
2619
0
  DistParam distParam;
2620
  // the second arguments to setDistParam is dummy and will be updated before being used
2621
0
  DistParam  distParamWholeBlk     = m_cRdCost.setDistParam( pu.cs->getOrgBuf().Y(), pu.cs->getOrgBuf().Y(), pu.cs->sps->bitDepths[ CH_L ], DF_SAD );
2622
0
  Distortion bestWholeBlkSad       = MAX_UINT64;
2623
0
  double     bestWholeBlkCost      = MAX_DOUBLE;
2624
0
  const ClpRng&  lclpRng           = pu.slice->clpRngs[COMP_Y];
2625
0
  const unsigned rshift            = std::max<int>( 2, ( IF_INTERNAL_PREC - lclpRng.bd ) );
2626
0
  const int      offset            = ( 1 << ( rshift - 1 ) ) + IF_INTERNAL_OFFS;
2627
0
  const int      numSamples        = pu.Y().area();
2628
0
  Distortion sadWholeBlk            [GEO_MAX_NUM_UNI_CANDS];
2629
0
  int        pocMrg                 [GEO_MAX_NUM_UNI_CANDS];
2630
0
  Mv         mergeMv                [GEO_MAX_NUM_UNI_CANDS];
2631
0
  bool       isSkipThisCand         [GEO_MAX_NUM_UNI_CANDS]
2632
0
                                   = { false, };
2633
0
  bool       sameMV                 [MRG_MAX_NUM_CANDS]
2634
0
                                   = { false, };
2635
0
  MergeBufVector geoTempBuf;
2636
2637
0
  if( m_pcEncCfg->m_Geo > 2 )
2638
0
  {
2639
0
    for( int m = 0; m < maxNumMergeCandidates; m++ )
2640
0
    {
2641
0
      if( !sameMV[m] )
2642
0
      {
2643
0
        for( int n = m + 1; n < maxNumMergeCandidates; n++ )
2644
0
        {
2645
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
2646
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
2647
0
        }
2648
0
      }
2649
0
    }
2650
0
  }
2651
2652
0
  for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2653
0
  {
2654
0
    geoBuffer .push_back ( m_aTmpStorageLCU[2                         + mergeCand].getCompactBuf( localUnitArea ) );
2655
0
    geoTempBuf.push_back ( m_aTmpStorageLCU[2 + GEO_MAX_NUM_UNI_CANDS + mergeCand].getCompactBuf( localUnitArea ) );
2656
2657
0
    const int  listIdx    = mergeCtx.mvFieldNeighbours[mergeCand][0]      .refIdx == -1 ? 1 : 0;
2658
0
    const auto refPicList = RefPicList(listIdx);
2659
0
    const int  refIdx     = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].refIdx;
2660
2661
0
    pocMrg [mergeCand]    = pu.cs->slice->getRefPic( refPicList, refIdx )->poc;
2662
0
    mergeMv[mergeCand]    = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].mv;
2663
2664
0
    for( int i = 0; i < mergeCand; i++ )
2665
0
    {
2666
0
      if( pocMrg[mergeCand] == pocMrg[i] && mergeMv[mergeCand] == mergeMv[i] )
2667
0
      {
2668
0
        isSkipThisCand[mergeCand] = true;
2669
0
        break;
2670
0
      }
2671
0
    }
2672
2673
0
    if( sameMV[mergeCand] )
2674
0
    {
2675
0
      continue;
2676
0
    }
2677
2678
0
    if( m_pcEncCfg->m_ifpLines ) 
2679
0
    {
2680
0
      bool isOutOfRange  = !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][0].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2681
0
           isOutOfRange |= !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][1].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2682
2683
      // use sameMV to surpress processing of this cand later on...
2684
0
      sameMV[mergeCand] |= isOutOfRange;
2685
2686
0
      if( isOutOfRange )
2687
0
        continue;
2688
0
    }
2689
2690
0
    mergeCtx.setMergeInfo            ( pu, mergeCand );
2691
0
    CU::spanMotionInfo               ( pu );
2692
0
    m_cInterSearch.motionCompensation( pu, geoBuffer[mergeCand], REF_PIC_LIST_X );
2693
2694
0
    g_pelBufOP.roundGeo( geoBuffer[mergeCand].Y().buf, geoTempBuf[mergeCand].Y().buf, numSamples, rshift, offset, lclpRng );
2695
2696
0
    distParamWholeBlk.cur  = geoTempBuf[mergeCand].Y();
2697
0
    sadWholeBlk[mergeCand] = distParamWholeBlk.distFunc( distParamWholeBlk );
2698
2699
0
    if( sadWholeBlk[mergeCand] < bestWholeBlkSad )
2700
0
    {
2701
0
      bestWholeBlkSad  = sadWholeBlk[mergeCand];
2702
0
      int bitsCand     = mergeCand + 1;
2703
0
      bestWholeBlkCost = ( double ) bestWholeBlkSad + ( double ) bitsCand * sqrtLambdaForFirstPass;
2704
0
    }
2705
0
  }
2706
2707
0
  bool allCandsAreSame = true;
2708
0
  for( uint8_t mergeCand = 1; mergeCand < maxNumMergeCandidates; mergeCand++ )
2709
0
  {
2710
0
    allCandsAreSame &= isSkipThisCand[mergeCand];
2711
0
  }
2712
0
  if( allCandsAreSame )
2713
0
  {
2714
0
    return false;
2715
0
  }
2716
2717
0
  const int wIdx = floorLog2( pu.lwidth() )  - GEO_MIN_CU_LOG2;
2718
0
  const int hIdx = floorLog2( pu.lheight() ) - GEO_MIN_CU_LOG2;
2719
2720
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2721
0
  {
2722
0
    int maskStride = 0, maskStride2 = 0;
2723
0
    int stepX = 1;
2724
0
    Pel *sadMask;
2725
0
    int16_t angle = g_GeoParams[splitDir][0];
2726
    
2727
0
    if( g_angle2mirror[angle] == 2 )
2728
0
    {
2729
0
      maskStride  = -GEO_WEIGHT_MASK_SIZE;
2730
0
      maskStride2 = -( int ) pu.lwidth();
2731
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2732
0
                      [( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][1] ) * GEO_WEIGHT_MASK_SIZE
2733
0
                                                  + g_weightOffset[hIdx][wIdx][splitDir][0]
2734
0
                      ];
2735
0
    }
2736
0
    else if( g_angle2mirror[angle] == 1 )
2737
0
    {
2738
0
      stepX       = -1;
2739
0
      maskStride2 = pu.lwidth();
2740
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2741
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2742
0
                      [     GEO_WEIGHT_MASK_SIZE *     g_weightOffset[hIdx][wIdx][splitDir][1]
2743
0
                        + ( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][0] )
2744
0
                      ];
2745
0
    }
2746
0
    else
2747
0
    {
2748
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2749
0
      maskStride2 = -( int ) pu.lwidth();
2750
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2751
0
                      [   g_weightOffset[hIdx][wIdx][splitDir][1] * GEO_WEIGHT_MASK_SIZE
2752
0
                        + g_weightOffset[hIdx][wIdx][splitDir][0]
2753
0
                      ];
2754
0
    }
2755
2756
0
    m_cRdCost.setDistParamGeo ( distParam, pu.cs->getOrgBuf().Y(),
2757
0
                                nullptr, 0,
2758
0
                                sadMask, maskStride, stepX, maskStride2,
2759
0
                                pu.cs->sps->bitDepths[CH_L], COMP_Y );
2760
2761
0
    for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2762
0
    {
2763
0
      if( sameMV[mergeCand] )
2764
0
      {
2765
0
        continue;
2766
0
      }
2767
2768
0
      distParam.cur.buf         = geoTempBuf[mergeCand].Y().buf;
2769
0
      distParam.cur.stride      = geoTempBuf[mergeCand].Y().stride;
2770
0
      const Distortion sadLarge = distParam.distFunc( distParam );
2771
0
      const Distortion sadSmall = sadWholeBlk[mergeCand] - sadLarge;
2772
2773
0
      const int bitsCand        = mergeCand + 1;
2774
2775
0
      const double cost0        = ( double ) sadLarge + ( double ) bitsCand * sqrtLambdaForFirstPass;
2776
0
      const double cost1        = ( double ) sadSmall + ( double ) bitsCand * sqrtLambdaForFirstPass;
2777
2778
0
      m_GeoCostList.insert( splitDir, 0, mergeCand, cost0 );
2779
0
      m_GeoCostList.insert( splitDir, 1, mergeCand, cost1 );
2780
0
    }
2781
2782
0
    if( m_pcEncCfg->m_Geo == 4 )
2783
0
    {
2784
0
      if( splitDir == 1 )
2785
0
      {
2786
0
        splitDir += 7;
2787
0
      }
2788
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2789
0
      {
2790
0
        splitDir++;
2791
0
      }
2792
0
      else
2793
0
      {
2794
0
        splitDir += 5;
2795
0
      }
2796
0
    }
2797
0
    else
2798
0
    {
2799
0
      splitDir++;
2800
0
    }
2801
0
  }
2802
2803
0
  comboList.list.clear();
2804
2805
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2806
0
  {
2807
0
    for( int geoMotionIdx = 0; geoMotionIdx < maxNumMergeCandidates * ( maxNumMergeCandidates - 1 ); geoMotionIdx++ )
2808
0
    {
2809
0
      const MergeIdxPair mergeIdxPair = m_GeoModeTest[geoMotionIdx];
2810
2811
0
      if( sameMV[mergeIdxPair[0]] || sameMV[mergeIdxPair[1]] )
2812
0
      {
2813
0
        continue;
2814
0
      }
2815
2816
0
      double tempCost = m_GeoCostList.getCost( splitDir, mergeIdxPair[0], mergeIdxPair[1] );
2817
2818
0
      if( tempCost > bestWholeBlkCost )
2819
0
      {
2820
0
        continue;
2821
0
      }
2822
2823
0
      tempCost = tempCost + ( double ) bitsForPartitionIdx * sqrtLambdaForFirstPass;
2824
0
      comboList.list.push_back( GeoMergeCombo{ splitDir, mergeIdxPair[0], mergeIdxPair[1], tempCost } );
2825
0
    }
2826
2827
0
    if( m_pcEncCfg->m_Geo == 4 )
2828
0
    {
2829
0
      if( splitDir == 1 )
2830
0
      {
2831
0
        splitDir += 7;
2832
0
      }
2833
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2834
0
      {
2835
0
        splitDir++;
2836
0
      }
2837
0
      else
2838
0
      {
2839
0
        splitDir += 5;
2840
0
      }
2841
0
    }
2842
0
    else
2843
0
    {
2844
0
      splitDir++;
2845
0
    }
2846
0
  }
2847
2848
0
  if( comboList.list.empty() )
2849
0
  {
2850
0
    return false;
2851
0
  }
2852
2853
0
  comboList.sortByCost();
2854
0
  return true;
2855
0
}
2856
2857
double EncCu::calcLumaCost4MergePrediction( const TempCtx &ctxStart, const PelUnitBuf &predBuf, double lambda, CodingUnit &cu, DistParam &distParam )
2858
0
{
2859
0
  distParam.cur = predBuf.Y();
2860
0
  auto dist     = distParam.distFunc(distParam);
2861
2862
0
  m_CABACEstimator->getCtx() = ctxStart;
2863
0
  auto fracBits = xCalcPuMeBits( cu );
2864
2865
0
  double cost   = ( double ) dist + ( double ) fracBits * lambda;
2866
2867
0
  m_uiSadBestForQPA = std::min( dist, m_uiSadBestForQPA );
2868
2869
0
  return cost;
2870
0
}
2871
2872
//////////////////////////////////////////////////////////////////////////////////////////////
2873
// ibc merge/skip mode check
2874
void EncCu::xCheckRDCostIBCModeMerge2Nx2N(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner, const EncTestMode& encTestMode)
2875
25.2k
{
2876
25.2k
  assert(partitioner.chType != CH_C); // chroma IBC is derived
2877
25.2k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128) // disable IBC mode larger than 64x64
2878
0
  {
2879
0
    return;
2880
0
  }
2881
2882
25.2k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
2883
0
  {
2884
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
2885
0
    {
2886
0
      return;
2887
0
    }
2888
0
  }
2889
2890
25.2k
  const SPS& sps = *tempCS->sps;
2891
2892
25.2k
  tempCS->initStructData(encTestMode.qp);
2893
25.2k
  MergeCtx mergeCtx;
2894
2895
25.2k
  {
2896
    // first get merge candidates
2897
25.2k
    CodingUnit cu(tempCS->area);
2898
25.2k
    cu.cs = tempCS;
2899
25.2k
    cu.predMode = MODE_IBC;
2900
25.2k
    cu.slice = tempCS->slice;
2901
25.2k
    cu.tileIdx = m_tileIdx;
2902
25.2k
    cu.initPuData();
2903
25.2k
    cu.cs = tempCS;
2904
25.2k
    cu.mmvdSkip = false;
2905
25.2k
    cu.mmvdMergeFlag = false;
2906
25.2k
    cu.geo = false;
2907
25.2k
    CU::getIBCMergeCandidates(cu, mergeCtx);
2908
25.2k
  }
2909
25.2k
  int candHasNoResidual[MRG_MAX_NUM_CANDS];
2910
176k
  for (unsigned int ui = 0; ui < mergeCtx.numValidMergeCand; ui++)
2911
151k
  {
2912
151k
    candHasNoResidual[ui] = 0;
2913
151k
  }
2914
2915
25.2k
  bool                                        bestIsSkip = false;
2916
25.2k
  unsigned                                    numMrgSATDCand = mergeCtx.numValidMergeCand;
2917
25.2k
  static_vector<unsigned, MRG_MAX_NUM_CANDS>  RdModeList(MRG_MAX_NUM_CANDS);
2918
176k
  for (unsigned i = 0; i < MRG_MAX_NUM_CANDS; i++)
2919
151k
  {
2920
151k
    RdModeList[i] = i;
2921
151k
  }
2922
2923
  //{
2924
25.2k
  static_vector<double, MRG_MAX_NUM_CANDS>  candCostList(MRG_MAX_NUM_CANDS, MAX_DOUBLE);
2925
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
2926
25.2k
  {
2927
25.2k
    const double sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda();
2928
2929
25.2k
    CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType,partitioner.treeType), partitioner.chType);
2930
2931
25.2k
    partitioner.setCUData(cu);
2932
25.2k
    cu.slice = tempCS->slice;
2933
25.2k
    cu.tileIdx = m_tileIdx;
2934
25.2k
    cu.skip = false;
2935
25.2k
    cu.predMode = MODE_IBC;
2936
25.2k
    cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
2937
25.2k
    cu.qp = encTestMode.qp;
2938
25.2k
    cu.mmvdSkip = false;
2939
25.2k
    cu.geo = false;
2940
25.2k
    DistParam distParam;
2941
25.2k
    cu.initPuData();
2942
25.2k
    cu.mmvdMergeFlag = false;
2943
25.2k
    Picture* refPic = cu.slice->pic;
2944
25.2k
    const UnitArea localUnitArea(tempCS->area.chromaFormat, Area(cu.blocks[COMP_Y].x, cu.blocks[COMP_Y].y, tempCS->area.Y().width, tempCS->area.Y().height));
2945
25.2k
    const CompArea& compArea = localUnitArea.block(COMP_Y);
2946
25.2k
    const CPelBuf refBuf = refPic->getRecoBuf(compArea);
2947
25.2k
    const Pel* piRefSrch = refBuf.buf;
2948
2949
25.2k
      distParam = m_cRdCost.setDistParam(tempCS->getOrgBuf(COMP_Y), refBuf, sps.bitDepths[CH_L], DF_HAD);
2950
25.2k
    int refStride = refBuf.stride;
2951
2952
25.2k
    int numValidBv = mergeCtx.numValidMergeCand;
2953
176k
    for (unsigned int mergeCand = 0; mergeCand < mergeCtx.numValidMergeCand; mergeCand++)
2954
151k
    {
2955
151k
      mergeCtx.setMergeInfo(cu, mergeCand); // set bv info in merge mode
2956
151k
      const int cuPelX = cu.Y().x;
2957
151k
      const int cuPelY = cu.Y().y;
2958
151k
      int roiWidth     = cu.lwidth();
2959
151k
      int roiHeight    = cu.lheight();
2960
151k
      const int picWidth  = cu.cs->slice->pps->picWidthInLumaSamples;
2961
151k
      const int picHeight = cu.cs->slice->pps->picHeightInLumaSamples;
2962
151k
      const unsigned int lcuWidth = cu.cs->slice->sps->CTUSize;
2963
2964
151k
      Mv bv = cu.mv[0][0];
2965
151k
      bv.changePrecision( MV_PRECISION_INTERNAL, MV_PRECISION_INT);
2966
151k
      int xPred = bv.hor;
2967
151k
      int yPred = bv.ver;
2968
      
2969
151k
      if( !m_cInterSearch.searchBvIBC( cu, cuPelX, cuPelY, roiWidth, roiHeight, picWidth, picHeight, xPred, yPred, lcuWidth ) ) // not valid bv derived
2970
151k
      {
2971
151k
        numValidBv--;
2972
151k
        continue;
2973
151k
      }
2974
1
      CU::spanMotionInfo(cu);
2975
1
      distParam.cur.buf = piRefSrch + refStride * yPred + xPred;
2976
2977
1
      Distortion sad = distParam.distFunc(distParam);
2978
1
      unsigned int bitsCand = mergeCand + 1;
2979
1
      if (mergeCand == tempCS->sps->maxNumIBCMergeCand - 1)
2980
0
      {
2981
0
        bitsCand--;
2982
0
      }
2983
1
      double cost = (double)sad + (double)bitsCand * sqrtLambdaForFirstPass;
2984
2985
1
      updateCandList( mergeCand, cost, RdModeList, candCostList, numMrgSATDCand );
2986
1
    }
2987
2988
    // Try to limit number of candidates using SATD-costs
2989
25.2k
    if (numValidBv)
2990
0
    {
2991
0
      numMrgSATDCand = numValidBv;
2992
0
      for (unsigned int i = 1; i < numValidBv; i++)
2993
0
      {
2994
0
        if (candCostList[i] > MRG_FAST_RATIO[0] * candCostList[0])
2995
0
        {
2996
0
          numMrgSATDCand = i;
2997
0
          break;
2998
0
        }
2999
0
      }
3000
0
    }
3001
25.2k
    else
3002
25.2k
    {
3003
25.2k
      tempCS->dist = 0;
3004
25.2k
      tempCS->fracBits = 0;
3005
25.2k
      tempCS->cost = MAX_DOUBLE;
3006
25.2k
      tempCS->costDbOffset = 0;
3007
25.2k
      tempCS->initStructData(encTestMode.qp);
3008
25.2k
      return;
3009
25.2k
    }
3010
3011
0
    tempCS->initStructData(encTestMode.qp);
3012
0
  }
3013
  //}
3014
3015
3016
0
  const unsigned int iteration = 2;
3017
 // m_bestModeUpdated = tempCS->cost = bestCS->cost = false;
3018
  // 2. Pass: check candidates using full RD test
3019
0
  for (unsigned int numResidualPass = 0; numResidualPass < iteration; numResidualPass++)
3020
0
  {
3021
0
    for (unsigned int mrgHADIdx = 0; mrgHADIdx < numMrgSATDCand; mrgHADIdx++)
3022
0
    {
3023
0
      unsigned int mergeCand = RdModeList[mrgHADIdx];
3024
0
      if (!(numResidualPass == 1 && candHasNoResidual[mergeCand] == 1))
3025
0
      {
3026
0
        if (!(bestIsSkip && (numResidualPass == 0)))
3027
0
        {
3028
0
          {
3029
3030
            // first get merge candidates
3031
0
            CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, (const ChannelType)partitioner.chType,partitioner.treeType), (const ChannelType)partitioner.chType);
3032
3033
0
            partitioner.setCUData(cu);
3034
0
            cu.slice = tempCS->slice;
3035
0
            cu.tileIdx = m_tileIdx;
3036
0
            cu.skip = false;
3037
0
            cu.predMode = MODE_IBC;
3038
0
            cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3039
0
            cu.qp = encTestMode.qp;
3040
0
            cu.sbtInfo = 0;
3041
0
            cu.initPuData();
3042
0
            cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3043
0
            cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3044
0
            cu.mmvdSkip = false;
3045
0
            cu.mmvdMergeFlag = false;
3046
0
            cu.geo = false;
3047
0
            mergeCtx.setMergeInfo(cu, mergeCand);
3048
0
            CU::spanMotionInfo(cu);
3049
3050
0
            assert(mergeCtx.mrgTypeNeighbours[mergeCand] == MRG_TYPE_IBC);
3051
0
            const bool chroma = !CU::isSepTree(cu);
3052
3053
            //  MC
3054
0
            cu.mcControl = chroma ? 0: 2;
3055
0
            m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3056
0
            m_CABACEstimator->getCtx() = m_CurrCtx->start;
3057
3058
0
            m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, (numResidualPass != 0));
3059
0
            cu.mcControl = 0;
3060
0
            xEncodeDontSplit(*tempCS, partitioner);
3061
0
            xCheckDQP(*tempCS, partitioner);
3062
0
            xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3063
3064
0
            tempCS->initStructData(encTestMode.qp);
3065
0
          }
3066
3067
0
          if (m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip)
3068
0
          {
3069
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType) == NULL)
3070
0
              bestIsSkip = 0;
3071
0
            else
3072
0
              bestIsSkip = bestCS->getCU(partitioner.chType, partitioner.treeType)->rootCbf == 0;
3073
0
          }
3074
0
        }
3075
0
      }
3076
0
    }
3077
0
  }
3078
0
}
3079
3080
void EncCu::xCheckRDCostIBCMode(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner,
3081
  const EncTestMode& encTestMode)
3082
25.2k
{
3083
25.2k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128)   // disable IBC mode larger than 64x64
3084
0
  {
3085
0
    return;
3086
0
  }
3087
25.2k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
3088
0
  {
3089
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3090
0
    {
3091
0
      return;
3092
0
    }
3093
0
  }
3094
3095
25.2k
  tempCS->initStructData(encTestMode.qp);
3096
3097
25.2k
  CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType, partitioner.treeType), partitioner.chType);
3098
3099
25.2k
  partitioner.setCUData(cu);
3100
25.2k
  cu.slice = tempCS->slice;
3101
25.2k
  cu.tileIdx = m_tileIdx;
3102
25.2k
  cu.skip = false;
3103
25.2k
  cu.predMode = MODE_IBC;
3104
25.2k
  cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3105
25.2k
  cu.qp = encTestMode.qp;
3106
25.2k
  cu.initPuData();
3107
25.2k
  cu.imv = IMV_OFF;
3108
25.2k
  cu.sbtInfo = 0;
3109
25.2k
  cu.mmvdSkip = false;
3110
25.2k
  cu.mmvdMergeFlag = false;
3111
3112
25.2k
  cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3113
25.2k
  cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3114
3115
25.2k
  cu.interDir = 1; // use list 0 for IBC mode
3116
25.2k
  cu.refIdx[REF_PIC_LIST_0] = MAX_NUM_REF; // last idx in the list
3117
25.2k
  bool bValid = m_cInterSearch.predIBCSearch(cu, partitioner);
3118
3119
25.2k
  if (bValid)
3120
22.5k
  {
3121
22.5k
    CU::spanMotionInfo(cu);
3122
22.5k
    const bool chroma = !CU::isSepTree(cu);
3123
    //  MC
3124
22.5k
    cu.mcControl = chroma ? 0 : 2;
3125
22.5k
    m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3126
3127
22.5k
    m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, false);
3128
22.5k
    cu.mcControl = 0;
3129
3130
22.5k
    xEncodeDontSplit(*tempCS, partitioner);
3131
22.5k
    xCheckDQP(*tempCS, partitioner);
3132
22.5k
    xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3133
22.5k
  } // bValid
3134
2.71k
  else
3135
2.71k
  {
3136
2.71k
    tempCS->dist = 0;
3137
2.71k
    tempCS->fracBits = 0;
3138
2.71k
    tempCS->cost = MAX_DOUBLE;
3139
2.71k
    tempCS->costDbOffset = 0;
3140
2.71k
  }
3141
25.2k
}
3142
3143
void EncCu::xCheckRDCostInter( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
3144
0
{
3145
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD, tempCS, partitioner.chType );
3146
0
  tempCS->initStructData( encTestMode.qp );
3147
3148
0
  m_cInterSearch.setAffineModeSelected( false );
3149
3150
0
  m_cInterSearch.resetBufferedUniMotions();
3151
3152
0
  int bcwLoopNum = BCW_NUM;
3153
3154
0
  if( tempCS->area.Y().area() < BCW_SIZE_CONSTRAINT || !tempCS->slice->isInterB() || !tempCS->sps->BCW )
3155
0
  {
3156
0
    bcwLoopNum = 1;
3157
0
  }
3158
  
3159
0
  double curBestCost = bestCS->cost;
3160
0
  double equBcwCost = MAX_DOUBLE;
3161
3162
0
  for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3163
0
  {
3164
0
    if( m_pcEncCfg->m_BCW == 2 )
3165
0
    {
3166
0
      bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3167
0
      uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3168
3169
0
      if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3170
0
      {
3171
0
        continue;
3172
0
      }
3173
0
    }
3174
    
3175
0
    if( !tempCS->slice->checkLDC )
3176
0
    {
3177
0
      if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3178
0
      {
3179
0
        continue;
3180
0
      }
3181
0
    }
3182
  
3183
0
    CodingUnit &cu      = tempCS->addCU( tempCS->area, partitioner.chType );
3184
3185
0
    partitioner.setCUData( cu );
3186
0
    cu.slice            = tempCS->slice;
3187
0
    cu.tileIdx          = m_tileIdx;
3188
0
    cu.skip             = false;
3189
0
    cu.mmvdSkip         = false;
3190
0
    cu.predMode         = MODE_INTER;
3191
0
    cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
3192
0
    cu.qp               = encTestMode.qp;
3193
0
    cu.initPuData();
3194
3195
0
    cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3196
0
    uint8_t bcwIdx = cu.BcwIdx;
3197
0
    bool testBcw = (bcwIdx != BCW_DEFAULT);
3198
3199
0
    bool StopInterRes = (m_pcEncCfg->m_FastInferMerge >> 3) & 1;
3200
0
    StopInterRes &= bestCS->slice->TLayer > (m_pcEncCfg->m_maxTLayer - (m_pcEncCfg->m_FastInferMerge & 7));
3201
0
    double bestCostInter = StopInterRes ? m_mergeBestSATDCost : MAX_DOUBLE;
3202
3203
0
    bool stopTest = m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3204
3205
0
    if (StopInterRes && (bestCostInter != m_mergeBestSATDCost))
3206
0
    {
3207
0
      int L = (cu.slice->TLayer <= 2) ? 0 : (cu.slice->TLayer - 2);
3208
0
      if ((bestCostInter > MRG_FAST_RATIOMYV[L] * m_mergeBestSATDCost))
3209
0
      {
3210
0
        stopTest = true;
3211
0
      }
3212
0
    }
3213
3214
0
    if( !stopTest )
3215
0
    {
3216
0
      bcwIdx   = CU::getValidBcwIdx(cu);
3217
0
      stopTest = testBcw && bcwIdx == BCW_DEFAULT;
3218
0
    }
3219
    
3220
0
    if( stopTest )
3221
0
    {
3222
0
      tempCS->initStructData(encTestMode.qp);
3223
0
      continue;
3224
0
    }
3225
3226
0
    CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3227
        
3228
0
    xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3229
    
3230
0
    if( bcwIdx == BCW_DEFAULT )
3231
0
    {
3232
0
      m_cInterSearch.setAffineModeSelected( bestCS->cus.front()->affine && !bestCS->cus.front()->mergeFlag );
3233
0
    }
3234
3235
0
    tempCS->initStructData(encTestMode.qp);
3236
  
3237
0
    double skipTH = MAX_DOUBLE;
3238
0
    skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3239
0
    if( equBcwCost > curBestCost * skipTH )
3240
0
    {
3241
0
      break;
3242
0
    }
3243
3244
0
    if( m_pcEncCfg->m_BCW == 2 )
3245
0
    {
3246
0
      if( ( cu.interDir != 3 && testBcw == 0 && ! m_pcEncCfg->m_picReordering )
3247
0
         || ( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) ) )
3248
0
      {
3249
0
        break;
3250
0
      }
3251
0
    }
3252
0
  }
3253
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3254
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3255
0
}
3256
3257
void EncCu::xCheckRDCostInterIMV(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode)
3258
0
{
3259
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD_IMV, tempCS, partitioner.chType );
3260
0
  bool Test_AMVR = m_pcEncCfg->m_AMVRspeed ? true: false;
3261
0
  if (m_pcEncCfg->m_AMVRspeed > 2 && m_pcEncCfg->m_AMVRspeed < 5 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3262
0
  {
3263
0
    Test_AMVR = false;
3264
0
  }
3265
0
  else if (m_pcEncCfg->m_AMVRspeed > 4 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->mergeFlag && !bestCS->getCU(partitioner.chType, partitioner.treeType)->ciip)
3266
0
  {
3267
0
    Test_AMVR = false;
3268
0
  }
3269
0
  bool Do_Limit = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 4 || m_pcEncCfg->m_AMVRspeed == 6) ? true : false;
3270
0
  bool Do_OnceRes = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 7) ? true : false;
3271
3272
0
  if( Test_AMVR )
3273
0
  {
3274
0
    double Fpel_cost    = m_pcEncCfg->m_AMVRspeed == 1 ? MAX_DOUBLE*0.5 : MAX_DOUBLE;
3275
0
    double costCurStart = m_pcEncCfg->m_AMVRspeed == 1 ? m_modeCtrl.comprCUCtx->bestCostNoImv : bestCS->cost;
3276
0
    double costCur      = MAX_DOUBLE;
3277
0
    double bestCostIMV  = MAX_DOUBLE;
3278
3279
0
    if (Do_OnceRes)
3280
0
    {
3281
0
      costCurStart = xCalcDistortion(bestCS, partitioner.chType, bestCS->sps->bitDepths[CH_L], 0);
3282
0
      Fpel_cost = costCurStart;
3283
0
      tempCS->initSubStructure(*m_pTempCS2, partitioner.chType, partitioner.currArea(), false);
3284
0
    }
3285
3286
0
    CodingStructure *tempCSbest = m_pTempCS2;
3287
3288
0
    m_cInterSearch.setAffineModeSelected( false );
3289
3290
0
    m_cInterSearch.resetBufferedUniMotions();
3291
3292
0
    int bcwLoopNum = (tempCS->slice->isInterB() ? BCW_NUM : 1);
3293
0
    bcwLoopNum = (tempCS->sps->BCW ? bcwLoopNum : 1);
3294
3295
0
    if( tempCS->area.lwidth() * tempCS->area.lheight() < BCW_SIZE_CONSTRAINT )
3296
0
    {
3297
0
      bcwLoopNum = 1;
3298
0
    }
3299
3300
0
    for (int i = 1; i <= IMV_HPEL; i++)
3301
0
    {
3302
0
      double curBestCost = bestCS->cost;
3303
0
      double equBcwCost  = MAX_DOUBLE;
3304
3305
0
      for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3306
0
      {
3307
0
        if( m_pcEncCfg->m_BCW == 2 )
3308
0
        {
3309
0
          bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3310
0
          uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3311
3312
0
          if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3313
0
          {
3314
0
            continue;
3315
0
          }
3316
          
3317
0
          if( tempCS->slice->checkLDC && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT
3318
0
            && (m_bestBcwIdx[0] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[0])
3319
0
            && (m_bestBcwIdx[1] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[1]))
3320
0
          {
3321
0
            continue;
3322
0
          }
3323
0
        }
3324
3325
0
        if( !tempCS->slice->checkLDC )
3326
0
        {
3327
0
          if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3328
0
          {
3329
0
            continue;
3330
0
          }
3331
0
        }
3332
3333
0
        bool testBcw;
3334
0
        uint8_t bcwIdx;
3335
0
        bool isEqualUni = false;
3336
3337
0
        if (i > IMV_FPEL)
3338
0
        {
3339
0
          bool nextimv = false;
3340
0
          double stopCost = i == IMV_HPEL ? 1.25 : 1.06;
3341
0
          if (Fpel_cost > stopCost * costCurStart)
3342
0
          {
3343
0
            nextimv = true;
3344
0
          }
3345
0
          if ( m_pcEncCfg->m_AMVRspeed == 1 )
3346
0
          {
3347
0
            costCurStart = bestCS->cost;
3348
0
          }
3349
0
          if (nextimv)
3350
0
          {
3351
0
            continue;
3352
0
          }
3353
0
        }
3354
3355
0
        bool Do_Search = Do_OnceRes ? false : true;
3356
3357
0
        if (Do_Limit)
3358
0
        {
3359
0
          Do_Search = i == IMV_FPEL ? true : false;
3360
3361
0
          if (i == IMV_HPEL)
3362
0
          {
3363
0
            if (bestCS->slice->TLayer > 3)
3364
0
            {
3365
0
              continue;
3366
0
            }
3367
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType)->imv != 0)
3368
0
            {
3369
0
              Do_Search = true; //do_est
3370
0
            }
3371
0
          }
3372
0
          if (bestCS->getCU(partitioner.chType, partitioner.treeType)->mmvdMergeFlag || bestCS->getCU(partitioner.chType, partitioner.treeType)->geo)
3373
0
          {
3374
0
            Do_Search = true;
3375
0
          }
3376
0
        }
3377
0
        tempCS->initStructData(encTestMode.qp);
3378
3379
0
        if (!Do_Search)
3380
0
        {
3381
0
          tempCS->copyStructure(*bestCS, partitioner.chType, TREE_D);
3382
0
        }
3383
0
        tempCS->dist = 0;
3384
0
        tempCS->fracBits = 0;
3385
0
        tempCS->cost = MAX_DOUBLE;
3386
0
        CodingUnit &cu = (Do_Search) ? tempCS->addCU(tempCS->area, partitioner.chType) : *tempCS->getCU(partitioner.chType, partitioner.treeType);
3387
0
        if (Do_Search)
3388
0
        {
3389
0
          partitioner.setCUData(cu);
3390
0
          cu.slice = tempCS->slice;
3391
0
          cu.tileIdx = m_tileIdx;
3392
0
          cu.skip = false;
3393
0
          cu.mmvdSkip = false;
3394
0
          cu.predMode = MODE_INTER;
3395
0
          cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3396
0
          cu.qp = encTestMode.qp;
3397
3398
0
          cu.initPuData();
3399
3400
0
          cu.imv = i;
3401
3402
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3403
0
          bcwIdx    = cu.BcwIdx;
3404
0
          testBcw   = (bcwIdx != BCW_DEFAULT);
3405
3406
0
          cu.interDir = 10;
3407
          
3408
0
          double bestCostInter = MAX_DOUBLE;
3409
0
          m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3410
          
3411
0
          if ( cu.interDir <= 3 )
3412
0
          {
3413
0
            bcwIdx = CU::getValidBcwIdx(cu);
3414
0
          }
3415
0
          else
3416
0
          {
3417
0
            continue;
3418
0
          }
3419
          
3420
0
          if( testBcw && bcwIdx == BCW_DEFAULT ) // Enabled Bcw but the search results is uni.
3421
0
          {
3422
0
            continue;
3423
0
          }
3424
0
          CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3425
3426
0
          if( m_pcEncCfg->m_BCW == 2 )
3427
0
          {
3428
0
            if( cu.interDir != 3 && testBcw == 0 )
3429
0
            {
3430
0
              isEqualUni = true;
3431
0
            }
3432
0
          }
3433
3434
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3435
0
          {
3436
0
            continue;
3437
0
          }
3438
0
        }
3439
0
        else
3440
0
        {
3441
0
          cu.smvdMode = 0;
3442
0
          cu.affine = false;
3443
0
          cu.imv = i ;
3444
0
          CU::resetMVDandMV2Int(cu);
3445
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3446
0
          {
3447
0
            continue;
3448
0
          }
3449
3450
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3451
3452
0
          cu.mvRefine = true;
3453
0
          m_cInterSearch.motionCompensation(cu, tempCS->getPredBuf() );
3454
0
          cu.mvRefine = false;
3455
0
        }
3456
3457
0
        if( Do_OnceRes )
3458
0
        {
3459
0
          costCur = xCalcDistortion(tempCS, partitioner.chType, tempCS->sps->bitDepths[CH_L], cu.imv );
3460
0
          if (costCur < bestCostIMV)
3461
0
          {
3462
0
            bestCostIMV = costCur;
3463
0
            tempCSbest->getPredBuf().copyFrom(tempCS->getPredBuf());
3464
0
            tempCSbest->clearCUs();
3465
0
            tempCSbest->clearTUs();
3466
0
            tempCSbest->copyStructure(*tempCS, partitioner.chType, TREE_D);
3467
0
          }
3468
0
          if (i > IMV_FPEL)
3469
0
          {
3470
0
            costCurStart = costCurStart > costCur ? costCur : costCurStart;
3471
0
          }
3472
0
        }
3473
0
        else
3474
0
        {
3475
0
          xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3476
0
          costCur = tempCS->cost;
3477
3478
0
          if (i > IMV_FPEL)
3479
0
          {
3480
0
            costCurStart = bestCS->cost;
3481
0
          }
3482
0
        }
3483
3484
0
        if (i == IMV_FPEL)
3485
0
        {
3486
0
           Fpel_cost = costCur;
3487
0
        }
3488
3489
0
        double skipTH = MAX_DOUBLE;
3490
0
        skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3491
0
        if( equBcwCost > curBestCost * skipTH )
3492
0
        {
3493
0
          break;
3494
0
        }
3495
3496
0
        if( m_pcEncCfg->m_BCW == 2 )
3497
0
        {
3498
0
          if( isEqualUni == true && ! m_pcEncCfg->m_picReordering )
3499
0
          {
3500
0
            break;
3501
0
          }
3502
0
          if( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) )
3503
0
          {
3504
0
            break;
3505
0
          }
3506
0
        }
3507
0
      }
3508
0
    }
3509
3510
0
    if (Do_OnceRes && (bestCostIMV != MAX_DOUBLE))
3511
0
    {
3512
0
      CodingStructure* CSCandBest = tempCSbest;
3513
0
      tempCS->initStructData(bestCS->currQP[partitioner.chType]);
3514
0
      tempCS->copyStructure(*CSCandBest, partitioner.chType, TREE_D);
3515
0
      tempCS->getPredBuf().copyFrom(tempCSbest->getPredBuf());
3516
0
      tempCS->dist = 0;
3517
0
      tempCS->fracBits = 0;
3518
0
      tempCS->cost = MAX_DOUBLE;
3519
3520
0
      xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, NULL);
3521
0
    }
3522
3523
0
    tempCS->initStructData(encTestMode.qp);
3524
0
  }
3525
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3526
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3527
0
}
3528
3529
void EncCu::xCalDebCost( CodingStructure &cs, Partitioner &partitioner )
3530
105k
{
3531
105k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_DEBLOCK_FILTER, &cs, partitioner.chType );
3532
105k
  if ( cs.slice->deblockingFilterDisable )
3533
0
  {
3534
0
    return;
3535
0
  }
3536
3537
105k
  const ChromaFormat format = cs.area.chromaFormat;
3538
105k
  CodingUnit*            cu = cs.getCU(partitioner.chType, partitioner.treeType);
3539
105k
  const Position    lumaPos = cu->Y().valid() ? cu->Y().pos() : recalcPosition( format, cu->chType, CH_L, cu->blocks[cu->chType].pos() );
3540
105k
  bool    topEdgeAvai = lumaPos.y > 0 && ((lumaPos.y % 4) == 0);
3541
105k
  bool   leftEdgeAvai = lumaPos.x > 0 && ((lumaPos.x % 4) == 0);
3542
3543
105k
  if( ! ( topEdgeAvai || leftEdgeAvai ))
3544
21.9k
  {
3545
21.9k
    return;
3546
21.9k
  }
3547
3548
83.1k
  ComponentID compStr = ( CU::isSepTree(*cu) && !isLuma( partitioner.chType ) ) ? COMP_Cb : COMP_Y;
3549
83.1k
  ComponentID compEnd = (( CU::isSepTree(*cu) && isLuma( partitioner.chType )) || cu->chromaFormat == VVENC_CHROMA_400 ) ? COMP_Y : COMP_Cr;
3550
83.1k
  const UnitArea currCsArea = clipArea( CS::getArea( cs, cs.area, partitioner.chType, partitioner.treeType ), *cs.picture );
3551
3552
83.1k
  PelStorage&  picDbBuf = m_dbBuffer; //th we could reduce the buffer size and do some relocate
3553
3554
  //deblock neighbour pixels
3555
83.1k
  const Size     lumaSize = cu->Y().valid() ? cu->Y().size() : recalcSize( format, cu->chType, CH_L, cu->blocks[cu->chType].size() );
3556
3557
83.1k
  int verOffset = lumaPos.y > 7 ? 8 : 4;
3558
83.1k
  int horOffset = lumaPos.x > 7 ? 8 : 4;
3559
3560
83.1k
  LoopFilter::calcFilterStrengths( *cu, true );
3561
3562
83.1k
  if( m_EDO == 2 && CS::isDualITree( cs ) && isLuma( partitioner.chType ) )
3563
45.9k
  {
3564
45.9k
    m_cLoopFilter.getMaxFilterLength( *cu, verOffset, horOffset );
3565
3566
45.9k
    if( 0== (verOffset + horOffset) )
3567
0
    {
3568
0
      return;
3569
0
    }
3570
3571
45.9k
    topEdgeAvai  &= verOffset != 0;
3572
45.9k
    leftEdgeAvai &= horOffset != 0;
3573
45.9k
  }
3574
3575
83.1k
  const UnitArea  areaTop  = UnitArea( format, Area( lumaPos.x,             lumaPos.y - verOffset, lumaSize.width, verOffset       ) );
3576
83.1k
  const UnitArea  areaLeft = UnitArea( format, Area( lumaPos.x - horOffset, lumaPos.y,             horOffset,      lumaSize.height ) );
3577
3578
203k
  for ( int compIdx = compStr; compIdx <= compEnd; compIdx++ )
3579
120k
  {
3580
120k
    ComponentID compId = (ComponentID)compIdx;
3581
3582
    //Copy current CU's reco to Deblock Pic Buffer
3583
120k
    const CompArea&  compArea = currCsArea.block( compId );
3584
120k
    CompArea         locArea  = compArea;
3585
120k
    locArea.x -= cu->blocks[compIdx].x;
3586
120k
    locArea.y -= cu->blocks[compIdx].y;
3587
120k
    PelBuf dbReco = picDbBuf.getBuf( locArea );
3588
120k
      PelBuf reco = cs.getRecoBuf( compId );
3589
120k
      dbReco.copyFrom( reco );
3590
    //left neighbour
3591
120k
    if ( leftEdgeAvai )
3592
83.6k
    {
3593
83.6k
      const CompArea&  compArea = areaLeft.block(compId);
3594
83.6k
      CompArea         locArea = compArea;
3595
83.6k
      locArea.x -= cu->blocks[compIdx].x;
3596
83.6k
      locArea.y -= cu->blocks[compIdx].y;
3597
83.6k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3598
83.6k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3599
83.6k
      }
3600
    //top neighbour
3601
120k
    if ( topEdgeAvai )
3602
87.3k
    {
3603
87.3k
      const CompArea&  compArea = areaTop.block( compId );
3604
87.3k
      CompArea         locArea = compArea;
3605
87.3k
      locArea.x -= cu->blocks[compIdx].x;
3606
87.3k
      locArea.y -= cu->blocks[compIdx].y;
3607
87.3k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3608
87.3k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3609
87.3k
      }
3610
120k
    }
3611
3612
83.1k
  ChannelType dbChType = CU::isSepTree(*cu) ? partitioner.chType : MAX_NUM_CH;
3613
3614
83.1k
  CHECK( CU::isSepTree(*cu) && !cu->Y().valid() && partitioner.chType == CH_L, "xxx" );
3615
3616
83.1k
  if( cu->Y() .valid() ) m_cLoopFilter.setOrigin( CH_L, cu->lumaPos() );
3617
83.1k
  if( cu->chromaFormat != VVENC_CHROMA_400 && cu->Cb().valid() ) m_cLoopFilter.setOrigin( CH_C, cu->chromaPos() );
3618
3619
  //deblock
3620
83.1k
  if( leftEdgeAvai )
3621
60.9k
  {
3622
60.9k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_VER, m_dbBuffer );
3623
60.9k
  }
3624
3625
83.1k
  if( topEdgeAvai )
3626
62.8k
  {
3627
62.8k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_HOR, m_dbBuffer );
3628
62.8k
  }
3629
3630
  //calculate difference between DB_before_SSE and DB_after_SSE for neighbouring CUs
3631
83.1k
  Distortion distBeforeDb = 0, distAfterDb = 0, distCur = 0;
3632
203k
  for (int compIdx = compStr; compIdx <= compEnd; compIdx++)
3633
120k
  {
3634
120k
    ComponentID compId = (ComponentID)compIdx;
3635
120k
    {
3636
120k
      CompArea compArea = currCsArea.block( compId );
3637
120k
      CompArea         locArea  = compArea;
3638
120k
      locArea.x -= cu->blocks[compIdx].x;
3639
120k
      locArea.y -= cu->blocks[compIdx].y;
3640
120k
      CPelBuf reco      = picDbBuf.getBuf( locArea );
3641
120k
      CPelBuf org       = cs.getOrgBuf( compId );
3642
120k
      distCur += xGetDistortionDb( cs, org, reco, compArea, false );
3643
120k
    }
3644
3645
120k
    if ( leftEdgeAvai )
3646
83.6k
    {
3647
83.6k
      const CompArea&  compArea = areaLeft.block( compId );
3648
83.6k
      CompArea         locArea  = compArea;
3649
83.6k
      locArea.x -= cu->blocks[compIdx].x;
3650
83.6k
      locArea.y -= cu->blocks[compIdx].y;
3651
83.6k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3652
83.6k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3653
0
      {
3654
0
        org = cs.picture->getFiltOrigBuf( compArea );
3655
0
      }
3656
83.6k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3657
83.6k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3658
83.6k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3659
83.6k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3660
83.6k
    }
3661
3662
120k
    if ( topEdgeAvai )
3663
87.3k
    {
3664
87.3k
      const CompArea&  compArea = areaTop.block( compId );
3665
87.3k
      CompArea         locArea  = compArea;
3666
87.3k
      locArea.x -= cu->blocks[compIdx].x;
3667
87.3k
      locArea.y -= cu->blocks[compIdx].y;
3668
87.3k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3669
87.3k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3670
0
      {
3671
0
        org = cs.picture->getFiltOrigBuf( compArea );
3672
0
      }
3673
87.3k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3674
87.3k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3675
87.3k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3676
87.3k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3677
87.3k
    }
3678
120k
  }
3679
3680
  //updated cost
3681
83.1k
  int64_t distTmp = distCur - cs.dist + distAfterDb - distBeforeDb;
3682
83.1k
  cs.costDbOffset = distTmp < 0 ? -m_cRdCost.calcRdCost( 0, -distTmp ) : m_cRdCost.calcRdCost( 0, distTmp );
3683
83.1k
}
3684
3685
Distortion EncCu::xGetDistortionDb(CodingStructure &cs, CPelBuf& org, CPelBuf& reco, const CompArea& compArea, bool beforeDb)
3686
462k
{
3687
462k
  Distortion dist;
3688
462k
  const ComponentID compID = compArea.compID;
3689
3690
462k
  dist = m_cRdCost.getDistPart(org, reco, cs.sps->bitDepths[toChannelType(compID)], compID, DF_SSE);
3691
462k
  return dist;
3692
462k
}
3693
3694
bool checkValidMvs( const CodingUnit& cu)
3695
0
{
3696
  // clang-format off
3697
0
  const int affineShiftTab[3] =
3698
0
  {
3699
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3700
0
    MV_PRECISION_INTERNAL - MV_PRECISION_SIXTEENTH,
3701
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT
3702
0
  };
3703
3704
0
  const int normalShiftTab[NUM_IMV_MODES] =
3705
0
  {
3706
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3707
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT,
3708
0
    MV_PRECISION_INTERNAL - MV_PRECISION_4PEL,
3709
0
    MV_PRECISION_INTERNAL - MV_PRECISION_HALF,
3710
0
  };
3711
  // clang-format on
3712
3713
0
  int mvShift;
3714
3715
0
  for (int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++)
3716
0
  {
3717
0
    if (cu.refIdx[refList] >= 0)
3718
0
    {
3719
0
      if (!cu.affine)
3720
0
      {
3721
0
        mvShift = normalShiftTab[cu.imv];
3722
0
        Mv signaledmvd(cu.mvd[refList][0].hor >> mvShift, cu.mvd[refList][0].ver >> mvShift);
3723
0
        if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3724
0
          return false;
3725
0
      }
3726
0
      else
3727
0
      {
3728
0
        for (int ctrlP = 1 + (cu.affineType == AFFINEMODEL_6PARAM); ctrlP >= 0; ctrlP--)
3729
0
        {
3730
0
          mvShift = affineShiftTab[cu.imv];
3731
0
          Mv signaledmvd(cu.mvd[refList][ctrlP].hor >> mvShift, cu.mvd[refList][ctrlP].ver >> mvShift);
3732
0
          if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3733
0
            return false;;
3734
0
        }
3735
0
      }
3736
0
    }
3737
0
  }
3738
  // avoid MV exceeding 18-bit dynamic range
3739
0
  const int maxMv = 1 << 17;
3740
0
  if (!cu.affine && !cu.mergeFlag)
3741
0
  {
3742
0
    if(    ( cu.refIdx[ 0 ] >= 0 && ( cu.mv[ 0 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 0 ][ 0 ].getAbsVer() >= maxMv ) )
3743
0
        || ( cu.refIdx[ 1 ] >= 0 && ( cu.mv[ 1 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 1 ][ 0 ].getAbsVer() >= maxMv ) ) )
3744
0
    {
3745
0
      return false;
3746
0
    }
3747
0
  }
3748
0
  if( cu.affine && !cu.mergeFlag )
3749
0
  {
3750
0
    for( int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++ )
3751
0
    {
3752
0
      if( cu.refIdx[ refList ] >= 0 )
3753
0
      {
3754
0
        for( int ctrlP = 1 + ( cu.affineType == AFFINEMODEL_6PARAM ); ctrlP >= 0; ctrlP-- )
3755
0
        {
3756
0
          if( cu.mv[ refList ][ ctrlP ].getAbsHor() >= maxMv || cu.mv[ refList ][ ctrlP ].getAbsVer() >= maxMv )
3757
0
          {
3758
0
            return false;
3759
0
          }
3760
0
        }
3761
0
      }
3762
0
    }
3763
0
  }
3764
0
  return true;
3765
0
}
3766
3767
3768
void EncCu::xEncodeInterResidual( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, int residualPass, bool* bestHasNonResi, double* equBcwCost )
3769
0
{
3770
0
  if( residualPass == 1 && encTestMode.lossless )
3771
0
  {
3772
0
    return;
3773
0
  }
3774
3775
0
  CodingUnit*            cu        = tempCS->getCU( partitioner.chType, partitioner.treeType );
3776
0
  double   bestCostInternal        = MAX_DOUBLE;
3777
3778
0
  if( !checkValidMvs( *cu ) )
3779
0
    return;
3780
3781
0
  double  currBestCost = MAX_DOUBLE;
3782
3783
  // For SBT
3784
0
  double     bestCost          = bestCS->cost;
3785
0
  double     bestCostBegin     = bestCS->cost;
3786
0
  const CodingUnit* prevBestCU = bestCS->getCU( partitioner.chType, partitioner.treeType );
3787
0
  uint8_t    prevBestSbt       = ( prevBestCU == nullptr ) ? 0 : prevBestCU->sbtInfo;
3788
0
  Distortion sbtOffDist        = 0;
3789
0
  bool       sbtOffRootCbf     = 0;
3790
0
  double     sbtOffCost        = MAX_DOUBLE;
3791
0
  uint8_t    currBestSbt       = 0;
3792
0
  uint8_t    histBestSbt       = MAX_UCHAR;
3793
0
  Distortion curPuSse          = MAX_DISTORTION;
3794
0
  uint8_t    numRDOTried       = 0;
3795
0
  bool       doPreAnalyzeResi  = false;
3796
0
  const bool mtsAllowed        =   tempCS->sps->MTSInter && cu->Y().maxDim() <= MTS_INTER_MAX_CU_SIZE;
3797
0
  const uint8_t sbtAllowed     = ( tempCS->pps->picWidthInLumaSamples < SBT_FAST64_WIDTH_THRESHOLD || m_pcEncCfg->m_SBT > 1 ) && cu->Y().maxDim() > 32 ? 0 : CU::checkAllowedSbt(*cu);
3798
3799
0
  if( sbtAllowed )
3800
0
  {
3801
    //SBT resolution-dependent fast algorithm: not try size-64 SBT in RDO for low-resolution sequences (now resolution below HD)
3802
0
    doPreAnalyzeResi = ( sbtAllowed || mtsAllowed ) && residualPass == 0;
3803
0
    m_cInterSearch.getBestSbt( tempCS, cu, histBestSbt, curPuSse, sbtAllowed, doPreAnalyzeResi, mtsAllowed );
3804
0
  }
3805
3806
0
  cu->skip    = false;
3807
0
  cu->sbtInfo = 0;
3808
3809
0
  const bool skipResidual = residualPass == 1;
3810
0
  if( skipResidual || histBestSbt == MAX_UCHAR || !CU::isSbtMode( histBestSbt ) )
3811
0
  {
3812
0
    m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3813
0
    xEncodeDontSplit( *tempCS, partitioner );
3814
0
    xCheckDQP       ( *tempCS, partitioner );
3815
3816
0
    if( NULL != bestHasNonResi && (bestCostInternal > tempCS->cost) )
3817
0
    {
3818
0
      bestCostInternal = tempCS->cost;
3819
0
      if( !cu->ciip )
3820
0
        *bestHasNonResi = !cu->rootCbf;
3821
0
    }
3822
3823
0
    if( cu->rootCbf == false )
3824
0
    {
3825
0
      if( cu->ciip )
3826
0
      {
3827
0
        tempCS->cost = MAX_DOUBLE;
3828
0
        tempCS->costDbOffset = 0;
3829
0
        return;
3830
0
      }
3831
0
    }
3832
0
    currBestCost = tempCS->cost;
3833
0
    if( sbtAllowed )
3834
0
    {
3835
0
      sbtOffCost    = tempCS->cost;
3836
0
      sbtOffDist    = tempCS->dist;
3837
0
      sbtOffRootCbf = cu->rootCbf;
3838
0
      currBestSbt   = cu->firstTU->mtsIdx[COMP_Y] > MTS_SKIP ? SBT_OFF_MTS : SBT_OFF_DCT;
3839
0
      numRDOTried  += mtsAllowed ? 2 : 1;
3840
0
    }
3841
3842
0
    DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3843
0
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3844
3845
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3846
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3847
0
  }
3848
3849
0
  if( sbtAllowed && ( m_pcEncCfg->m_SBT == 1 || sbtOffRootCbf ) )
3850
0
  {
3851
0
    bool swapped = false; // avoid unwanted data copy
3852
0
    uint8_t numSbtRdo = CU::numSbtModeRdo( sbtAllowed );
3853
    //early termination if all SBT modes are not allowed
3854
    //normative
3855
0
    if( !sbtAllowed || skipResidual )
3856
0
    {
3857
0
      numSbtRdo = 0;
3858
0
    }
3859
    //fast algorithm
3860
0
    if( ( histBestSbt != MAX_UCHAR && !CU::isSbtMode( histBestSbt ) ) || m_cInterSearch.getSkipSbtAll() )
3861
0
    {
3862
0
      numSbtRdo = 0;
3863
0
    }
3864
0
    if( bestCost != MAX_DOUBLE && sbtOffCost != MAX_DOUBLE )
3865
0
    {
3866
0
      double th = 1.07;
3867
0
      if( !( prevBestSbt == 0 || m_sbtCostSave[0] == MAX_DOUBLE ) )
3868
0
      {
3869
0
        assert( m_sbtCostSave[1] <= m_sbtCostSave[0] );
3870
0
        th *= ( m_sbtCostSave[0] / m_sbtCostSave[1] );
3871
0
      }
3872
0
      if( sbtOffCost > bestCost * th )
3873
0
      {
3874
0
        numSbtRdo = 0;
3875
0
      }
3876
0
    }
3877
0
    if( !sbtOffRootCbf && sbtOffCost != MAX_DOUBLE )
3878
0
    {
3879
0
      double th = Clip3( 0.05, 0.55, ( 27 - cu->qp ) * 0.02 + 0.35 );
3880
0
      if( sbtOffCost < m_cRdCost.calcRdCost( ( cu->lwidth() * cu->lheight() ) << SCALE_BITS, 0 ) * th )
3881
0
      {
3882
0
        numSbtRdo = 0;
3883
0
      }
3884
0
    }
3885
3886
0
    if( histBestSbt != MAX_UCHAR && numSbtRdo != 0 )
3887
0
    {
3888
0
      numSbtRdo = 1;
3889
0
      m_cInterSearch.initSbtRdoOrder( CU::getSbtMode( CU::getSbtIdx( histBestSbt ), CU::getSbtPos( histBestSbt ) ) );
3890
0
    }
3891
3892
0
    for( int sbtModeIdx = 0; sbtModeIdx < numSbtRdo; sbtModeIdx++ )
3893
0
    {
3894
0
      uint8_t sbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx );
3895
0
      uint8_t sbtIdx = CU::getSbtIdxFromSbtMode( sbtMode );
3896
0
      uint8_t sbtPos = CU::getSbtPosFromSbtMode( sbtMode );
3897
3898
      //fast algorithm (early skip, save & load)
3899
0
      if( histBestSbt == MAX_UCHAR )
3900
0
      {
3901
0
        uint8_t skipCode = m_cInterSearch.skipSbtByRDCost( cu->lwidth(), cu->lheight(), cu->mtDepth, sbtIdx, sbtPos, bestCS->cost, sbtOffDist, sbtOffCost, sbtOffRootCbf );
3902
0
        if( skipCode != MAX_UCHAR )
3903
0
        {
3904
0
          continue;
3905
0
        }
3906
3907
0
        if( sbtModeIdx > 0 )
3908
0
        {
3909
0
          uint8_t prevSbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx - 1 );
3910
          //make sure the prevSbtMode is the same size as the current SBT mode (otherwise the estimated dist may not be comparable)
3911
0
          if( CU::isSameSbtSize( prevSbtMode, sbtMode ) )
3912
0
          {
3913
0
            Distortion currEstDist = m_cInterSearch.getEstDistSbt( sbtMode );
3914
0
            Distortion prevEstDist = m_cInterSearch.getEstDistSbt( prevSbtMode );
3915
0
            if( currEstDist > prevEstDist * 1.15 )
3916
0
            {
3917
0
              continue;
3918
0
            }
3919
0
          }
3920
0
        }
3921
0
      }
3922
3923
      //init tempCS and TU
3924
0
      if( bestCost == bestCS->cost ) //The first EMT pass didn't become the bestCS, so we clear the TUs generated
3925
0
      {
3926
0
        tempCS->clearTUs();
3927
0
      }
3928
0
      else if( !swapped )
3929
0
      {
3930
0
        tempCS->initStructData( encTestMode.qp );
3931
0
        tempCS->copyStructure( *bestCS, partitioner.chType, partitioner.treeType );
3932
0
        tempCS->getPredBuf().copyFrom( bestCS->getPredBuf() );
3933
0
        bestCost = bestCS->cost;
3934
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3935
0
        swapped = true;
3936
0
      }
3937
0
      else
3938
0
      {
3939
0
        tempCS->clearTUs();
3940
0
        bestCost = bestCS->cost;
3941
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3942
0
      }
3943
3944
      //we need to restart the distortion for the new tempCS, the bit count and the cost
3945
0
      tempCS->dist     = 0;
3946
0
      tempCS->fracBits = 0;
3947
0
      tempCS->cost     = MAX_DOUBLE;
3948
0
      cu->skip         = false;
3949
3950
3951
      //set SBT info
3952
0
      cu->sbtInfo = (sbtPos << 4) + sbtIdx;
3953
3954
      //try residual coding
3955
0
      m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3956
0
      numRDOTried++;
3957
3958
0
      xEncodeDontSplit( *tempCS, partitioner );
3959
0
      xCheckDQP( *tempCS, partitioner );
3960
3961
0
      if( NULL != bestHasNonResi && ( bestCostInternal > tempCS->cost ) )
3962
0
      {
3963
0
        bestCostInternal = tempCS->cost;
3964
0
        if( !( cu->ciip ) )
3965
0
          *bestHasNonResi = !cu->rootCbf;
3966
0
      }
3967
3968
0
      if( tempCS->cost < currBestCost )
3969
0
      {
3970
0
        currBestSbt = cu->sbtInfo;
3971
0
        currBestCost = tempCS->cost;
3972
0
      }
3973
0
      else if( m_pcEncCfg->m_SBT > 2 )
3974
0
      {
3975
0
        sbtModeIdx = numSbtRdo;
3976
0
      }
3977
3978
0
      DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3979
0
      xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3980
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3981
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3982
0
    }
3983
3984
0
    if( bestCostBegin != bestCS->cost )
3985
0
    {
3986
0
      m_sbtCostSave[0] = sbtOffCost;
3987
0
      m_sbtCostSave[1] = currBestCost;
3988
0
    }
3989
3990
0
    if( histBestSbt == MAX_UCHAR && doPreAnalyzeResi && numRDOTried > 1 )
3991
0
    {
3992
0
      auto slsSbt = static_cast<CacheBlkInfoCtrl&>( m_modeCtrl );
3993
0
      int slShift = 4 + std::min( Log2( cu->lwidth() ) + Log2( cu->lheight() ), 9 );
3994
0
      slsSbt.saveBestSbt( cu->cs->area, (uint32_t)( curPuSse >> slShift ), currBestSbt );
3995
0
    }
3996
    
3997
0
    if( ETM_INTER_ME == encTestMode.type )
3998
0
    {
3999
0
      if( equBcwCost != NULL )
4000
0
      {
4001
0
        if( tempCS->cost < ( *equBcwCost ) && cu->BcwIdx == BCW_DEFAULT )
4002
0
        {
4003
0
          ( *equBcwCost ) = tempCS->cost;
4004
0
        }
4005
0
      }
4006
0
      else
4007
0
      {
4008
0
        CHECK( equBcwCost == NULL, "equBcwCost == NULL" );
4009
0
      }
4010
0
      if( tempCS->slice->checkLDC && !cu->imv && cu->BcwIdx != BCW_DEFAULT && tempCS->cost < m_bestBcwCost[1] )
4011
0
      {
4012
0
        if( tempCS->cost < m_bestBcwCost[0] )
4013
0
        {
4014
0
          m_bestBcwCost[1] = m_bestBcwCost[0];
4015
0
          m_bestBcwCost[0] = tempCS->cost;
4016
0
          m_bestBcwIdx[1] = m_bestBcwIdx[0];
4017
0
          m_bestBcwIdx[0] = cu->BcwIdx;
4018
0
        }
4019
0
        else
4020
0
        {
4021
0
          m_bestBcwCost[1] = tempCS->cost;
4022
0
          m_bestBcwIdx[1] = cu->BcwIdx;
4023
0
        }
4024
0
      }
4025
0
    }
4026
0
  }
4027
4028
0
  tempCS->cost = currBestCost;
4029
0
}
4030
4031
void EncCu::xEncodeDontSplit( CodingStructure &cs, Partitioner &partitioner )
4032
105k
{
4033
105k
  m_CABACEstimator->resetBits();
4034
4035
105k
  m_CABACEstimator->split_cu_mode( CU_DONT_SPLIT, cs, partitioner );
4036
105k
  if( partitioner.treeType == TREE_C )
4037
105k
    CHECK( m_CABACEstimator->getEstFracBits() != 0, "must be 0 bit" );
4038
4039
105k
  cs.fracBits += m_CABACEstimator->getEstFracBits(); // split bits
4040
105k
  cs.cost      = m_cRdCost.calcRdCost( cs.fracBits, cs.dist );
4041
105k
}
4042
4043
void EncCu::xReuseCachedResult( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner )
4044
0
{
4045
0
  EncTestMode cachedMode;
4046
4047
0
  if( ! m_modeCtrl.setCsFrom( *tempCS, cachedMode, partitioner ) )
4048
0
  {
4049
0
    THROW( "Should never happen!" );
4050
0
  }
4051
4052
0
  CodingUnit& cu = *tempCS->cus.front();
4053
0
  partitioner.setCUData( cu );
4054
4055
0
  if( CU::isIntra( cu ) )
4056
0
  {
4057
0
    if( isLuma( cu.chType ) )
4058
0
    {
4059
0
      cu.getMotionBuf().memset( -1 ); // clear motion buf
4060
0
    }
4061
0
    xReconIntraQT( cu );
4062
0
  }
4063
0
  else
4064
0
  {
4065
0
    xDeriveCUMV( cu );
4066
0
    xReconInter( cu );
4067
0
  }
4068
4069
0
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
4070
0
  m_CABACEstimator->resetBits();
4071
4072
0
  CUCtx cuCtx;
4073
0
  cuCtx.isDQPCoded = true;
4074
0
  cuCtx.isChromaQpAdjCoded = true;
4075
0
  m_CABACEstimator->coding_unit( cu, partitioner, cuCtx );
4076
4077
0
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
4078
0
  tempCS->cost     = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
4079
4080
0
  xEncodeDontSplit( *tempCS,         partitioner );
4081
0
  xCheckDQP       ( *tempCS,         partitioner );
4082
0
  xCheckBestMode  (  tempCS, bestCS, partitioner, cachedMode, m_EDO );
4083
0
}
4084
4085
uint64_t EncCu::xCalcPuMeBits( const CodingUnit &cu )
4086
0
{
4087
0
  CHECK( !cu.mergeFlag, "Should only be used for merge!" );
4088
0
  CHECK( CU::isIBC( cu ), "Shound not be used for IBC" );
4089
4090
0
  m_CABACEstimator->resetBits();
4091
0
  m_CABACEstimator->merge_flag(cu);
4092
0
  if( cu.mergeFlag )
4093
0
  {
4094
0
    m_CABACEstimator->merge_data( cu );
4095
0
  }
4096
0
  return m_CABACEstimator->getEstFracBits();
4097
0
}
4098
4099
double EncCu::xCalcDistortion(CodingStructure *&cur_CS, ChannelType chType, int BitDepth, int imv)
4100
0
{
4101
0
  const auto currDist1 = m_cRdCost.getDistPart(cur_CS->getOrgBuf( COMP_Y ), cur_CS->getPredBuf( COMP_Y ), BitDepth, COMP_Y, m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
4102
0
  unsigned int uiMvBits = 0;
4103
0
  unsigned imvShift = imv == IMV_HPEL ? 1 : (imv << 1);
4104
0
  const CodingUnit& cu = *cur_CS->getCU( chType, TREE_D);
4105
0
  if (cu.interDir != 2)
4106
0
  {
4107
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[0][0].hor, cu.mvd[0][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4108
0
  }
4109
0
  if (cu.interDir != 1)
4110
0
  {
4111
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[1][0].hor, cu.mvd[1][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4112
0
  }
4113
0
  return (double(currDist1) + (double)m_cRdCost.getCost(uiMvBits));
4114
0
}
4115
4116
int EncCu::xCheckMMVDCand(MmvdIdx& mmvdMergeCand, int& bestDir, int tempNum, double& bestCostOffset, double& bestCostMerge, double bestCostList )
4117
0
{
4118
0
  int baseIdx = mmvdMergeCand.val / MMVD_MAX_REFINE_NUM;
4119
0
  int CandCur = mmvdMergeCand.val - MMVD_MAX_REFINE_NUM * baseIdx;
4120
4121
0
  if( m_pcEncCfg->m_MMVD > 2 )
4122
0
  {
4123
0
    if( CandCur % 4 == 0 )
4124
0
    {
4125
0
      if( ( bestCostOffset >= bestCostMerge ) && ( CandCur >= 4 ) )
4126
0
      {
4127
0
        if( mmvdMergeCand.val > MMVD_MAX_REFINE_NUM )
4128
0
        {
4129
0
          return 2;
4130
0
        }
4131
0
        else
4132
0
        {
4133
0
          mmvdMergeCand.val = MMVD_MAX_REFINE_NUM;
4134
0
          if( tempNum == mmvdMergeCand.val )
4135
0
          {
4136
0
            return 2;
4137
0
          }
4138
0
        }
4139
0
      }
4140
      //reset
4141
0
      bestCostOffset = MAX_DOUBLE;
4142
0
      bestCostMerge  = bestCostList;
4143
0
    }
4144
0
  }
4145
4146
0
  if( mmvdMergeCand.val == MMVD_MAX_REFINE_NUM )
4147
0
  {
4148
0
    bestDir = 0;
4149
0
  }
4150
0
  if( CandCur >= 4 )
4151
0
  {
4152
0
    if( CandCur % 4 != bestDir )
4153
0
    {
4154
0
      return 1;
4155
0
    }
4156
0
  }
4157
0
  return 0;
4158
0
}
4159
4160
void EncCu::xInitLumaDeltaQpLUT()
4161
0
{
4162
  //TODO: to be discussed: add config parameters for these?
4163
0
  std::vector<int> defaultLumaLevelTodQp_QpChangePoints   =  { -3,  -2,  -1,   0,   1,   2,   3,   4,   5,   6 };
4164
0
  std::vector<int> defaultLumaLevelTodQp_LumaChangePoints =  {  0, 301, 367, 434, 501, 567, 634, 701, 767, 834 };
4165
  
4166
0
  int         lastDeltaQPValue = 0;
4167
0
  std::size_t nextSparseIndex = 0;
4168
0
  for( int index = 0; index < LUMA_LEVEL_TO_DQP_LUT_MAXSIZE; index++ )
4169
0
  {
4170
0
    while( nextSparseIndex < defaultLumaLevelTodQp_QpChangePoints.size() && index >= defaultLumaLevelTodQp_LumaChangePoints[nextSparseIndex] )
4171
0
    {
4172
0
      lastDeltaQPValue = defaultLumaLevelTodQp_QpChangePoints[nextSparseIndex];
4173
0
      nextSparseIndex++;
4174
0
    }
4175
0
    m_lumaLevelToDeltaQPLUT[index] = lastDeltaQPValue;
4176
0
  }
4177
0
}
4178
4179
MergeItem::MergeItem()
4180
154k
{
4181
4182
154k
}
4183
MergeItem::~MergeItem()
4184
154k
{
4185
4186
154k
}
4187
4188
void MergeItem::create( ChromaFormat chromaFormat, const Area &area )
4189
154k
{
4190
154k
  if( m_pelStorage.bufs.empty() )
4191
154k
  {
4192
154k
    m_pelStorage.create( chromaFormat, area );
4193
154k
    m_mvStorage .resize( area.area() >> ( MIN_CU_LOG2 << 1 ) );
4194
154k
  }
4195
4196
154k
  init();
4197
154k
}
4198
4199
void MergeItem::init()
4200
154k
{
4201
  // reset data
4202
154k
  cost        = MAX_DOUBLE;
4203
154k
  mergeIdx    = 0;
4204
154k
  bcwIdx      = 0;
4205
154k
  interDir    = 0;
4206
154k
  useAltHpelIf  = false;
4207
154k
  affineType    = AFFINEMODEL_4PARAM;
4208
154k
  mergeItemType = MergeItemType::NUM;
4209
4210
154k
  noBdofRefine  = false;
4211
154k
  noResidual    = false;
4212
4213
154k
  lumaPredReady   = false;
4214
154k
  chromaPredReady = false;
4215
154k
}
4216
4217
void MergeItem::importMergeInfo(const MergeCtx& mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu)
4218
0
{
4219
0
  mergeIdx      = _mergeIdx;
4220
0
  mergeItemType = _mergeItemType;
4221
4222
0
  if( mergeItemType != MergeItemType::GPM && mergeItemType != MergeItemType::MMVD )
4223
0
  {
4224
0
    mvField[REF_PIC_LIST_0][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_0];
4225
0
    mvField[REF_PIC_LIST_1][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_1];
4226
0
    interDir                   = mergeCtx.interDirNeighbours[mergeIdx];
4227
0
    bcwIdx                     = mergeCtx.BcwIdx            [mergeIdx];
4228
0
    useAltHpelIf               = mergeCtx.useAltHpelIf      [mergeIdx];
4229
0
  }
4230
4231
0
  switch( _mergeItemType )
4232
0
  {
4233
0
  case MergeItemType::REGULAR:
4234
0
  case MergeItemType::CIIP:
4235
0
    break;
4236
4237
0
  case MergeItemType::MMVD:
4238
0
  {
4239
0
    MmvdIdx candIdx;
4240
4241
0
    candIdx.val                = mergeIdx;
4242
0
    mvField[L0][0]             . setMvField( pu.mv[L0][0], pu.refIdx[0] );
4243
0
    mvField[L1][0]             . setMvField( pu.mv[L1][0], pu.refIdx[1] );
4244
0
    interDir                   = pu.interDir;
4245
0
    bcwIdx                     = pu.BcwIdx;
4246
0
    useAltHpelIf               = mergeCtx.useAltHpelIf[candIdx.pos.baseIdx];
4247
4248
0
    break;
4249
0
  }
4250
4251
0
  case MergeItemType::GPM:
4252
0
    mvField[L0][0]             . setMvField( Mv( 0, 0 ), -1 );
4253
0
    mvField[L1][0]             . setMvField( Mv( 0, 0 ), -1 );
4254
0
    bcwIdx                     = BCW_DEFAULT;
4255
0
    useAltHpelIf               = false;
4256
4257
0
    break;
4258
4259
0
  case MergeItemType::IBC:
4260
0
  default:
4261
0
    THROW( "Wrong merge item type" );
4262
0
  }
4263
4264
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4265
0
}
4266
4267
void MergeItem::importMergeInfo( const AffineMergeCtx &mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu )
4268
0
{
4269
0
  mergeIdx      = _mergeIdx;
4270
0
  mergeItemType = _mergeItemType;
4271
4272
0
  affineType    = mergeCtx.affineType         [mergeIdx];
4273
0
  interDir      = mergeCtx.interDirNeighbours [mergeIdx];
4274
0
  bcwIdx        = mergeCtx.BcwIdx             [mergeIdx];
4275
0
  useAltHpelIf  = false;
4276
4277
0
  switch( _mergeItemType )
4278
0
  {
4279
0
  case MergeItemType::SBTMVP:
4280
    // the pu motion was already generated preparing for IFP check (unconditional)
4281
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4282
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4283
4284
0
    break;
4285
4286
0
  case MergeItemType::AFFINE:
4287
    // the pu motion was already generated preparing for IFP check (unconditional)
4288
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4289
0
    mvField[L0][1] . setMvField( pu.mv[L0][1], pu.refIdx[L0] );
4290
0
    mvField[L0][2] . setMvField( pu.mv[L0][2], pu.refIdx[L0] );
4291
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4292
0
    mvField[L1][1] . setMvField( pu.mv[L1][1], pu.refIdx[L1] );
4293
0
    mvField[L1][2] . setMvField( pu.mv[L1][2], pu.refIdx[L1] );
4294
4295
0
    break;
4296
4297
0
  default:
4298
0
    THROW( "Wrong merge item type" );
4299
0
  }
4300
4301
  // the MI buf was already generated preparing for IFP check (unconditional)
4302
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4303
0
}
4304
4305
bool MergeItem::exportMergeInfo( CodingUnit &pu, bool forceNoResidual ) const
4306
0
{
4307
0
  pu.mergeFlag        = true;
4308
0
  pu.mmvdMergeFlag    = false;
4309
0
  pu.interDir         = interDir;
4310
0
  pu.mergeIdx         = mergeIdx;
4311
0
  pu.mergeType        = MRG_TYPE_DEFAULT_N;
4312
0
  pu.mv[REF_PIC_LIST_0][0]  = mvField[REF_PIC_LIST_0][0].mv;
4313
0
  pu.mv[REF_PIC_LIST_1][0]  = mvField[REF_PIC_LIST_1][0].mv;
4314
0
  pu.refIdx[REF_PIC_LIST_0] = mvField[REF_PIC_LIST_0][0].refIdx;
4315
0
  pu.refIdx[REF_PIC_LIST_1] = mvField[REF_PIC_LIST_1][0].refIdx;
4316
0
  pu.mvd[REF_PIC_LIST_0][0] = Mv();
4317
0
  pu.mvd[REF_PIC_LIST_1][0] = Mv();
4318
0
  pu.mvpIdx[REF_PIC_LIST_0] = NOT_VALID;
4319
0
  pu.mvpIdx[REF_PIC_LIST_1] = NOT_VALID;
4320
0
  pu.mvpNum[REF_PIC_LIST_0] = NOT_VALID;
4321
0
  pu.mvpNum[REF_PIC_LIST_1] = NOT_VALID;
4322
0
  pu.BcwIdx         = ( interDir == 3 ) ? bcwIdx : BCW_DEFAULT;
4323
0
  pu.mcControl      = 0;
4324
0
  pu.mmvdSkip       = false;
4325
0
  pu.affine         = false;
4326
0
  pu.affineType     = AFFINEMODEL_4PARAM;
4327
0
  pu.geo            = false;
4328
0
  pu.mtsFlag        = false;
4329
0
  pu.ciip           = false;
4330
0
  pu.imv            = ( !pu.geo && useAltHpelIf ) ? IMV_HPEL : IMV_OFF;
4331
0
  pu.mvRefine       = false;
4332
4333
0
  const bool resetCiip2Regular = mergeItemType == MergeItemType::CIIP && forceNoResidual;
4334
0
  MergeItemType updatedType    = resetCiip2Regular ? MergeItemType::REGULAR : mergeItemType;
4335
4336
0
  switch( updatedType )
4337
0
  {
4338
0
  case MergeItemType::REGULAR:
4339
0
    CU::restrictBiPredMergeCandsOne( pu );
4340
0
    break;
4341
4342
0
  case MergeItemType::CIIP:
4343
0
    CHECK( forceNoResidual, "Cannot force no residuals for CIIP" );
4344
0
    pu.ciip           = true;
4345
0
    pu.intraDir[CH_L] = PLANAR_IDX;
4346
0
    pu.intraDir[CH_C] = DM_CHROMA_IDX;
4347
0
    break;
4348
4349
0
  case MergeItemType::MMVD:
4350
0
    pu.mmvdMergeFlag    = true;
4351
0
    pu.mmvdMergeIdx.val = mergeIdx;
4352
0
    if( forceNoResidual )
4353
0
    {
4354
0
      pu.mmvdSkip       = true;
4355
0
    }
4356
0
    CU::restrictBiPredMergeCandsOne( pu );
4357
0
    break;
4358
4359
0
  case MergeItemType::SBTMVP:
4360
0
    pu.affine    = true;
4361
0
    pu.mergeType = MRG_TYPE_SUBPU_ATMVP;
4362
0
    break;
4363
4364
0
  case MergeItemType::AFFINE:
4365
0
    pu.affine     = true;
4366
0
    pu.affineType = affineType;
4367
0
    pu.mv[L0][0]  = mvField[L0][0].mv;
4368
0
    pu.mv[L1][0]  = mvField[L1][0].mv;
4369
0
    pu.mv[L0][1]  = mvField[L0][1].mv;
4370
0
    pu.mv[L1][1]  = mvField[L1][1].mv;
4371
0
    pu.mv[L0][2]  = mvField[L0][2].mv;
4372
0
    pu.mv[L1][2]  = mvField[L1][2].mv;
4373
0
    pu.refIdx[L0] = mvField[L0][0].refIdx;
4374
0
    pu.refIdx[L1] = mvField[L1][0].refIdx;
4375
0
    break;
4376
4377
0
  case MergeItemType::GPM:
4378
0
    pu.mergeIdx = -1;
4379
0
    pu.geo      = true;
4380
0
    pu.BcwIdx   = BCW_DEFAULT;
4381
0
    updateGpmIdx( mergeIdx, pu.geoSplitDir, pu.geoMergeIdx );
4382
0
    pu.imv      = IMV_OFF;
4383
0
    break;
4384
4385
0
  case MergeItemType::IBC:
4386
0
  default:
4387
0
    THROW( "Wrong merge item type" );
4388
0
  }
4389
4390
0
  pu.getMotionBuf().copyFrom( getMvBuf( pu ) );
4391
4392
0
  return resetCiip2Regular;
4393
0
}
4394
4395
MergeItemList::MergeItemList()
4396
19.2k
{
4397
4398
19.2k
}
4399
4400
MergeItemList::~MergeItemList()
4401
19.2k
{
4402
19.2k
  for( MergeItem* p : m_list )
4403
0
  {
4404
0
    delete p;
4405
0
  }
4406
19.2k
  m_list.clear();
4407
4408
19.2k
  for( MergeItem *p : m_mergeItems )
4409
154k
  {
4410
154k
    delete p;
4411
154k
  }
4412
19.2k
  m_mergeItems.clear();
4413
19.2k
}
4414
4415
void MergeItemList::init( size_t maxSize, size_t maxExtSize, ChromaFormat chromaFormat, SizeType ctuWidth, SizeType ctuHeight )
4416
19.2k
{
4417
19.2k
  CHECK( !m_mergeItems.empty() || !m_list.empty(), "MergeItemList already initialized" );
4418
4419
19.2k
  m_list      . reserve( maxSize + 1 ); // to avoid reallocation when inserting a new item
4420
19.2k
  m_mergeItems. reserve( maxSize + 1 );
4421
19.2k
  m_maxSize   = maxSize;
4422
19.2k
  m_maxExtSize= maxExtSize;
4423
19.2k
  m_numExt    = 0;
4424
4425
173k
  for( int i = 0; i < maxSize + m_maxExtSize; i++ )
4426
154k
  {
4427
154k
    MergeItem *p = new MergeItem;
4428
154k
    p->create( chromaFormat, Area{ 0, 0, ctuWidth, ctuHeight } );
4429
154k
    m_mergeItems.push_back( p );
4430
154k
  }
4431
19.2k
}
4432
4433
MergeItem *MergeItemList::allocateNewMergeItem()
4434
0
{
4435
0
  m_numExt++;
4436
0
  CHECK( m_mergeItems.empty(), "Missing merge items!" );
4437
0
  CHECK( m_numExt > m_maxExtSize, "Taking out more external items than specified during list allocation!" );
4438
0
  MergeItem *p = m_mergeItems.back();
4439
0
  m_mergeItems.pop_back();
4440
0
  p->init();
4441
0
  return p;
4442
0
}
4443
4444
bool MergeItemList::insertMergeItemToList( MergeItem *p )
4445
0
{
4446
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4447
4448
0
  m_numExt--;
4449
4450
0
  if( m_list.empty() )
4451
0
  {
4452
0
    m_list.push_back( p );
4453
0
  }
4454
0
  else if( m_list.size() == m_maxTrackingNum && p->cost >= m_list.back()->cost )
4455
0
  {
4456
0
    m_mergeItems.push_back( p );
4457
0
    return false;
4458
0
  }
4459
0
  else
4460
0
  {
4461
0
    if( m_list.size() == m_maxTrackingNum )
4462
0
    {
4463
0
      m_mergeItems.push_back( m_list.back() );
4464
0
      m_list      .pop_back();
4465
0
    }
4466
0
    auto it = std::find_if( m_list.begin(), m_list.end(), [&p]( const MergeItem *mi ) { return p->cost < mi->cost; } );
4467
0
    m_list.insert( it, p );
4468
0
  }
4469
4470
0
  return true;
4471
0
}
4472
4473
void MergeItemList::giveBackMergeItem( MergeItem *p )
4474
0
{
4475
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4476
4477
0
  m_numExt--;
4478
4479
0
  m_mergeItems.push_back( p );
4480
0
}
4481
4482
MergeItem *MergeItemList::getMergeItemInList( size_t index )
4483
0
{
4484
0
  return index < m_maxTrackingNum ? m_list[index] : nullptr;
4485
0
}
4486
4487
void MergeItemList::resetList( size_t maxTrackingNum )
4488
0
{
4489
0
  CHECK( maxTrackingNum > m_maxSize, "Not enough items allocated to track " << maxTrackingNum << " items" );
4490
4491
0
  for( auto p : m_list )
4492
0
  {
4493
0
    m_mergeItems.push_back( p );
4494
0
  }
4495
0
  m_list.clear  ();
4496
4497
0
  m_maxTrackingNum = maxTrackingNum;
4498
0
}
4499
4500
void MergeItemList::shrinkList( size_t reduceTo )
4501
0
{
4502
0
  CHECK( reduceTo > m_maxSize, "Not enough items allocated to track " << reduceTo << " items" );
4503
4504
0
  while( m_list.size() > reduceTo )
4505
0
  {
4506
0
    m_mergeItems.push_back( m_list.back() );
4507
0
    m_list      .pop_back();
4508
0
  }
4509
0
}
4510
4511
} // namespace vvenc
4512
4513
//! \}