Coverage Report

Created: 2026-09-02 06:43

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/EncoderLib/EncCu.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or
4
other Intellectual Property Rights other than the copyrights concerning
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     EncCu.cpp
45
    \brief    Coding Unit (CU) encoder class
46
*/
47
48
#include "EncCu.h"
49
#include "EncLib.h"
50
#include "Analyze.h"
51
#include "EncPicture.h"
52
#include "EncModeCtrl.h"
53
#include "BitAllocation.h"
54
#include "EncStage.h"
55
56
#include "CommonLib/dtrace_codingstruct.h"
57
#include "CommonLib/Picture.h"
58
#include "CommonLib/UnitTools.h"
59
#include "CommonLib/dtrace_buffer.h"
60
#include "CommonLib/TimeProfiler.h"
61
#include "CommonLib/SearchSpaceCounter.h"
62
63
#include <mutex>
64
#include <cmath>
65
#include <algorithm>
66
67
//! \ingroup EncoderLib
68
//! \{
69
70
namespace vvenc {
71
72
const MergeIdxPair EncCu::m_GeoModeTest[GEO_MAX_NUM_CANDS] = { MergeIdxPair{0, 1}, MergeIdxPair{1, 0}, MergeIdxPair{0, 2}, MergeIdxPair{1, 2}, MergeIdxPair{2, 0},
73
                                                               MergeIdxPair{2, 1}, MergeIdxPair{0, 3}, MergeIdxPair{1, 3}, MergeIdxPair{2, 3}, MergeIdxPair{3, 0},
74
                                                               MergeIdxPair{3, 1}, MergeIdxPair{3, 2}, MergeIdxPair{0, 4}, MergeIdxPair{1, 4}, MergeIdxPair{2, 4},
75
                                                               MergeIdxPair{3, 4}, MergeIdxPair{4, 0}, MergeIdxPair{4, 1}, MergeIdxPair{4, 2}, MergeIdxPair{4, 3},
76
                                                               MergeIdxPair{0, 5}, MergeIdxPair{1, 5}, MergeIdxPair{2, 5}, MergeIdxPair{3, 5}, MergeIdxPair{4, 5},
77
                                                               MergeIdxPair{5, 0}, MergeIdxPair{5, 1}, MergeIdxPair{5, 2}, MergeIdxPair{5, 3}, MergeIdxPair{5, 4} };
78
79
80
// Shape coefSquareCUsFasterFastMedium (2 x 5 x 2 x 2 x 2): preset (faster and fast + medium) x cusize x nspred x sptype x numcoef
81
82
const double EncCu::coefSquareCUsFasterFastMedium[2][5][2][2][2] = {
83
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
84
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
85
{{{-1.00000000, -1.00000000, }, {0.07848505, 0.00225808, }, },  {{-1.00000000, -1.00000000, }, {0.07509575, 0.00204789, }, },  },
86
{{{-1.00000000, -1.00000000, }, {0.10833051, 0.00053144, }, },  {{-1.00000000, -1.00000000, }, {0.08304352, 0.00142876, }, },  },
87
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
88
},
89
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
90
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
91
{{{0.06852235, 0.00388054, }, {0.09236045, 0.00084528, }, },  {{0.06955832, 0.00289679, }, {0.09598522, 0.00096187, }, },  },
92
{{{0.07268085, 0.00302796, }, {0.09323753, 0.00050996, }, },  {{0.06123618, 0.00471601, }, {0.09253389, 0.00046826, }, },  },
93
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
94
},
95
};
96
97
// Shape coefSquareCUsSlowSlower (2 x 5 x 2 x 5 x 2 x 2 x 2): preset (Slow + Slower) x cusize x nspred x qtdepth x mtdepth x sptype x numcoef
98
99
const double EncCu::coefSquareCUsSlowSlower[2][5][2][5][2][2][2] = {
100
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
101
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
102
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
103
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
104
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
105
},
106
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
107
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
108
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
109
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
110
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
111
},
112
},
113
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
114
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
115
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
116
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
117
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
118
},
119
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
120
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
121
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
122
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
123
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
124
},
125
},
126
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
127
{{{0.06057349, 0.00447803, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
128
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06948736, 0.00327774, }, {0.06396861, 0.00789923, }, }, },
129
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
130
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
131
},
132
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
133
{{{0.05526461, 0.00436703, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
134
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05661641, 0.00546358, }, {0.06680048, 0.00670604, }, }, },
135
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
136
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
137
},
138
},
139
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
140
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06556926, 0.00455697, }, {0.06803347, 0.00365396, }, }, },
141
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
142
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
143
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
144
},
145
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
146
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06394328, 0.00403533, }, {0.06688063, 0.00272684, }, }, },
147
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
148
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
149
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
150
},
151
},
152
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05533525, 0.00235007, }, {0.07184475, 0.00086154, }, }, },
153
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
154
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
155
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
156
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
157
},
158
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
159
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
160
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
161
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
162
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
163
},
164
},
165
},
166
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
167
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
168
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
169
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
170
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
171
},
172
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
173
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
174
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
175
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
176
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
177
},
178
},
179
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
180
{{{0.04770815, 0.00774597, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
181
{{{0.05509813, 0.00812349, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
182
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04719064, 0.01023887, }, {0.04641434, 0.01279769, }, }, },
183
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
184
},
185
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
186
{{{0.02777251, 0.00811989, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
187
{{{0.04848973, 0.00628111, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
188
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04924294, 0.00769725, }, {0.04491680, 0.01120645, }, }, },
189
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
190
},
191
},
192
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
193
{{{0.06856533, 0.00418949, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
194
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.07059085, 0.00348835, }, {0.06051544, 0.00821365, }, }, },
195
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
196
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
197
},
198
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
199
{{{0.07461087, 0.00229937, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
200
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05893714, 0.00513168, }, {0.05809189, 0.00772994, }, }, },
201
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
202
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
203
},
204
},
205
{{{{{0.06922080, 0.00274816, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
206
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06176300, 0.00374356, }, {0.06085891, 0.00361274, }, }, },
207
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
208
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
209
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
210
},
211
{{{{0.06611224, 0.00256748, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
212
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06364765, 0.00282819, }, {0.05888407, 0.00320947, }, }, },
213
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
214
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
215
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
216
},
217
},
218
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05637817, 0.00184637, }, {0.06937475, 0.00091860, }, }, },
219
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
220
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
221
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
222
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
223
},
224
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
225
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
226
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
227
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
228
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
229
},
230
},
231
},
232
};
233
234
// ====================================================================================================================
235
EncCu::EncCu()
236
18.3k
  : m_CtxCache          ( nullptr )
237
18.3k
  , m_globalCtuQpVector ( nullptr )
238
18.3k
  , m_wppMutex          ( nullptr )
239
18.3k
  , m_CABACEstimator    ( nullptr )
240
18.3k
{
241
18.3k
}
242
243
void EncCu::initPic( Picture* pic )
244
4.58k
{
245
4.58k
  m_cInterSearch.setSearchRange( pic->cs->slice, *m_pcEncCfg );
246
247
4.58k
  m_wppMutex = (m_pcEncCfg->m_numThreads > 0 ) ? &pic->wppMutex : nullptr;
248
4.58k
}
249
250
void EncCu::initSlice( const Slice* slice )
251
4.58k
{
252
4.58k
  m_cTrQuant.setLambdas( slice->getLambdas() );
253
4.58k
  m_cRdCost.setLambda( slice->getLambdas()[0], slice->sps->bitDepths );
254
4.58k
}
255
256
void EncCu::setCtuEncRsrc( CABACWriter* cabacEstimator, CtxCache* ctxCache, ReuseUniMv* pReuseUniMv, BlkUniMvInfoBuffer* pBlkUniMvInfoBuffer, AffineProfList* pAffineProfList, IbcBvCand* pCachedBvs )
257
3.59k
{
258
3.59k
  m_CABACEstimator = cabacEstimator;
259
3.59k
  m_CtxCache       = ctxCache;
260
3.59k
  m_cIntraSearch.setCtuEncRsrc( cabacEstimator, ctxCache );
261
3.59k
  m_cInterSearch.setCtuEncRsrc( cabacEstimator, ctxCache, pReuseUniMv, pBlkUniMvInfoBuffer, pAffineProfList, pCachedBvs );
262
3.59k
}
263
264
void EncCu::setUpLambda (Slice& slice, const double dLambda, const int iQP, const bool setSliceLambda, const bool saveUnadjusted)
265
8.17k
{
266
  // store lambda
267
8.17k
  m_cRdCost.setLambda( dLambda, slice.sps->bitDepths );
268
269
  // for RDO
270
  // in RdCost there is only one lambda because the luma and chroma bits are not separated, instead we weight the distortion of chroma.
271
8.17k
  double dLambdas[MAX_NUM_COMP] = { dLambda };
272
24.5k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
273
16.3k
  {
274
16.3k
    const ComponentID compID = ComponentID( compIdx );
275
16.3k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
276
16.3k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, iQP) + chromaQPOffset;
277
16.3k
    double tmpWeight         = pow( 2.0, ( iQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
278
16.3k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
279
16.3k
    {
280
16.3k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
281
16.3k
    }
282
16.3k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
283
16.3k
    dLambdas[compIdx] = dLambda / tmpWeight;
284
16.3k
  }
285
286
  // for RDOQ
287
8.17k
  m_cTrQuant.setLambdas( dLambdas );
288
289
  // for SAO, ALF
290
8.17k
  if (setSliceLambda)
291
4.58k
  {
292
4.58k
    slice.setLambdas( dLambdas );
293
4.58k
  }
294
8.17k
  if( saveUnadjusted )
295
8.17k
  {
296
8.17k
    m_cRdCost.saveUnadjustedLambda();
297
8.17k
  }
298
8.17k
}
299
300
void EncCu::updateLambda(const Slice& slice, const double ctuLambda, const int ctuQP, const int newQP, const bool saveUnadjusted)
301
3.11k
{
302
3.11k
  const double  corrFactor = pow (2.0, double (newQP - ctuQP) / 3.0);
303
3.11k
  const double  newLambda  = ctuLambda * corrFactor;
304
  
305
3.11k
  double newLambdas[MAX_NUM_COMP] = { newLambda };
306
9.33k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
307
6.22k
  {
308
6.22k
    const ComponentID compID = ComponentID( compIdx );
309
6.22k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
310
6.22k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, newQP) + chromaQPOffset;
311
6.22k
    double tmpWeight         = pow( 2.0, ( newQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
312
6.22k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
313
6.22k
    {
314
6.22k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
315
6.22k
    }
316
6.22k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
317
6.22k
    newLambdas[compIdx] = newLambda / tmpWeight;
318
6.22k
  }
319
320
3.11k
  m_cTrQuant.setLambdas ( newLambdas);
321
3.11k
  m_cRdCost.setLambda   ( newLambda, slice.sps->bitDepths);
322
323
3.11k
  if (saveUnadjusted)
324
3.11k
  {
325
3.11k
    m_cRdCost.saveUnadjustedLambda(); // TODO hlm: check if this actually improves the overall quality
326
3.11k
  }
327
3.11k
}
328
329
void EncCu::init( const VVEncCfg& encCfg, const SPS& sps, std::vector<int>* const globalCtuQpVector, Ctx* syncPicCtx, RateCtrl* pRateCtrl )
330
18.3k
{
331
18.3k
  DecCu::init( &m_cTrQuant, &m_cIntraSearch, &m_cInterSearch, encCfg.m_internChromaFormat );
332
18.3k
  m_cRdCost.create     ();
333
18.3k
  m_cRdCost.setCostMode( encCfg.m_costMode );
334
18.3k
  if ( encCfg.m_lumaLevelToDeltaQPEnabled == 1 )
335
0
  {
336
0
    m_cRdCost.setChromaFormat( encCfg.m_internChromaFormat );
337
0
    m_cRdCost.initLumaLevelToWeightTable( encCfg.m_internalBitDepth[ CH_L ] );
338
0
    xInitLumaDeltaQpLUT();
339
0
  }
340
341
18.3k
  m_modeCtrl.init     ( encCfg, &m_cRdCost );
342
18.3k
  m_cIntraSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_SortedPelUnitBufs, m_unitCache );
343
18.3k
  m_cInterSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_modeCtrl, m_cIntraSearch.getSaveCSBuf() );
344
18.3k
  m_cTrQuant.init     ( nullptr, encCfg.m_RDOQ, encCfg.m_useRDOQTS, false, true, encCfg.m_quantThresholdVal );
345
346
18.3k
  m_syncPicCtx = syncPicCtx;                         ///< context storage for state of contexts at the wavefront/WPP/entropy-coding-sync second CTU of tile-row used for estimation
347
18.3k
  m_pcRateCtrl = pRateCtrl;
348
349
  // Initialise scaling lists: The encoder will only use the SPS scaling lists. The PPS will never be marked present.
350
18.3k
  const int maxLog2TrDynamicRange[ MAX_NUM_CH ] = { sps.getMaxLog2TrDynamicRange(), sps.getMaxLog2TrDynamicRange() };
351
18.3k
  m_cTrQuant.getQuant()->setFlatScalingList( maxLog2TrDynamicRange, sps.bitDepths );
352
353
18.3k
  m_pcEncCfg       = &encCfg;
354
355
18.3k
  m_GeoCostList.init( encCfg.m_maxNumGeoCand );
356
357
18.3k
  unsigned      uiMaxSize    = encCfg.m_CTUSize;
358
18.3k
  ChromaFormat  chromaFormat = encCfg.m_internChromaFormat;
359
360
18.3k
  Area ctuArea = Area( 0, 0, uiMaxSize, uiMaxSize );
361
362
18.3k
  m_mergeItemList.init( encCfg.m_maxMergeRdCandNumTotal, m_pcEncCfg->m_Geo > 1 ? 3 : 1, chromaFormat, uiMaxSize, uiMaxSize );
363
364
238k
  for( int i = 0; i < maxCuDepth; i++ )
365
220k
  {
366
220k
    Area area = Area( 0, 0, uiMaxSize >> ( i >> 1 ), uiMaxSize >> ( ( i + 1 ) >> 1 ) );
367
368
220k
    if( area.width < (1 << MIN_CU_LOG2) || area.height < (1 << MIN_CU_LOG2) )
369
18.3k
    {
370
18.3k
      m_pTempCS[i] = m_pBestCS[i] = nullptr;
371
18.3k
      continue;
372
18.3k
    }
373
374
201k
    m_pTempCS[i] = new CodingStructure( m_unitCache, nullptr );
375
201k
    m_pBestCS[i] = new CodingStructure( m_unitCache, nullptr );
376
377
201k
    m_pTempCS[i]->createForSearch( chromaFormat, area );
378
201k
    m_pBestCS[i]->createForSearch( chromaFormat, area );
379
380
201k
    m_pOrgBuffer[i].create( chromaFormat, area );
381
201k
    m_pRspBuffer[i].create( CHROMA_400, area );
382
201k
  }
383
384
18.3k
  m_pTempCS2 = new CodingStructure( m_unitCache, nullptr );
385
18.3k
  m_pBestCS2 = new CodingStructure( m_unitCache, nullptr );
386
387
18.3k
  m_pTempCS2->createForSearch( chromaFormat, ctuArea );
388
18.3k
  m_pBestCS2->createForSearch( chromaFormat, ctuArea );
389
390
18.3k
  m_cuChromaQpOffsetIdxPlus1 = 0;
391
18.3k
  m_tempQpDiff = 0;
392
18.3k
  m_globalCtuQpVector = globalCtuQpVector;
393
394
18.3k
  m_SortedPelUnitBufs.create( chromaFormat, uiMaxSize, uiMaxSize );
395
396
275k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
397
256k
  {
398
256k
    m_aTmpStorageLCU[i].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
399
256k
  }
400
128k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
401
110k
  {
402
110k
    m_acMergeTmpBuffer[ui].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
403
110k
  }
404
405
18.3k
  const unsigned maxDepth = 2 * MAX_CU_SIZE_IDX;
406
18.3k
  m_CtxBuffer.resize( maxDepth );
407
18.3k
  m_CurrCtx = 0;
408
18.3k
  if( encCfg.m_EDO )
409
18.3k
    m_dbBuffer.create( chromaFormat, Area( 0, 0, uiMaxSize, uiMaxSize ), 0, 8 );
410
411
18.3k
  m_MergeSimpleFlag = 0;
412
18.3k
  m_tileIdx = 0;
413
18.3k
}
414
415
416
void EncCu::destroy()
417
18.3k
{
418
238k
  for( int i = 0; i < maxCuDepth; i++ )
419
220k
  {
420
220k
    if( m_pTempCS[i] )
421
201k
    {
422
201k
      m_pTempCS[i]->destroy();
423
201k
      delete m_pTempCS[i]; m_pTempCS[i] = nullptr;
424
201k
    }
425
426
220k
    if( m_pBestCS[i] )
427
201k
    {
428
201k
      m_pBestCS[i]->destroy();
429
201k
      delete m_pBestCS[i]; m_pBestCS[i] = nullptr;
430
201k
    }
431
432
220k
    m_pOrgBuffer[i].destroy();
433
220k
    m_pRspBuffer[i].destroy();
434
220k
  }
435
436
18.3k
  m_pTempCS2->destroy();
437
18.3k
  m_pBestCS2->destroy();
438
439
18.3k
  delete m_pTempCS2; m_pTempCS2 = nullptr;
440
18.3k
  delete m_pBestCS2; m_pBestCS2 = nullptr;
441
442
18.3k
  m_SortedPelUnitBufs.destroy();
443
444
275k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
445
256k
  {
446
256k
    m_aTmpStorageLCU[i].destroy();
447
256k
  }
448
128k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
449
110k
  {
450
110k
    m_acMergeTmpBuffer[ui].destroy();
451
110k
  }
452
453
454
18.3k
  m_dbBuffer.destroy();
455
18.3k
}
456
457
458
EncCu::~EncCu()
459
18.3k
{
460
18.3k
  destroy();
461
18.3k
}
462
463
// ====================================================================================================================
464
// Public member functions
465
// ====================================================================================================================
466
467
void EncCu::encodeCtu( Picture* pic, int (&prevQP)[MAX_NUM_CH], uint32_t ctuXPosInCtus, uint32_t ctuYPosInCtus )
468
3.59k
{
469
3.59k
  CodingStructure&     cs          = *pic->cs;
470
3.59k
  Slice*               slice       = cs.slice;
471
3.59k
  const PreCalcValues& pcv         = *cs.pcv;
472
473
#if ENABLE_MEASURE_SEARCH_SPACE
474
  if( ctuXPosInCtus == 0 && ctuYPosInCtus == 0 )
475
  {
476
    g_searchSpaceAcc.picW = pic->lwidth();
477
    g_searchSpaceAcc.picH = pic->lheight();
478
    g_searchSpaceAcc.addSlice( slice->isIntra(), slice->depth );
479
  }
480
481
#endif
482
3.59k
  const int ctuRsAddr                 = ctuYPosInCtus * pcv.widthInCtus + ctuXPosInCtus;
483
484
3.59k
  const Position pos (ctuXPosInCtus * pcv.maxCUSize, ctuYPosInCtus * pcv.maxCUSize);
485
3.59k
  const UnitArea ctuArea( cs.area.chromaFormat, Area( pos.x, pos.y, pcv.maxCUSize, pcv.maxCUSize ) );
486
3.59k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "ctu", ctuRsAddr ) );
487
488
3.59k
  const int tileXPosInCtus = cs.pps->tileColBd[cs.pps->ctuToTileCol[ctuXPosInCtus]];
489
3.59k
  const int tileYPosInCtus = cs.pps->tileRowBd[cs.pps->ctuToTileRow[ctuYPosInCtus]];
490
491
3.59k
  if( ( cs.slice->sliceType != VVENC_I_SLICE || cs.sps->IBC ) && ctuXPosInCtus == tileXPosInCtus )
492
2.03k
  {
493
2.03k
    const int tileRowId = cs.pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus );
494
2.03k
    cs.motionLutBuf[tileRowId].lut.resize( 0 );
495
2.03k
    cs.motionLutBuf[tileRowId].lutIbc.resize( 0 );
496
2.03k
  }
497
498
3.59k
  if( ( m_pcEncCfg->m_ensureWppBitEqual || m_pcEncCfg->m_entropyCodingSyncEnabled ) && ctuXPosInCtus == tileXPosInCtus )
499
2.03k
  {
500
2.03k
    m_CABACEstimator->initCtxModels( *slice );
501
502
2.03k
    if( m_pcEncCfg->m_entropyCodingSyncEnabled && ( ctuYPosInCtus > tileYPosInCtus ) )
503
0
    {
504
0
      m_CABACEstimator->getCtx() = m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus - 1 )];
505
0
    }
506
507
2.03k
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
508
2.03k
  }
509
1.56k
  else if( ctuXPosInCtus == tileXPosInCtus && ctuYPosInCtus == tileYPosInCtus )
510
0
  {
511
0
    m_CABACEstimator->initCtxModels( *slice );
512
0
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
513
0
  }
514
515
3.59k
  xCompressCtu( cs, ctuArea, ctuRsAddr, prevQP );
516
517
3.59k
  m_CABACEstimator->resetBits();
518
3.59k
  m_CABACEstimator->coding_tree_unit( cs, ctuArea, prevQP, ctuRsAddr, true, true );
519
520
  // Store probabilities of second CTU in line into buffer - used only if wavefront-parallel-processing is enabled.
521
3.59k
  if( ctuXPosInCtus == tileXPosInCtus && m_pcEncCfg->m_entropyCodingSyncEnabled )
522
0
  {
523
0
    m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus )] = m_CABACEstimator->getCtx();
524
0
  }
525
526
3.59k
  DTRACE_AREA_CRC( g_trace_ctx, D_CRC, cs, ctuArea );
527
3.59k
}
528
529
// ====================================================================================================================
530
// Protected member functions
531
// ====================================================================================================================
532
533
void EncCu::xCompressCtu( CodingStructure& cs, const UnitArea& area, const unsigned ctuRsAddr, const int prevQP[] )
534
3.59k
{
535
3.59k
  m_tileIdx = cs.pps->getTileIdx( area.lumaPos() );
536
537
3.59k
  m_modeCtrl.initCTUEncoding( *cs.slice, m_tileIdx );
538
539
  // init the partitioning manager
540
3.59k
  Partitioner *partitioner = &m_partitioner;
541
3.59k
  partitioner->initCtu( area, CH_L, *cs.slice );
542
  
543
3.59k
  const Position& lumaPos = area.lumaPos();
544
3.59k
  const bool leftSameTile  = lumaPos.x == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset(-1, 0) );
545
3.59k
  const bool aboveSameTile = lumaPos.y == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset( 0,-1) );
546
3.59k
  m_EDO = (!m_pcEncCfg->m_tileParallelCtuEnc || (leftSameTile && aboveSameTile)) ? m_pcEncCfg->m_EDO : 0;
547
  
548
3.59k
  if( m_pcEncCfg->m_IBCMode )
549
3.59k
  {
550
3.59k
    m_cInterSearch.resetCtuRecordIBC();
551
3.59k
  }
552
553
  // init current context pointer
554
3.59k
  m_CurrCtx = m_CtxBuffer.data();
555
556
3.59k
  PelStorage* orgBuffer = &m_pOrgBuffer[0];
557
3.59k
  PelStorage* rspBuffer = &m_pRspBuffer[0];
558
3.59k
  CodingStructure *tempCS =  m_pTempCS [0];
559
3.59k
  CodingStructure *bestCS =  m_pBestCS [0];
560
3.59k
  cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
561
3.59k
  cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
562
3.59k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_COMPRESS_CU, tempCS, CH_L );
563
564
  // copy the relevant area
565
3.59k
  UnitArea clippedArea = clipArea( partitioner->currArea(), cs.area );
566
3.59k
  CPelUnitBuf org = cs.picture->getFilteredOrigBuffer().valid() ? cs.picture->getFiltOrigBuf( clippedArea ) : cs.picture->getOrigBuf( clippedArea );
567
3.59k
  tempCS->getOrgBuf( clippedArea ).copyFrom( org );
568
3.59k
  tempCS->currQP[CH_L] = bestCS->currQP[CH_L] =
569
3.59k
  tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
570
3.59k
  tempCS->prevQP[CH_L] = bestCS->prevQP[CH_L] = prevQP[CH_L];
571
572
3.59k
  xCompressCU( tempCS, bestCS, *partitioner );
573
  // all signals were already copied during compression if the CTU was split - at this point only the structures are copied to the top level CS
574
  
575
  // Ensure that a coding was found
576
  // Selected mode's RD-cost must be not MAX_DOUBLE.
577
3.59k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
578
3.59k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
579
3.59k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
580
581
3.59k
  if ( m_wppMutex ) m_wppMutex->lock();
582
583
3.59k
  cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
584
585
3.59k
  if ( m_wppMutex ) m_wppMutex->unlock();
586
587
3.59k
  if( CS::isDualITree( cs ) && isChromaEnabled( cs.pcv->chrFormat ) )
588
3.59k
  {
589
3.59k
    m_CABACEstimator->getCtx() = m_CurrCtx->start;
590
591
3.59k
    partitioner->initCtu( area, CH_C, *cs.slice );
592
593
3.59k
    cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
594
3.59k
    cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
595
3.59k
    tempCS->currQP[CH_C] = bestCS->currQP[CH_C] =
596
3.59k
    tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
597
3.59k
    tempCS->prevQP[CH_C] = bestCS->prevQP[CH_C] = prevQP[CH_C];
598
599
3.59k
    xCompressCU( tempCS, bestCS, *partitioner );
600
    
601
    // Ensure that a coding was found
602
    // Selected mode's RD-cost must be not MAX_DOUBLE.
603
3.59k
    CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
604
3.59k
    CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
605
3.59k
    CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
606
607
3.59k
    if ( m_wppMutex ) m_wppMutex->lock();
608
609
3.59k
    cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
610
611
3.59k
    if ( m_wppMutex ) m_wppMutex->unlock();
612
3.59k
  }
613
614
  // reset context states and uninit context pointer
615
3.59k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
616
3.59k
  m_CurrCtx                  = 0;
617
3.59k
}
618
619
620
621
bool EncCu::xCheckBestMode( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const bool useEDO )
622
268k
{
623
268k
  bool bestCSUpdated = false;
624
625
268k
  if( !tempCS->cus.empty() )
626
163k
  {
627
163k
    if( tempCS->cus.size() == 1 )
628
123k
    {
629
123k
      const CodingUnit& cu = *tempCS->cus.front();
630
123k
      CHECK( cu.skip && !cu.mergeFlag, "Skip flag without a merge flag is not allowed!" );
631
123k
    }
632
633
163k
    DTRACE_BEST_MODE( tempCS, bestCS, m_cRdCost.getLambda(true), useEDO );
634
635
163k
    if( m_modeCtrl.useModeResult( encTestMode, tempCS, partitioner, useEDO ) )
636
147k
    {
637
147k
      std::swap( tempCS, bestCS );
638
      // store temp best CI for next CU coding
639
147k
      m_CurrCtx->best = m_CABACEstimator->getCtx();
640
147k
      bestCSUpdated = true;
641
147k
    }
642
163k
  }
643
644
  // reset context states
645
268k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
646
268k
  return bestCSUpdated;
647
648
268k
}
649
650
void xCheckFastCuChromaSplitting( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner&  partitioner, ComprCUCtx& cuECtx )
651
0
{
652
0
  const uint32_t uiLPelX = tempCS->area.Cb().lumaPos().x;
653
0
  const uint32_t uiTPelY = tempCS->area.Cb().lumaPos().y;
654
655
0
  int lumaw = 0, lumah = 0;
656
0
  bool splitver      = true;
657
0
  bool splithor      = true;
658
0
  bool qtSplitChroma = true;
659
660
0
  if( partitioner.isSepTree( *tempCS ) && isChroma( partitioner.chType ) )
661
0
  {
662
0
    Position lumaRefPos( uiLPelX, uiTPelY );
663
0
    CodingUnit* colLumaCu = bestCS->lumaCS->getCU( lumaRefPos, CH_L, TREE_D );
664
665
0
    if( colLumaCu )
666
0
    {
667
0
      lumah = colLumaCu->Y().height;
668
0
      lumaw = colLumaCu->Y().width;
669
0
    }
670
0
  }
671
0
  else
672
0
  {
673
0
    return;
674
0
  }
675
676
0
  if( partitioner.getImplicitSplit( *tempCS ) != CU_DONT_SPLIT ) return;
677
678
0
  const CPelBuf orgCb = tempCS->getOrgBuf( COMP_Cb );
679
0
  const CPelBuf orgCr = tempCS->getOrgBuf( COMP_Cr );
680
681
0
  int th1 = FCBP_TH1;
682
683
0
  if( ( lumaw >> getChannelTypeScaleX( CH_C, tempCS->area.chromaFormat ) ) == orgCb.width )
684
0
  {
685
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
686
0
    {
687
0
      splitver      = false;
688
0
      qtSplitChroma = false;
689
0
    }
690
0
  }
691
692
0
  if( ( lumah >> getChannelTypeScaleY( CH_C, tempCS->area.chromaFormat ) ) == orgCb.height )
693
0
  {
694
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
695
0
    {
696
0
      splithor      = false;
697
0
      qtSplitChroma = false;
698
0
    }
699
0
  }
700
701
0
  cuECtx.doHorChromaSplit = splithor;
702
0
  cuECtx.doVerChromaSplit = splitver;
703
0
  cuECtx.doQtChromaSplit  = qtSplitChroma;
704
705
0
  if( orgCb.width == orgCb.height )
706
0
  {
707
0
    int varh_cb, varv_cb;
708
0
    int varh_cr, varv_cr;
709
710
0
    orgCb.calcVarianceSplit( orgCb, orgCb.width, varh_cb, varv_cb );
711
0
    orgCr.calcVarianceSplit( orgCr, orgCr.width, varh_cr, varv_cr );
712
713
0
    if( ( varh_cr*FCBP_TH2 < varv_cr * 100 ) && ( varh_cb*FCBP_TH2 < varv_cb * 100 ) )
714
0
    {
715
0
      cuECtx.doVerChromaSplit = false;
716
0
    }
717
0
    else if( ( varv_cr*FCBP_TH2 < varh_cr * 100 ) && ( varv_cb*FCBP_TH2 < varh_cb * 100 ) )
718
0
    {
719
0
      cuECtx.doHorChromaSplit = false;
720
0
    }
721
0
  }
722
0
}
723
724
void EncCu::xCompressCU( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner )
725
123k
{
726
123k
  const Area& lumaArea = tempCS->area.Y();
727
728
123k
  Slice&   slice      = *tempCS->slice;
729
123k
  const PPS &pps      = *tempCS->pps;
730
123k
  const SPS &sps      = *tempCS->sps;
731
123k
  const uint32_t uiLPelX  = tempCS->area.Y().lumaPos().x;
732
123k
  const uint32_t uiTPelY  = tempCS->area.Y().lumaPos().y;
733
123k
  const bool isBimEnabled = (m_pcEncCfg->m_blockImportanceMapping && !bestCS->picture->m_picShared->m_ctuBimQpOffset.empty());
734
735
123k
  m_modeCtrl.initBlk( tempCS->area, slice.pic->poc );
736
123k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
737
738
123k
  if ((m_pcEncCfg->m_usePerceptQPA || isBimEnabled || m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1) && ( pps.useDQP || m_pcEncCfg->m_maxDeltaQP == 0 ) && isLuma (partitioner.chType) && partitioner.currQgEnable())
739
6.70k
  {
740
6.70k
    const PreCalcValues &pcv = *pps.pcv;
741
6.70k
    Picture* const pic = bestCS->picture;
742
6.70k
    const uint32_t ctuRsAddr = getCtuAddr (partitioner.currQgPos, pcv);
743
744
6.70k
    if (partitioner.currSubdiv == 0) // CTU-level QP adaptation
745
3.59k
    {
746
3.59k
      if (m_pcEncCfg->m_usePerceptQPA)
747
3.59k
      {
748
3.59k
        if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
749
0
        {
750
0
          m_tempQpDiff = pic->ctuAdaptedQP[ctuRsAddr] - BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
751
0
        }
752
753
3.59k
        int clippedMaxDQP = 0;
754
755
3.59k
        if ((!slice.isIntra()) && (pcv.maxCUSize > 64) && // sub-CTU QPA behavior - Museum fix
756
0
            (uiLPelX + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceWidth)) &&
757
0
            (uiTPelY + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceHeight)))
758
0
        {
759
0
          const uint32_t h = lumaArea.height >> 1;
760
0
          const uint32_t w = lumaArea.width  >> 1;
761
0
          const int adQPTL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
762
0
          const int adQPTR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
763
0
          const int adQPBL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
764
0
          const int adQPBR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
765
766
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
767
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = std::min (std::min (adQPTL, adQPTR), std::min (adQPBL, adQPBR));
768
769
0
          if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
770
0
          {
771
0
            if ((m_globalCtuQpVector->size() > ctuRsAddr) && (slice.TLayer == 0) && // last CTU row of non-Intra key-frame
772
0
                (m_pcEncCfg->m_IntraPeriod == 2 * m_pcEncCfg->m_GOPSize) && (ctuRsAddr >= pcv.widthInCtus) && (uiTPelY + pcv.maxCUSize > m_pcEncCfg->m_PadSourceHeight))
773
0
            {
774
0
              m_globalCtuQpVector->at (ctuRsAddr) = m_globalCtuQpVector->at (ctuRsAddr - pcv.widthInCtus); // copy the pumping reducing QP offset from the top CTU neighbor
775
0
              tempCS->currQP[partitioner.chType] = tempCS->baseQP =
776
0
              bestCS->currQP[partitioner.chType] = bestCS->baseQP = tempCS->baseQP - m_globalCtuQpVector->at (ctuRsAddr);
777
0
            }
778
0
            tempCS->currQP[partitioner.chType] = tempCS->baseQP =
779
0
            bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
780
0
          }
781
0
        }
782
3.59k
        else
783
3.59k
        {
784
3.59k
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
785
3.59k
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = pic->ctuAdaptedQP[ctuRsAddr];
786
3.59k
        }
787
788
3.59k
        int deltaQP       = slice.sliceQp - tempCS->baseQP;
789
3.59k
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
790
0
        {
791
0
          clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
792
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
793
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
794
0
        }
795
796
        //the ctuAdaptedQP correspones with the ctuQpaLambda
797
3.59k
        setUpLambda (slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], false, true);
798
3.59k
      }
799
0
      else // isBimEnabled without QPA
800
0
      {
801
0
        const int baseQp  = tempCS->baseQP;
802
0
        int adaptedLumaQP = 0;
803
0
        int lumaQPoffset  = 0;
804
        
805
0
        if( m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1 )
806
0
        {
807
0
          uint32_t meanLuma = 0;
808
0
          if( ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples ) || ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) )
809
0
          {
810
0
            int wdth = ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples )   ? pps.picWidthInLumaSamples - uiLPelX  : lumaArea.width;
811
0
            int hght = ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) ? pps.picHeightInLumaSamples - uiTPelY : lumaArea.height;
812
            
813
0
            int sum = 0;
814
0
            for( int y = 0; y < hght; y++ )
815
0
            {
816
0
              for( int x = 0; x < wdth; x++ )
817
0
              {
818
0
                sum += tempCS->getOrgBuf().Y().at( x, y );
819
0
              }
820
0
            }
821
0
            double avg = (double) sum / ( wdth*hght );
822
0
            meanLuma = (uint32_t)( avg + 0.5 );
823
0
          }
824
0
          else
825
0
          {
826
0
            meanLuma = tempCS->getOrgBuf().Y().getAvg();
827
0
          }
828
829
0
          int lumaBD     = m_pcEncCfg->m_internalBitDepth[ CH_L ];
830
0
          int lumaIdxOrg = Clip3<int>( 0, int( 1 << lumaBD ) - 1, meanLuma );
831
0
          int lumaIdx    = lumaBD < 10 ? lumaIdxOrg << (10 - lumaBD) : lumaBD > 10 ? lumaIdxOrg >> (lumaBD - 10) : lumaIdxOrg;
832
0
          lumaQPoffset   = m_lumaLevelToDeltaQPLUT[lumaIdx];
833
0
          adaptedLumaQP  = Clip3( -sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset );
834
0
        }
835
        
836
0
        if( isBimEnabled )
837
0
        {
838
0
          const unsigned bimQpSize = (unsigned) bestCS->picture->m_picShared->m_ctuBimQpOffset.size();
839
0
          uint32_t ctuAddr         = ctuRsAddr;
840
841
0
          double avgBimOffset = 0.0;
842
843
0
          if (bimQpSize != pcv.sizeInCtus) // re-calculate correct address of BIM CTU QP offset
844
0
          {
845
0
            const unsigned bimCtuSize  = m_pcEncCfg->m_bimCtuSize;
846
0
            const unsigned bimCtuWidth = (pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize;
847
848
0
            ctuAddr = getCtuAddrFromCtuSize (partitioner.currQgPos, Log2 (bimCtuSize), bimCtuWidth);
849
0
            CHECK (ctuAddr >= bimQpSize, "ctuAddr exceeds size of m_ctuBimQpOffset");
850
851
0
            int maxY = m_pcEncCfg->m_CTUSize / bimCtuSize;
852
0
            int maxX = m_pcEncCfg->m_CTUSize / bimCtuSize;
853
0
            if( ( partitioner.currQgPos.y + maxY * bimCtuSize ) > pcv.lumaHeight )
854
0
            {
855
0
              int reduceHeight = ( partitioner.currQgPos.y + ( maxY - 1 ) * bimCtuSize ) - pcv.lumaHeight;
856
0
              int reduceY = ( reduceHeight + bimCtuSize ) / bimCtuSize;
857
0
              maxY = maxY - reduceY;
858
0
            }
859
0
            if( ( partitioner.currQgPos.x + maxX * bimCtuSize ) > pcv.lumaWidth )
860
0
            {
861
0
              int reduceWidth = ( partitioner.currQgPos.x + ( maxX - 1 ) * bimCtuSize ) - pcv.lumaWidth;
862
0
              int reduceX = ( reduceWidth + bimCtuSize ) / bimCtuSize;
863
0
              maxX = maxX - reduceX;
864
0
            }
865
0
            for (int y = 0; y < maxY * bimCtuSize; y = y + bimCtuSize)
866
0
            {
867
0
              for (int x = 0; x < maxX * bimCtuSize; x = x + bimCtuSize)
868
0
              {
869
0
                int theBlockId = ctuAddr + (y / bimCtuSize) * ((pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize) + (x / bimCtuSize);
870
0
                avgBimOffset += pic->m_picShared->m_ctuBimQpOffset[theBlockId];
871
0
              }
872
0
            }
873
0
            if (maxY * maxX > 0)
874
0
            {
875
0
              avgBimOffset = avgBimOffset / (double)(maxY * maxX);
876
0
            }
877
0
            else
878
0
            {
879
0
              avgBimOffset = 0.0;
880
0
            }
881
0
          }
882
0
          else
883
0
          {
884
0
            avgBimOffset = pic->m_picShared->m_ctuBimQpOffset[ctuAddr];
885
0
          }
886
0
          int intAvgBimOffset = avgBimOffset >= 0 ? (int)(0.5 + avgBimOffset) : (int)(-0.5 + avgBimOffset);
887
0
          adaptedLumaQP       = Clip3 (-sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset + intAvgBimOffset );
888
0
        }
889
        
890
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP =
891
0
        bestCS->currQP[partitioner.chType] = bestCS->baseQP = adaptedLumaQP;
892
        
893
0
        int deltaQP         = slice.sliceQp - adaptedLumaQP;
894
0
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
895
0
        {
896
0
          int clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
897
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
898
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
899
0
        }
900
901
0
        updateLambda (slice, slice.getLambdas()[0], baseQp, adaptedLumaQP, true);
902
0
      }
903
3.59k
    }
904
3.11k
    else if (m_pcEncCfg->m_usePerceptQPA && slice.isIntra()) // currSubdiv 2 - use sub-CTU QPA
905
3.11k
    {
906
3.11k
      CHECK ((partitioner.currArea().lwidth() >= pcv.maxCUSize) || (partitioner.currArea().lheight() >= pcv.maxCUSize), "sub-CTU delta-QP error");
907
3.11k
      tempCS->currQP[partitioner.chType] = tempCS->baseQP = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
908
909
3.11k
      if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
910
0
      {
911
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
912
0
      }
913
914
3.11k
      int backupBaseQP  = tempCS->baseQP;
915
3.11k
      int deltaQP       = slice.sliceQp - tempCS->baseQP;
916
3.11k
      int clippedMaxDQP = 0;
917
3.11k
      if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
918
0
      {
919
0
        clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
920
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = clippedMaxDQP;
921
0
      }
922
3.11k
      updateLambda( slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], backupBaseQP, true );
923
3.11k
    }
924
6.70k
  }
925
926
123k
  if (partitioner.currQtDepth == 0)
927
7.18k
  {
928
7.18k
    m_MergeSimpleFlag = 0;
929
7.18k
  }
930
123k
  m_modeCtrl.initCULevel( partitioner, *tempCS, m_MergeSimpleFlag );
931
123k
  m_sbtCostSave[0] = m_sbtCostSave[1] = MAX_DOUBLE;
932
933
123k
  m_CurrCtx->start = m_CABACEstimator->getCtx();
934
935
123k
  m_cuChromaQpOffsetIdxPlus1 = 0;
936
937
123k
  if( slice.chromaQpAdjEnabled && partitioner.currQgChromaEnable() )
938
0
  {
939
    // TODO M0133 : double check encoder decisions with respect to chroma QG detection and actual encode
940
0
    int cuChromaQpOffsetSubdiv = slice.isIntra() ? slice.picHeader->cuChromaQpOffsetSubdivIntra : slice.picHeader->cuChromaQpOffsetSubdivInter;
941
0
    int lgMinCuSize = sps.log2MinCodingBlockSize +
942
0
      std::max<int>(0, floorLog2(sps.CTUSize) - sps.log2MinCodingBlockSize - int((cuChromaQpOffsetSubdiv + 1) / 2));
943
0
    m_cuChromaQpOffsetIdxPlus1 = ( ( uiLPelX >> lgMinCuSize ) + ( uiTPelY >> lgMinCuSize ) ) % ( pps.chromaQpOffsetListLen + 1 );
944
0
  }
945
946
123k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cux", uiLPelX ) );
947
123k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuy", uiTPelY ) );
948
123k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuw", tempCS->area.lwidth() ) );
949
123k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuh", tempCS->area.lheight() ) );
950
123k
  DTRACE( g_trace_ctx, D_COMMON, "@(%4d,%4d) [%2dx%2d]\n", tempCS->area.lx(), tempCS->area.ly(), tempCS->area.lwidth(), tempCS->area.lheight() );
951
952
123k
  if( tempCS->slice->checkLDC )
953
123k
  {
954
123k
    m_bestBcwCost[0] = m_bestBcwCost[1] = std::numeric_limits<double>::max();
955
123k
    m_bestBcwIdx[0] = m_bestBcwIdx[1] = -1;
956
123k
  }
957
958
123k
  m_cInterSearch.resetSavedAffineMotion();
959
123k
  {
960
123k
    const ComprCUCtx &cuECtx      = *m_modeCtrl.comprCUCtx;
961
123k
    const CodingStructure& cs     = *tempCS;
962
123k
    const PartSplit implicitSplit = partitioner.getImplicitSplit( cs );
963
123k
    const bool isBoundary         = implicitSplit != CU_DONT_SPLIT;
964
123k
    const bool lossless           = false;
965
123k
    int qp                        = cs.baseQP;
966
967
#if ENABLE_MEASURE_SEARCH_SPACE
968
    if( !isBoundary )
969
    {
970
      g_searchSpaceAcc.addPartition( partitioner.currArea(), partitioner.isSepTree( *tempCS ) ? partitioner.chType : MAX_NUM_CH );
971
    }
972
973
#endif
974
123k
    if( ! isBoundary )
975
77.9k
    {
976
77.9k
      if (pps.useDQP && partitioner.isSepTree (*tempCS) && isChroma (partitioner.chType))
977
54.0k
      {
978
54.0k
        const ChromaFormat chromaFm = tempCS->area.chromaFormat;
979
54.0k
        const Position chromaCentral (tempCS->area.Cb().chromaPos().offset (tempCS->area.Cb().chromaSize().width >> 1, tempCS->area.Cb().chromaSize().height >> 1));
980
54.0k
        const Position lumaRefPos (chromaCentral.x << getChannelTypeScaleX (CH_C, chromaFm), chromaCentral.y << getChannelTypeScaleY (CH_C, chromaFm));
981
54.0k
        const CodingUnit* colLumaCu = bestCS->lumaCS->getCU (lumaRefPos, CH_L, TREE_D);
982
        // update qp
983
54.0k
        qp = colLumaCu->qp;
984
54.0k
      }
985
986
77.9k
      m_cIntraSearch.reset();
987
988
77.9k
      bool isReuseCU = m_modeCtrl.isReusingCuValid( cs, partitioner, qp );
989
990
77.9k
      bool checkIbc = m_pcEncCfg->m_IBCMode && bestCS->picture->useIBC && (partitioner.chType == CH_L);
991
77.9k
      if ((m_pcEncCfg->m_IBCFastMethod>3) && (cs.area.lwidth() * cs.area.lheight()) > (16 * 16))
992
0
      {
993
0
        checkIbc = false;
994
0
      }
995
77.9k
      if( isReuseCU )
996
0
      {
997
0
        xReuseCachedResult( tempCS, bestCS, partitioner );
998
0
      }
999
77.9k
      else
1000
77.9k
      {
1001
        // add first pass modes
1002
77.9k
        if ( !slice.isIntra() && !slice.isIRAP() && !( cs.area.lwidth() == 4 && cs.area.lheight() == 4 ) && !partitioner.isConsIntra() )
1003
0
        {
1004
          // add inter modes
1005
0
          EncTestMode encTestModeSkip = { ETM_MERGE_SKIP, ETO_STANDARD, qp, lossless };
1006
0
          if (m_modeCtrl.tryMode(encTestModeSkip, cs, partitioner))
1007
0
          {
1008
0
            xCheckRDCostUnifiedMerge(tempCS, bestCS, partitioner, encTestModeSkip);
1009
1010
0
            CodingUnit* cu = bestCS->getCU(partitioner.chType, partitioner.treeType);
1011
0
            if (cu)
1012
0
              cu->mmvdSkip = cu->skip == false ? false : cu->mmvdSkip;
1013
0
          }
1014
0
          EncTestMode encTestMode = { ETM_INTER_ME, ETO_STANDARD, qp, lossless };
1015
0
          if (m_modeCtrl.tryMode(encTestMode, cs, partitioner))
1016
0
          {
1017
0
            xCheckRDCostInter(tempCS, bestCS, partitioner, encTestMode);
1018
0
          }
1019
1020
0
          if (m_pcEncCfg->m_AMVRspeed)
1021
0
          {
1022
0
            double bestIntPelCost = MAX_DOUBLE;
1023
1024
0
            EncTestMode encTestMode = {ETM_INTER_IMV, ETO_STANDARD, qp, lossless};
1025
0
            if( m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1026
0
            {
1027
0
              const bool skipAltHpelIF = ( int( ( encTestMode.opts & ETO_IMV ) >> ETO_IMV_SHIFT ) == 4 ) && ( bestIntPelCost > 1.25 * bestCS->cost );
1028
0
              if (!skipAltHpelIF)
1029
0
              {
1030
0
                xCheckRDCostInterIMV(tempCS, bestCS, partitioner, encTestMode );
1031
0
              }
1032
0
            }
1033
0
          }
1034
0
        }
1035
1036
77.9k
        if (checkIbc && !partitioner.isConsInter())
1037
23.9k
        {
1038
23.9k
          EncTestMode encTestModeIBCMerge = { ETM_IBC_MERGE, ETO_STANDARD, qp, lossless };
1039
23.9k
          if ((m_pcEncCfg->m_IBCFastMethod < 4) && (partitioner.chType == CH_L) && m_modeCtrl.tryMode(encTestModeIBCMerge, cs, partitioner))
1040
23.9k
          {
1041
23.9k
            xCheckRDCostIBCModeMerge2Nx2N(tempCS, bestCS, partitioner, encTestModeIBCMerge);
1042
23.9k
          }
1043
1044
23.9k
          EncTestMode encTestModeIBC = { ETM_IBC, ETO_STANDARD, qp, lossless };
1045
23.9k
          if (m_modeCtrl.tryMode(encTestModeIBC, cs, partitioner))
1046
23.9k
          {
1047
23.9k
            xCheckRDCostIBCMode(tempCS, bestCS, partitioner, encTestModeIBC);
1048
23.9k
          }
1049
23.9k
        }
1050
77.9k
        if( m_EDO && bestCS->cost != MAX_DOUBLE )
1051
21.3k
        {
1052
21.3k
          xCalDebCost(*bestCS, partitioner);
1053
21.3k
        }
1054
1055
        // add intra modes
1056
77.9k
        EncTestMode encTestMode( {ETM_INTRA, ETO_STANDARD, qp, lossless} );
1057
77.9k
        if( !partitioner.isConsInter() && m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1058
77.9k
        {
1059
77.9k
          xCheckRDCostIntra( tempCS, bestCS, partitioner, encTestMode );
1060
77.9k
        }
1061
77.9k
      } // reusing cu
1062
1063
77.9k
      m_modeCtrl.beforeSplit( partitioner );
1064
1065
77.9k
      if (cuECtx.bestCS && ((cuECtx.bestCostNoImv == (MAX_DOUBLE * .5) || cuECtx.isReusingCu) && !slice.isIntra()) )
1066
0
      {
1067
0
        m_cInterSearch.loadGlobalUniMvs( lumaArea, *pps.pcv );
1068
0
      }
1069
1070
77.9k
      if (!cs.slice->isIntra() && (partitioner.chType == CH_L) && ( m_pcEncCfg->m_qtbttSpeedUpMode & 2) && (partitioner.currQtDepth < 3) && bestCS->cus.size())
1071
0
      {
1072
0
        int flagDbefore = (bestCS->cus[0]->mergeFlag && !bestCS->cus[0]->mmvdMergeFlag && !bestCS->cus[0]->ispMode && !bestCS->cus[0]->geo) ? 1 : 0;
1073
0
        if (partitioner.currQtDepth == 0)
1074
0
        {
1075
0
          m_MergeSimpleFlag = flagDbefore;
1076
0
        }
1077
0
        else
1078
0
        {
1079
0
          int markFlag = (partitioner.currQtDepth == 1) ? 1 : 3;
1080
0
          m_MergeSimpleFlag = (flagDbefore << partitioner.currQtDepth) | (m_MergeSimpleFlag & markFlag);
1081
0
        }
1082
0
      }
1083
77.9k
    } //boundary
1084
1085
123k
    if( ( m_pcEncCfg->m_IntraPeriod == 1 ) && ( partitioner.chType == CH_C ) )
1086
0
    {
1087
0
      xCheckFastCuChromaSplitting( tempCS, bestCS, partitioner, *m_modeCtrl.comprCUCtx );
1088
0
    }
1089
    //////////////////////////////////////////////////////////////////////////
1090
    // split modes
1091
123k
    EncTestMode lastTestMode;
1092
1093
123k
    if( cuECtx.qtBeforeBt )
1094
19.7k
    {
1095
19.7k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1096
19.7k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1097
12.8k
      {
1098
12.8k
        lastTestMode = encTestMode;
1099
12.8k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1100
12.8k
      }
1101
19.7k
    }
1102
1103
123k
    if( partitioner.canSplit( CU_HORZ_SPLIT, cs ) )
1104
65.4k
    {
1105
      // add split modes
1106
65.4k
      EncTestMode encTestMode( { ETM_SPLIT_BT_H, ETO_STANDARD, qp, false } );
1107
65.4k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1108
65.4k
      {
1109
65.4k
        lastTestMode = encTestMode;
1110
65.4k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1111
65.4k
      }
1112
65.4k
    }
1113
1114
123k
    if( partitioner.canSplit( CU_VERT_SPLIT, cs ) )
1115
62.5k
    {
1116
      // add split modes
1117
62.5k
      EncTestMode encTestMode( { ETM_SPLIT_BT_V, ETO_STANDARD, qp, false } );
1118
62.5k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1119
62.5k
      {
1120
62.5k
        lastTestMode = encTestMode;
1121
62.5k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1122
62.5k
      }
1123
62.5k
    }
1124
1125
123k
    if( partitioner.canSplit( CU_TRIH_SPLIT, cs ) )
1126
27.6k
    {
1127
      // add split modes
1128
27.6k
      EncTestMode encTestMode( { ETM_SPLIT_TT_H, ETO_STANDARD, qp, false } );
1129
27.6k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1130
1.47k
      {
1131
1.47k
        lastTestMode = encTestMode;
1132
1.47k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1133
1.47k
      }
1134
27.6k
    }
1135
1136
123k
    if( partitioner.canSplit( CU_TRIV_SPLIT, cs ) )
1137
21.7k
    {
1138
      // add split modes
1139
21.7k
      EncTestMode encTestMode( { ETM_SPLIT_TT_V, ETO_STANDARD, qp, false } );
1140
21.7k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1141
1.12k
      {
1142
1.12k
        lastTestMode = encTestMode;
1143
1.12k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1144
1.12k
      }
1145
21.7k
    }
1146
1147
123k
    if( !cuECtx.qtBeforeBt )
1148
103k
    {
1149
103k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1150
103k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1151
25.8k
      {
1152
25.8k
        lastTestMode = encTestMode;
1153
25.8k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1154
25.8k
      }
1155
103k
    }
1156
123k
  }
1157
1158
123k
  if( bestCS->cus.empty() )
1159
0
  {
1160
0
    m_modeCtrl.finishCULevel( partitioner );
1161
0
    return;
1162
0
  }
1163
1164
  //////////////////////////////////////////////////////////////////////////
1165
  // Finishing CU
1166
  // set context states
1167
123k
  m_CABACEstimator->getCtx() = m_CurrCtx->best;
1168
1169
  // QP from last processed CU for further processing
1170
  //copy the qp of the last non-chroma CU
1171
123k
  int numCUInThisNode = (int)bestCS->cus.size();
1172
123k
  if( numCUInThisNode > 1 && bestCS->cus.back()->chType == CH_C && !CS::isDualITree( *bestCS ) )
1173
0
  {
1174
0
    CHECK( bestCS->cus[numCUInThisNode-2]->chType != CH_L, "wrong chType" );
1175
0
    bestCS->prevQP[partitioner.chType] = bestCS->cus[numCUInThisNode-2]->qp;
1176
0
  }
1177
123k
  else
1178
123k
  {
1179
123k
    bestCS->prevQP[partitioner.chType] = bestCS->cus.back()->qp;
1180
123k
  }
1181
123k
  if( ( !slice.isIntra() || slice.sps->IBC )
1182
123k
    && partitioner.chType == CH_L
1183
49.2k
    && bestCS->cus.size() == 1 && ( bestCS->cus.back()->predMode == MODE_INTER || bestCS->cus.back()->predMode == MODE_IBC )
1184
0
    && bestCS->area.Y() == (*bestCS->cus.back()).Y() )
1185
0
  {
1186
0
    const CodingUnit& cu = *bestCS->cus.front();
1187
0
    bool isIbcSmallBlk = CU::isIBC(cu) && (cu.lwidth() * cu.lheight() <= 16);
1188
0
    if (!cu.affine && !cu.geo && !isIbcSmallBlk)
1189
0
    {
1190
0
      const MotionInfo &mi = cu.getMotionInfo();
1191
0
      HPMVInfo hMi( mi, ( mi.interDir() == 3 ) ? cu.BcwIdx : BCW_DEFAULT, cu.imv == IMV_HPEL, CU::isIBC( cu ) );
1192
0
      cu.cs->addMiToLut( CU::isIBC( cu ) ? cu.cs->motionLut.lutIbc : cu.cs->motionLut.lut, hMi );
1193
0
    }
1194
0
  }
1195
1196
123k
  m_modeCtrl.finishCULevel( partitioner );
1197
123k
  if( m_cIntraSearch.getSaveCuCostInSCIPU() && bestCS->cus.size() == 1 )
1198
0
  {
1199
0
    m_cIntraSearch.saveCuAreaCostInSCIPU( Area( partitioner.currArea().lumaPos(), partitioner.currArea().lumaSize() ), bestCS->cost );
1200
0
  }
1201
1202
  // Assert if Best prediction mode is NONE
1203
  // Selected mode's RD-cost must be not MAX_DOUBLE.
1204
123k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
1205
123k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
1206
123k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
1207
123k
}
1208
1209
1210
void EncCu::xCheckModeSplit(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1211
169k
{
1212
169k
  const ModeType modeTypeParent  = partitioner.modeType;
1213
169k
  const TreeType treeTypeParent  = partitioner.treeType;
1214
169k
  const ChannelType chTypeParent = partitioner.chType;
1215
1216
169k
  int signalModeConsVal = CS::signalModeCons( *tempCS, partitioner.currArea(), getPartSplit(encTestMode), modeTypeParent);
1217
169k
  int numRoundRdo = signalModeConsVal == LDT_MODE_TYPE_SIGNAL ? 2 : 1;
1218
169k
  bool skipInterPass = false;
1219
338k
  for( int i = 0; i < numRoundRdo; i++ )
1220
169k
  {
1221
    //change cons modes
1222
169k
    if( signalModeConsVal == LDT_MODE_TYPE_SIGNAL )
1223
0
    {
1224
0
      CHECK( numRoundRdo != 2, "numRoundRdo shall be 2 - [LDT_MODE_TYPE_SIGNAL]" );
1225
0
      partitioner.modeType = (i == 0) ? MODE_TYPE_INTER : MODE_TYPE_INTRA;
1226
0
    }
1227
169k
    else if( signalModeConsVal == LDT_MODE_TYPE_INFER )
1228
0
    {
1229
0
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INFER]" );
1230
0
      partitioner.modeType = MODE_TYPE_INTRA;
1231
0
    }
1232
169k
    else if( signalModeConsVal == LDT_MODE_TYPE_INHERIT )
1233
169k
    {
1234
169k
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INHERIT]" );
1235
169k
      partitioner.modeType = modeTypeParent;
1236
169k
    }
1237
1238
    //for lite intra encoding fast algorithm, set the status to save inter coding info
1239
169k
    if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType == MODE_TYPE_INTER )
1240
0
    {
1241
0
      m_cIntraSearch.setSaveCuCostInSCIPU( true );
1242
0
      m_cIntraSearch.setNumCuInSCIPU( 0 );
1243
0
    }
1244
169k
    else if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType != MODE_TYPE_INTER )
1245
169k
    {
1246
169k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1247
169k
      if( partitioner.modeType == MODE_TYPE_ALL )
1248
169k
      {
1249
169k
        m_cIntraSearch.setNumCuInSCIPU( 0 );
1250
169k
      }
1251
169k
    }
1252
1253
169k
    xCheckModeSplitInternal( tempCS, bestCS, partitioner, encTestMode, modeTypeParent, skipInterPass );
1254
    //recover cons modes
1255
169k
    partitioner.modeType = modeTypeParent;
1256
169k
    partitioner.treeType = treeTypeParent;
1257
169k
    partitioner.chType = chTypeParent;
1258
169k
    if( modeTypeParent == MODE_TYPE_ALL )
1259
169k
    {
1260
169k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1261
169k
      if( numRoundRdo == 2 && partitioner.modeType == MODE_TYPE_INTRA )
1262
0
      {
1263
0
        m_cIntraSearch.initCuAreaCostInSCIPU();
1264
0
      }
1265
169k
    }
1266
169k
    if( skipInterPass )
1267
0
    {
1268
0
      break;
1269
0
    }
1270
169k
  }
1271
169k
}
1272
1273
void EncCu::xCheckModeSplitInternal(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const ModeType modeTypeParent, bool& skipInterPass )
1274
169k
{
1275
169k
  const int qp                     = encTestMode.qp;
1276
169k
  const int oldPrevQp              = tempCS->prevQP[partitioner.chType];
1277
169k
  const auto oldMotionLut          = tempCS->motionLut;
1278
                                   
1279
169k
  const PartSplit split            = getPartSplit( encTestMode );
1280
169k
  const ModeType  modeTypeChild    = partitioner.modeType;
1281
1282
169k
  CHECK( !( split == CU_QUAD_SPLIT || split == CU_HORZ_SPLIT || split == CU_VERT_SPLIT
1283
169k
         || split == CU_TRIH_SPLIT || split == CU_TRIV_SPLIT ), "invalid split type" );
1284
1285
169k
  tempCS->initStructData( qp );
1286
1287
169k
  m_CABACEstimator->getCtx()       = m_CurrCtx->start;
1288
1289
169k
  const uint16_t split_ctx_size    = Ctx::SplitFlag.size() + Ctx::SplitQtFlag.size() + Ctx::SplitHvFlag.size() + Ctx::Split12Flag.size() + Ctx::ModeConsFlag.size();
1290
169k
  const TempCtx  ctxSplitFlags     ( m_CtxCache, SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), m_CABACEstimator->getCtx() ) );
1291
1292
169k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
1293
169k
  m_CABACEstimator->resetBits           ();
1294
169k
  m_CABACEstimator->split_cu_mode       ( split, *tempCS, partitioner );
1295
169k
  partitioner     . modeType            = modeTypeParent;
1296
169k
  m_CABACEstimator->mode_constraint     ( split, *tempCS, partitioner, modeTypeChild );
1297
169k
  partitioner     . modeType            = modeTypeChild;
1298
1299
169k
  const int64_t splitBits   = m_CABACEstimator->getEstFracBits();
1300
1301
169k
  const bool chromaNotSplit = modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTRA;
1302
169k
  const bool isChromaTooBig = isChromaEnabled( tempCS->pps->pcv->chrFormat ) && tempCS->area.Y().maxDim() > tempCS->sps->getMaxTbSize();
1303
169k
  bool       skipSplitTest  = chromaNotSplit && isChromaTooBig;
1304
1305
169k
  if( !skipSplitTest )
1306
169k
  {
1307
169k
    double         a = -1, b = -1;
1308
169k
    const unsigned w       = partitioner.currArea().lwidth();
1309
169k
    const unsigned h       = partitioner.currArea().lheight();
1310
169k
    const bool contextCond = w == h && tempCS->slice->sliceType == VVENC_B_SLICE && isLuma( partitioner.chType ) && m_pcEncCfg->m_splitCostThrParamId >= 0 && m_pcEncCfg->m_splitCostThrParamId <= 3;
1311
1312
169k
    if( contextCond )
1313
0
    {
1314
0
      uint8_t nsPredInd = m_modeCtrl.comprCUCtx->bestNsPredMode.type == ETM_INTRA;
1315
0
      uint8_t szInd     = getLog2( w ) - 3;
1316
0
      uint8_t splitInd  = split == CU_QUAD_SPLIT ? 1 : 0;
1317
0
      if ( m_pcEncCfg->m_splitCostThrParamId <= 1 )
1318
0
      {
1319
0
        a = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][0];
1320
0
        b = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][1];
1321
0
      }
1322
0
      else
1323
0
      {
1324
0
        uint8_t mtInd = (partitioner.currMtDepth == 0);
1325
0
        a = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][0];
1326
0
        b = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][1];
1327
0
      }
1328
0
    }
1329
1330
169k
    if( a > -1 && b > -1 )
1331
0
    {
1332
0
      const double bestNsCost    = m_modeCtrl.comprCUCtx->bestCostBeforeSplit == MAX_DOUBLE ? -1 : m_modeCtrl.comprCUCtx->bestCostBeforeSplit;
1333
0
      const double factor        = 1.0 + b * exp( a * qp );
1334
0
      const double predSplitCost = bestNsCost / factor + splitBits;
1335
0
      skipSplitTest              = bestNsCost >= 0 && predSplitCost >= bestNsCost;
1336
0
    }
1337
169k
    else
1338
169k
    {
1339
169k
      int numChild = 3;
1340
169k
      if( split == CU_VERT_SPLIT || split == CU_HORZ_SPLIT ) numChild--;
1341
41.2k
      else if( split == CU_QUAD_SPLIT ) numChild++;
1342
1343
18.4E
      int64_t approxBits = m_pcEncCfg->m_qtbttSpeedUp > 0 ? numChild << SCALE_BITS : 0;
1344
1345
169k
      const double factor     = ( tempCS->currQP[partitioner.chType] > 30                              ? 1.1  : 1.075 ) +
1346
18.4E
                                (   m_pcEncCfg->m_qtbttSpeedUp > 0                                     ? 0.01 : 0.0   ) +
1347
169k
                                ( ( m_pcEncCfg->m_qtbttSpeedUp > 0 && isChroma( partitioner.chType ) ) ? 0.2  : 0.0   );
1348
       
1349
169k
      const double baseCost   = bestCS->cost + bestCS->costDbOffset;
1350
169k
      const double predCost   = baseCost / factor + splitBits + approxBits;
1351
169k
      skipSplitTest           = predCost >= baseCost;
1352
169k
    }
1353
169k
  }
1354
1355
169k
  if( skipSplitTest )
1356
105k
  {
1357
105k
    m_CABACEstimator->getCtx() = SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), ctxSplitFlags );
1358
105k
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1359
105k
    return;
1360
105k
  }
1361
1362
63.8k
  if( partitioner.treeType == TREE_D )
1363
63.8k
  {
1364
63.8k
    if( chromaNotSplit )
1365
0
    {
1366
0
      CHECK( partitioner.chType != CH_L, "chType must be luma" );
1367
0
      partitioner.treeType = TREE_L;
1368
0
    }
1369
63.8k
    else
1370
63.8k
    {
1371
63.8k
      partitioner.treeType = TREE_D;
1372
63.8k
    }
1373
63.8k
  }
1374
1375
63.8k
  partitioner.splitCurrArea( split, *tempCS );
1376
63.8k
  bool qgEnableChildren = partitioner.currQgEnable(); // QG possible at children level
1377
1378
63.8k
  m_CurrCtx++;
1379
1380
63.8k
  AffineMVInfo tmpMVInfo;
1381
63.8k
  bool isAffMVInfoSaved = m_cInterSearch.m_AffineProfList->savePrevAffMVInfo( 0, tmpMVInfo );
1382
1383
63.8k
  BlkUniMvInfo tmpUniMvInfo;
1384
63.8k
  bool         isUniMvInfoSaved = false;
1385
63.8k
  if( !tempCS->slice->isIntra() )
1386
0
  {
1387
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->savePrevUniMvInfo( tempCS->area.Y(), tmpUniMvInfo, isUniMvInfoSaved );
1388
0
  }
1389
1390
63.8k
  DeriveCtx deriveCtx = m_CABACEstimator->getDeriveCtx();
1391
1392
63.8k
  do
1393
163k
  {
1394
163k
    const auto &subCUArea  = partitioner.currArea();
1395
1396
163k
    if( tempCS->picture->Y().contains( subCUArea.lumaPos() ) )
1397
116k
    {
1398
116k
      PelStorage* orgBuffer =  &m_pOrgBuffer[partitioner.currDepth];
1399
116k
      PelStorage* rspBuffer =  &m_pRspBuffer[partitioner.currDepth];
1400
116k
      CodingStructure *tempSubCS = m_pTempCS[partitioner.currDepth];
1401
116k
      CodingStructure *bestSubCS = m_pBestCS[partitioner.currDepth];
1402
1403
116k
      tempCS->initSubStructure( *tempSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1404
116k
      tempCS->initSubStructure( *bestSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1405
1406
      // copy org buffer, need to be done after initSubStructure because of reshaping!
1407
116k
      orgBuffer->copyFrom( tempCS->getOrgBuf( subCUArea ) );
1408
1409
116k
      tempSubCS->bestParent = bestSubCS->bestParent = bestCS;
1410
1411
116k
      xCompressCU(tempSubCS, bestSubCS, partitioner );
1412
1413
116k
      tempSubCS->bestParent = bestSubCS->bestParent = nullptr;
1414
1415
116k
      if( bestSubCS->cost == MAX_DOUBLE )
1416
0
      {
1417
0
        CHECK( split == CU_QUAD_SPLIT, "Split decision reusing cannot skip quad split" );
1418
0
        tempCS->cost = MAX_DOUBLE;
1419
0
        tempCS->costDbOffset = 0;
1420
0
        m_CurrCtx--;
1421
0
        partitioner.exitCurrSplit();
1422
0
        xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1423
0
        if( partitioner.chType == CH_L )
1424
0
        {
1425
0
          tempCS->motionLut = oldMotionLut;
1426
0
        }
1427
1428
0
        m_CABACEstimator->getDeriveCtx() = deriveCtx;
1429
0
        return;
1430
0
      }
1431
1432
116k
      tempCS->useSubStructure( *bestSubCS, partitioner.chType, TREE_D, CS::getArea( *tempCS, subCUArea, partitioner.chType, partitioner.treeType ), partitioner.hasNextPart() || chromaNotSplit );
1433
1434
116k
      if( partitioner.currQgEnable() )
1435
6.22k
      {
1436
6.22k
        tempCS->prevQP[partitioner.chType] = bestSubCS->prevQP[partitioner.chType];
1437
6.22k
      }
1438
116k
      if( partitioner.isConsInter() )
1439
0
      {
1440
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1441
0
        {
1442
0
          CHECK( bestSubCS->cus[i]->predMode != MODE_INTER, "all CUs must be inter mode in an Inter coding region (SCIPU)" );
1443
0
        }
1444
0
      }
1445
116k
      else if( partitioner.isConsIntra() )
1446
0
      {
1447
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1448
0
        {
1449
0
          CHECK( bestSubCS->cus[i]->predMode == MODE_INTER, "all CUs must not be inter mode in an Intra coding region (SCIPU)" );
1450
0
        }
1451
0
      }
1452
1453
116k
      tempSubCS->releaseIntermediateData();
1454
116k
      bestSubCS->releaseIntermediateData();
1455
116k
      if( !tempCS->slice->isIntra() && partitioner.isConsIntra() )
1456
0
      {
1457
0
        tempCS->cost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1458
0
        if( tempCS->cost > bestCS->cost )
1459
0
        {
1460
0
          tempCS->cost = MAX_DOUBLE;
1461
0
          tempCS->costDbOffset = 0;
1462
0
          m_CurrCtx--;
1463
0
          partitioner.exitCurrSplit();
1464
0
          if( partitioner.chType == CH_L )
1465
0
          {
1466
0
            tempCS->motionLut = oldMotionLut;
1467
0
          }
1468
1469
0
          m_CABACEstimator->getDeriveCtx() = deriveCtx;
1470
0
          return;
1471
0
        }
1472
0
      }
1473
116k
    }
1474
163k
  } while( partitioner.nextPart( *tempCS ) );
1475
1476
63.8k
  partitioner.exitCurrSplit();
1477
1478
63.8k
  m_CurrCtx--;
1479
1480
63.8k
  m_CABACEstimator->getDeriveCtx() = deriveCtx;
1481
1482
63.8k
  if( chromaNotSplit )
1483
0
  {
1484
    //Note: In local dual tree region, the chroma CU refers to the central luma CU's QP.
1485
    //If the luma CU QP shall be predQP (no residual in it and before it in the QG), it must be revised to predQP before encoding the chroma CU
1486
    //Otherwise, the chroma CU uses predQP+deltaQP in encoding but is decoded as using predQP, thus causing encoder-decoded mismatch on chroma qp.
1487
0
    if( tempCS->pps->useDQP )
1488
0
    {
1489
      //find parent CS that including all coded CUs in the QG before this node
1490
0
      CodingStructure* qgCS = tempCS;
1491
0
      bool deltaQpCodedBeforeThisNode = false;
1492
0
      if( partitioner.currArea().lumaPos() != partitioner.currQgPos )
1493
0
      {
1494
0
        int numParentNodeToQgCS = 0;
1495
0
        while( qgCS->area.lumaPos() != partitioner.currQgPos )
1496
0
        {
1497
0
          CHECK( qgCS->parent == nullptr, "parent of qgCS shall exsit" );
1498
0
          qgCS = qgCS->parent;
1499
0
          numParentNodeToQgCS++;
1500
0
        }
1501
1502
        //check whether deltaQP has been coded (in luma CU or luma&chroma CU) before this node
1503
0
        CodingStructure* parentCS = tempCS->parent;
1504
0
        for( int i = 0; i < numParentNodeToQgCS; i++ )
1505
0
        {
1506
          //checking each parent
1507
0
          CHECK( parentCS == nullptr, "parentCS shall exsit" );
1508
0
          for( const auto &cu : parentCS->cus )
1509
0
          {
1510
0
            if( cu->rootCbf && !isChroma( cu->chType ) )
1511
0
            {
1512
0
              deltaQpCodedBeforeThisNode = true;
1513
0
              break;
1514
0
            }
1515
0
          }
1516
0
          parentCS = parentCS->parent;
1517
0
        }
1518
0
      }
1519
1520
      //revise luma CU qp before the first luma CU with residual in the SCIPU to predQP
1521
0
      if( !deltaQpCodedBeforeThisNode )
1522
0
      {
1523
        //get pred QP of the QG
1524
0
        const CodingUnit* cuFirst = qgCS->getCU( CH_L, TREE_D );
1525
0
        CHECK( cuFirst->lumaPos() != partitioner.currQgPos, "First cu of the Qg is wrong" );
1526
0
        int predQp = CU::predictQP( *cuFirst, qgCS->prevQP[CH_L] );
1527
1528
        //revise to predQP
1529
0
        int firstCuHasResidual = (int)tempCS->cus.size();
1530
0
        for( int i = 0; i < tempCS->cus.size(); i++ )
1531
0
        {
1532
0
          if( tempCS->cus[i]->rootCbf )
1533
0
          {
1534
0
            firstCuHasResidual = i;
1535
0
            break;
1536
0
          }
1537
0
        }
1538
1539
0
        for( int i = 0; i < firstCuHasResidual; i++ )
1540
0
        {
1541
0
          tempCS->cus[i]->qp = predQp;
1542
0
        }
1543
0
      }
1544
0
    }
1545
0
    partitioner.chType   = CH_C;
1546
0
    partitioner.treeType = TREE_C;
1547
1548
0
    m_CurrCtx++;
1549
1550
0
    CodingStructure *tempCSChroma = m_pTempCS2;
1551
0
    CodingStructure *bestCSChroma = m_pBestCS2;
1552
1553
0
    tempCS->initSubStructure( *tempCSChroma, partitioner.chType, partitioner.currArea(), false );
1554
0
    tempCS->initSubStructure( *bestCSChroma, partitioner.chType, partitioner.currArea(), false );
1555
0
    tempCSChroma->lumaCS = tempCS;
1556
0
    bestCSChroma->lumaCS = tempCS;
1557
0
    xCompressCU( tempCSChroma, bestCSChroma, partitioner );
1558
1559
    //attach chromaCS to luma CS and update cost
1560
0
    tempCS->useSubStructure( *bestCSChroma, partitioner.chType, TREE_D, CS::getArea( *bestCSChroma, partitioner.currArea(), partitioner.chType, partitioner.treeType ), false );
1561
1562
    //release tmp resource
1563
0
    tempCSChroma->releaseIntermediateData();
1564
0
    bestCSChroma->releaseIntermediateData();
1565
1566
0
    m_CurrCtx--;
1567
    //recover luma tree status
1568
0
    partitioner.chType = CH_L;
1569
0
    partitioner.treeType = TREE_D;
1570
0
    partitioner.modeType = MODE_TYPE_ALL;
1571
0
  }
1572
1573
  // Finally, add split-signaling bits for RD-cost check
1574
63.8k
  tempCS->fracBits += splitBits; // split bits
1575
63.8k
  tempCS->cost      = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1576
63.8k
  partitioner.modeType = modeTypeParent;
1577
1578
  // Check Delta QP bits for splitted structure
1579
63.8k
  if( !qgEnableChildren ) // check at deepest QG level only
1580
61.4k
    xCheckDQP( *tempCS, partitioner, true );
1581
1582
  // If the configuration being tested exceeds the maximum number of bytes for a slice / slice-segment, then
1583
  // a proper RD evaluation cannot be performed. Therefore, termination of the
1584
  // slice/slice-segment must be made prior to this CTU.
1585
  // This can be achieved by forcing the decision to be that of the rpcTempCU.
1586
  // The exception is each slice / slice-segment must have at least one CTU.
1587
63.8k
  if( bestCS->cost == MAX_DOUBLE )
1588
45.5k
  {
1589
45.5k
    bestCS->costDbOffset = 0;
1590
45.5k
  }
1591
1592
63.8k
  if( tempCS->cus.size() > 0 && modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTER )
1593
0
  {
1594
0
    int areaSizeNoResiCu = 0;
1595
0
    for( int k = 0; k < tempCS->cus.size(); k++ )
1596
0
    {
1597
0
      areaSizeNoResiCu += (tempCS->cus[k]->rootCbf == false) ? tempCS->cus[k]->lumaSize().area() : 0;
1598
0
    }
1599
0
    if( areaSizeNoResiCu >= (tempCS->area.lumaSize().area() >> 1) )
1600
0
    {
1601
0
      skipInterPass = true;
1602
0
    }
1603
0
  }
1604
1605
  // RD check for sub partitioned coding structure.
1606
63.8k
  xCheckBestMode( tempCS, bestCS, partitioner, encTestMode, m_EDO );
1607
1608
63.8k
  if( isAffMVInfoSaved )
1609
0
  {
1610
0
    m_cInterSearch.m_AffineProfList->addAffMVInfo(tmpMVInfo);
1611
0
  }
1612
1613
63.8k
  if( !tempCS->slice->isIntra() && isUniMvInfoSaved )
1614
0
  {
1615
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->addUniMvInfo(tmpUniMvInfo);
1616
0
  }
1617
1618
63.8k
  tempCS->motionLut = oldMotionLut;
1619
63.8k
  tempCS->releaseIntermediateData();
1620
63.8k
  tempCS->prevQP[partitioner.chType] = oldPrevQp;
1621
63.8k
}
1622
1623
1624
void EncCu::xCheckRDCostIntra( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1625
77.9k
{
1626
77.9k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTRA, tempCS, partitioner.chType );
1627
1628
77.9k
  tempCS->initStructData( encTestMode.qp, false ); // clear motion buffer
1629
1630
77.9k
  CodingUnit &cu      = tempCS->addCU( CS::getArea( *tempCS, tempCS->area, partitioner.chType, partitioner.treeType ), partitioner.chType );
1631
1632
77.9k
  partitioner.setCUData( cu );
1633
77.9k
  cu.slice            = tempCS->slice;
1634
77.9k
  cu.tileIdx          = m_tileIdx;
1635
77.9k
  cu.skip             = false;
1636
77.9k
  cu.mmvdSkip         = false;
1637
77.9k
  cu.predMode         = MODE_INTRA;
1638
77.9k
  cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
1639
77.9k
  cu.qp               = encTestMode.qp;
1640
77.9k
  cu.ispMode          = NOT_INTRA_SUBPARTITIONS;
1641
77.9k
  cu.initPuData();
1642
1643
77.9k
  m_cIntraSearch.m_ispTestedModes[0].init(0, 0, 1);
1644
77.9k
  if (m_pcEncCfg->m_FastIntraTools)
1645
77.9k
  {
1646
77.9k
    m_modeCtrl.comprCUCtx->intraWasTested = false;
1647
77.9k
    m_cIntraSearch.m_ispTestedModes[0].relatedCuIsValid = m_modeCtrl.comprCUCtx->relatedCuIsValid;
1648
77.9k
    if (!bestCS->cus.empty())
1649
21.3k
    {
1650
21.3k
      if ((bestCS->cus[0]->mergeFlag || bestCS->cus[0]->imv || bestCS->cus[0]->affine) && (!bestCS->cus[0]->ciip))
1651
21.3k
      {
1652
21.3k
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[0] = -1;
1653
21.3k
      }
1654
21.3k
    }
1655
77.9k
    if (!bestCS->slice->isIntra())
1656
0
    {
1657
0
      const Position posBL = cu.Y().bottomLeft();
1658
0
      const Position posTR = cu.Y().topRight();
1659
0
      for (int i = 0; i < 2; i++)
1660
0
      {
1661
0
        const CodingUnit* neigh = i ? cu.cs->getCURestricted(posTR.offset(0, -1), cu, CH_L) :cu.cs->getCURestricted(posBL.offset(-1, 0), cu, CH_L);
1662
0
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = -1;
1663
0
        if (neigh != nullptr)
1664
0
        {
1665
0
          int bestMode = neigh->firstTU->mtsIdx[0] ? 4 : 0;
1666
0
          bestMode |= neigh->lfnstIdx ? 2 : 0;
1667
0
          bestMode |= neigh->ispMode ? 1 : 0;
1668
0
          m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = bestMode;
1669
0
        }
1670
0
      }
1671
0
    }
1672
77.9k
  }
1673
1674
77.9k
  tempCS->interHad    = m_modeCtrl.comprCUCtx->interHad;
1675
77.9k
  double maxCostAllowedForChroma = MAX_DOUBLE;
1676
77.9k
  if( isLuma( partitioner.chType ) )
1677
23.9k
  {
1678
23.9k
    if (!tempCS->slice->isIntra() && bestCS)
1679
0
    {
1680
0
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner, bestCS->cost);
1681
0
    }
1682
23.9k
    else
1683
23.9k
    {
1684
23.9k
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner);
1685
23.9k
    }
1686
23.9k
    if (m_pcEncCfg->m_FastIntraTools)
1687
23.9k
    {
1688
23.9k
      if (m_cIntraSearch.m_ispTestedModes[0].intraWasTested)
1689
23.9k
      {
1690
23.9k
        m_modeCtrl.comprCUCtx->intraWasTested = m_cIntraSearch.m_ispTestedModes[0].intraWasTested;
1691
23.9k
      }
1692
23.9k
    }
1693
1694
23.9k
    if( !partitioner.isSepTree( *tempCS ) )
1695
0
    {
1696
0
      tempCS->lumaCost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1697
0
    }
1698
23.9k
    if (m_pcEncCfg->m_usePbIntraFast && tempCS->dist == MAX_DISTORTION && tempCS->interHad == 0)
1699
0
    {
1700
      // JEM assumes only perfect reconstructions can from now on beat the inter mode
1701
0
      m_modeCtrl.comprCUCtx->interHad = 0;
1702
0
      return;
1703
0
    }
1704
23.9k
  }
1705
1706
77.9k
  if( tempCS->area.chromaFormat != CHROMA_400 && ( partitioner.chType == CH_C || !CU::isSepTree(cu) ) )
1707
54.0k
  {
1708
54.0k
    bool useIntraSubPartitions = cu.ispMode != NOT_INTRA_SUBPARTITIONS;
1709
54.0k
    Partitioner subTuPartitioner = partitioner;
1710
54.0k
    if ((m_pcEncCfg->m_ISP >= 3) && (!partitioner.isSepTree(*tempCS) && useIntraSubPartitions))
1711
0
    {
1712
0
      maxCostAllowedForChroma = bestCS->cost < MAX_DOUBLE ? bestCS->cost - tempCS->lumaCost : MAX_DOUBLE;
1713
0
    }
1714
54.0k
    m_cIntraSearch.estIntraPredChromaQT(
1715
54.0k
      cu, (!useIntraSubPartitions || (CU::isSepTree(cu) && !isLuma(CH_C))) ? partitioner : subTuPartitioner,
1716
54.0k
      maxCostAllowedForChroma);
1717
54.0k
    if ((m_pcEncCfg->m_ISP >= 3) && useIntraSubPartitions && !cu.ispMode)
1718
0
    {
1719
0
      return;
1720
0
    }
1721
54.0k
  }
1722
1723
77.9k
  cu.rootCbf = false;
1724
1725
311k
  for (uint32_t t = 0; t < getNumberValidTBlocks(*cu.cs->pcv); t++)
1726
233k
  {
1727
233k
    cu.rootCbf |= cu.firstTU->cbf[t] != 0;
1728
233k
  }
1729
1730
  // Get total bits for current mode: encode CU
1731
77.9k
  m_CABACEstimator->resetBits();
1732
1733
77.9k
  if ((!cu.cs->slice->isIntra() || cu.cs->slice->sps->IBC) && cu.Y().valid())
1734
23.9k
  {
1735
23.9k
    m_CABACEstimator->cu_skip_flag(cu);
1736
23.9k
  }
1737
77.9k
  m_CABACEstimator->pred_mode(cu);
1738
77.9k
  m_CABACEstimator->cu_pred_data(cu);
1739
1740
  // Encode Coefficients
1741
77.9k
  CUCtx cuCtx;
1742
77.9k
  cuCtx.isDQPCoded = true;
1743
77.9k
  cuCtx.isChromaQpAdjCoded = true;
1744
77.9k
  m_CABACEstimator->cu_residual(cu, partitioner, cuCtx);
1745
1746
77.9k
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
1747
77.9k
  tempCS->cost = m_cRdCost.calcRdCost(tempCS->fracBits, tempCS->dist);
1748
1749
77.9k
  xEncodeDontSplit(*tempCS, partitioner);
1750
1751
77.9k
  xCheckDQP(*tempCS, partitioner);
1752
1753
77.9k
  if( m_EDO )
1754
77.9k
  {
1755
77.9k
    xCalDebCost(*tempCS, partitioner);
1756
77.9k
  }
1757
1758
77.9k
  DTRACE_MODE_COST(*tempCS, m_cRdCost.getLambda(true));
1759
77.9k
  xCheckBestMode(tempCS, bestCS, partitioner, encTestMode, m_EDO);
1760
1761
77.9k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
1762
77.9k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
1763
77.9k
}
1764
1765
void EncCu::xCheckDQP( CodingStructure& cs, Partitioner& partitioner, bool bKeepCtx )
1766
160k
{
1767
160k
  if( !cs.pps->useDQP )
1768
0
  {
1769
0
    return;
1770
0
  }
1771
1772
160k
  if (partitioner.isSepTree(cs) && isChroma(partitioner.chType))
1773
90.8k
  {
1774
90.8k
    return;
1775
90.8k
  }
1776
1777
69.8k
  if( !partitioner.currQgEnable() ) // do not consider split or leaf/not leaf QG condition (checked by caller)
1778
63.2k
  {
1779
63.2k
    return;
1780
63.2k
  }
1781
1782
6.67k
  CodingUnit* cuFirst = cs.getCU( partitioner.chType, partitioner.treeType );
1783
1784
6.67k
  CHECK( bKeepCtx && cs.cus.size() <= 1 && partitioner.getImplicitSplit( cs ) == CU_DONT_SPLIT, "bKeepCtx should only be set in split case" );
1785
6.67k
  CHECK( !bKeepCtx && cs.cus.size() > 1, "bKeepCtx should never be set for non-split case" );
1786
6.67k
  CHECK( !cuFirst, "No CU available" );
1787
1788
6.67k
  bool hasResidual = false;
1789
6.67k
  for( const auto &cu : cs.cus )
1790
19.1k
  {
1791
    //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1792
19.1k
    if( cu->rootCbf && !isChroma( cu->chType ))
1793
1.15k
    {
1794
1.15k
      hasResidual = true;
1795
1.15k
      break;
1796
1.15k
    }
1797
19.1k
  }
1798
1799
6.67k
  int predQP = CU::predictQP( *cuFirst, cs.prevQP[partitioner.chType] );
1800
1801
6.67k
  if( hasResidual )
1802
1.15k
  {
1803
1.15k
    TempCtx ctxTemp( m_CtxCache );
1804
1.15k
    if( !bKeepCtx ) ctxTemp = SubCtx( Ctx::DeltaQP, m_CABACEstimator->getCtx() );
1805
1806
1.15k
    m_CABACEstimator->resetBits();
1807
1.15k
    m_CABACEstimator->cu_qp_delta( *cuFirst, predQP, cuFirst->qp );
1808
1809
1.15k
    cs.fracBits += m_CABACEstimator->getEstFracBits(); // dQP bits
1810
1.15k
    cs.cost      = m_cRdCost.calcRdCost(cs.fracBits, cs.dist);
1811
1812
1813
1.15k
    if( !bKeepCtx ) m_CABACEstimator->getCtx() = SubCtx( Ctx::DeltaQP, ctxTemp );
1814
1815
    // NOTE: reset QPs for CUs without residuals up to first coded CU
1816
1.15k
    for( const auto &cu : cs.cus )
1817
1.16k
    {
1818
      //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1819
1.16k
      if( cu->rootCbf && !isChroma( cu->chType ))
1820
1.15k
      {
1821
1.15k
        break;
1822
1.15k
      }
1823
7
      cu->qp = predQP;
1824
7
    }
1825
1.15k
  }
1826
5.51k
  else
1827
5.51k
  {
1828
    // No residuals: reset CU QP to predicted value
1829
5.51k
    for( const auto &cu : cs.cus )
1830
17.9k
    {
1831
17.9k
      cu->qp = predQP;
1832
17.9k
    }
1833
5.51k
  }
1834
6.67k
}
1835
1836
CodingUnit *EncCu::getCuForInterPrediction( CodingStructure *cs, const EncTestMode& encTestMode )
1837
0
{
1838
0
  CodingUnit *cu = cs->getCU( CH_L, TREE_D );
1839
1840
0
  if( cu == nullptr )
1841
0
  {
1842
0
    CHECK( cs->getCU( CH_L, TREE_D ) != nullptr, "Wrong CU/PU setting in CS" );
1843
0
    cu = &cs->addCU( cs->area, CH_L );
1844
0
  }
1845
1846
0
  cu->slice       = cs->slice;
1847
0
  cu->tileIdx     = m_tileIdx;
1848
0
  cu->skip        = false;
1849
0
  cu->mmvdSkip    = false;
1850
0
  cu->mmvdMergeFlag
1851
0
                  = false;
1852
0
  cu->geo         = false;
1853
0
  cu->predMode    = MODE_INTER;
1854
0
  cu->chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
1855
0
  cu->qp          = encTestMode.qp;
1856
0
  cu->affine      = false;
1857
0
  cu->multiRefIdx = 0;
1858
0
  cu->mipFlag     = false;
1859
0
  cu->ciip        = false;
1860
1861
0
  return cu;
1862
0
}
1863
1864
int getDmvrMvdNum( const CodingUnit &cu )
1865
0
{
1866
0
  const int dx = std::max<int>( cu.lwidth()  >> DMVR_SUBCU_SIZE_LOG2, 1 );
1867
0
  const int dy = std::max<int>( cu.lheight() >> DMVR_SUBCU_SIZE_LOG2, 1 );
1868
0
  return dx * dy;
1869
0
}
1870
1871
void EncCu::xCheckRDCostUnifiedMerge( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, EncTestMode &encTestMode )
1872
0
{
1873
0
  const Slice &slice = *tempCS->slice;
1874
1875
0
  CHECK( slice.sliceType == VVENC_I_SLICE, "Merge modes not available for I-slices" );
1876
1877
0
  tempCS->initStructData( encTestMode.qp );
1878
1879
0
  MergeCtx          mergeCtx, gpmMergeCtx;
1880
0
  AffineMergeCtx    affineMergeCtx;
1881
0
  GeoComboCostList &comboList = m_comboList;
1882
0
  const SPS        &sps       = *tempCS->sps;
1883
1884
0
  if( sps.SbtMvp )
1885
0
  {
1886
0
    const Size bufSize           = g_miScaling.scale( tempCS->area.lumaSize() );
1887
0
    affineMergeCtx.subPuMvpMiBuf = MotionBuf        ( m_subPuMiBuf, bufSize );
1888
0
  }
1889
1890
0
  m_mergeBestSATDCost = MAX_DOUBLE;
1891
1892
0
  CodingUnit *cu = getCuForInterPrediction( tempCS, encTestMode );
1893
0
  partitioner.setCUData            ( *cu );
1894
0
  CU::getInterMergeCandidates      ( *cu, mergeCtx, 0 );
1895
0
  if( sps.MMVD )
1896
0
    CU::getInterMMVDMergeCandidates( *cu, mergeCtx );
1897
1898
0
  bool sameMV[MRG_MAX_NUM_CANDS] = { false, };
1899
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
1900
0
  {
1901
0
    for( int m = 0; m < mergeCtx.numValidMergeCand - 1; m++ )
1902
0
    {
1903
0
      if( !sameMV[m] )
1904
0
      {
1905
0
        for( int n = m + 1; n < mergeCtx.numValidMergeCand; n++ )
1906
0
        {
1907
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
1908
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
1909
0
        }
1910
0
      }
1911
0
    }
1912
0
  }
1913
1914
0
  MergeBufVector mrgPredBufNoCiip;
1915
0
  MergeBufVector geoBuffer;
1916
0
  const double  sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda() * FRAC_BITS_SCALE;
1917
1918
0
  const UnitArea localUnitArea( cu->chromaFormat, Area( 0, 0, cu->Y().width, cu->Y().height ) );
1919
0
  for( int i = 0; i < mergeCtx.numValidMergeCand; i++ )
1920
0
  {
1921
0
    mrgPredBufNoCiip.push_back( m_acMergeTmpBuffer[i].getCompactBuf( localUnitArea ) );
1922
0
  }
1923
1924
0
  int numMergeSatdCand = std::min( bestCS->area.lumaSize().area() >= 64 ? m_pcEncCfg->m_mergeRdCandQuotaRegular : m_pcEncCfg->m_mergeRdCandQuotaRegularSmallBlk, mergeCtx.numValidMergeCand );
1925
1926
0
  bool isCiipEnabled  = sps.CIIP && bestCS->area.lumaSize().area() >= 64 && bestCS->area.lumaSize().maxDim() < MAX_CU_SIZE;
1927
0
       isCiipEnabled &= m_pcEncCfg->m_CIIP <= 1 || !m_modeCtrl.getBlkInfo( tempCS->area ).isSkip; //5
1928
1929
0
  if( isCiipEnabled )
1930
0
  {
1931
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaCiip, mergeCtx.numValidMergeCand );
1932
0
  }
1933
1934
0
  const bool affineMrgAvail = ( m_pcEncCfg->m_Affine <= 2 || slice.TLayer <= 3 || m_pcEncCfg->m_SbTMVP )
1935
0
                           && ( m_pcEncCfg->m_Affine || sps.SbtMvp ) && m_pcEncCfg->m_maxNumAffineMergeCand && bestCS->area.Y().minDim() >= 8;
1936
1937
0
  if( affineMrgAvail )
1938
0
  {
1939
0
    CU::getAffineMergeCand( *cu, affineMergeCtx );
1940
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaSubBlk, affineMergeCtx.numValidMergeCand );
1941
0
  }
1942
1943
0
  int numSatdCandPreGeo = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1944
0
  bool toAddGpmCand     = false;
1945
0
  if( sps.GEO && slice.isInterB() // base checks
1946
0
      && cu->lumaSize().minDim() >= GEO_MIN_CU_SIZE  && cu->lumaSize().maxDim() <= GEO_MAX_CU_SIZE && cu->lumaSize().maxDim() < 8 * cu->lumaSize().minDim() // size checks
1947
0
      && !( m_pcEncCfg->m_Geo > 2 && slice.TLayer <= 1 ) ) // speedups
1948
0
  {
1949
0
    cu->mergeFlag            = true;
1950
0
    cu->geo                  = true;
1951
0
    CU::getGeoMergeCandidates( *cu, gpmMergeCtx );
1952
0
    toAddGpmCand             = prepareGpmComboList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, comboList, geoBuffer, *cu );
1953
0
    numMergeSatdCand        += toAddGpmCand ? std::min( m_pcEncCfg->m_mergeRdCandQuotaGpm, ( int ) comboList.list.size() ) : 0;
1954
0
  }
1955
1956
0
  numMergeSatdCand  = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1957
1958
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
1959
0
  m_mergeItemList.resetList( numMergeSatdCand );
1960
0
  const TempCtx ctxStart   ( m_CtxCache, m_CABACEstimator->getCtx() );
1961
0
  const DFunc   dfunc      = encTestMode.lossless ? DF_SAD : ( m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
1962
0
  DistParam     distParam  = m_cRdCost.setDistParam( tempCS->getOrgBuf().Y(), tempCS->getOrgBuf().Y(), sps.bitDepths[CH_L], dfunc );
1963
0
  m_uiSadBestForQPA        = MAX_DISTORTION;
1964
1965
0
  addRegularCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV, mrgPredBufNoCiip );
1966
1967
  // add CIIP candidates directly after adding regular cands
1968
0
  if( isCiipEnabled )
1969
0
  {
1970
0
    addCiipCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV );
1971
0
  }
1972
1973
0
  if( sps.MMVD && !!m_mergeItemList.size() && !( m_pcEncCfg->m_useFastMrg >= 2 && m_mergeItemList.size() <= 1 ) )
1974
0
  {
1975
0
    addMmvdCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1976
0
  }
1977
1978
0
  if( affineMergeCtx.numValidMergeCand > 0 )
1979
0
  {
1980
0
    addAffineCandsToPruningList( affineMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1981
0
  }
1982
1983
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
1984
0
  {
1985
0
    m_mergeBestSATDCost    = m_mergeItemList.getMergeItemInList( 0 )->cost;
1986
0
    const double threshold = m_mergeBestSATDCost * MRG_FAST_RATIO[tempCS->picture->useFastMrg];
1987
0
    const   int shrinkSize = std::min( numSatdCandPreGeo, ( int ) updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand ) );
1988
0
    m_mergeItemList        . shrinkList( shrinkSize );
1989
0
  }
1990
0
  else
1991
0
  {
1992
0
    m_mergeItemList        . shrinkList( numSatdCandPreGeo );
1993
0
  }
1994
1995
0
  if( toAddGpmCand )
1996
0
  {
1997
0
    addGpmCandsToPruningList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, comboList, geoBuffer, distParam, *cu );
1998
0
  }
1999
2000
0
  if(    m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2 && m_uiSadBestForQPA < MAX_DISTORTION && slice.TLayer == 0 // non-Intra key-frame
2001
0
      && m_pcEncCfg->m_salienceBasedOpt
2002
0
      && m_pcEncCfg->m_usePerceptQPA && partitioner.currQgEnable() && partitioner.currSubdiv == 0 ) // CTU-level luma quantization group
2003
0
  {
2004
0
    CHECK( bestCS->cost < MAX_DOUBLE, "This has to be the first test performed!" );
2005
2006
0
    const Picture *pic         = slice.pic;
2007
0
    const bool     isBIM       = m_pcEncCfg->m_RCNumPasses != 2 && m_pcEncCfg->m_blockImportanceMapping && !pic->m_picShared->m_ctuBimQpOffset.empty();
2008
0
    const uint32_t rsAddr      = getCtuAddr( partitioner.currQgPos, *pic->cs->pcv );
2009
0
    const int      pumpReducQP = BitAllocation::getCtuPumpingReducingQP( &slice, tempCS->getOrgBuf( COMP_Y ), m_uiSadBestForQPA, *m_globalCtuQpVector, rsAddr,
2010
0
                                                                         m_pcEncCfg->m_QP, isBIM );
2011
2012
0
    if( pumpReducQP != 0 ) // subtract QP offset, reduces Intra-period pumping or overcoding
2013
0
    {
2014
0
      encTestMode.qp = Clip3( 0, MAX_QP, encTestMode.qp - pumpReducQP );
2015
0
      tempCS->currQP[partitioner.chType] = tempCS->baseQP =
2016
0
      bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3( 0, MAX_QP, tempCS->baseQP - pumpReducQP );
2017
2018
0
      updateLambda( slice, pic->ctuQpaLambda[rsAddr], pic->ctuAdaptedQP[rsAddr], tempCS->baseQP, true );
2019
0
    }
2020
0
  }
2021
2022
  // Try to limit number of candidates using SATD-costs
2023
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
2024
0
  {
2025
    // shrink GEO list as well
2026
0
    const double threshold = m_mergeItemList.getMergeItemInList( 0 )->cost * MRG_FAST_RATIO[0];
2027
0
    numMergeSatdCand       = updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand );
2028
0
    m_mergeBestSATDCost    = m_mergeItemList.size() != 0 ? m_mergeItemList.getMergeItemInList( 0 )->cost : MAX_DOUBLE;
2029
0
  }
2030
0
  else
2031
0
  {
2032
0
    numMergeSatdCand       = std::min<int>( numMergeSatdCand, ( int ) m_mergeItemList.size() );
2033
0
  }
2034
2035
  // 2. Pass: RD checking 
2036
0
  tempCS->initStructData( encTestMode.qp );
2037
0
  m_CABACEstimator->getCtx() = ctxStart;
2038
2039
0
  double bestEndCost                            =   MAX_DOUBLE;
2040
0
  bool bestIsSkip                               =   false;
2041
0
  PelUnitBuf ciipBuf                            =   m_aTmpStorageLCU[1].getCompactBuf( *cu );
2042
0
  bool ciipChromaDone                           =   false;
2043
0
  bool isRegularTestedAsSkip[MRG_MAX_NUM_CANDS] = { false, };
2044
0
  bool geoWasTested                             =   false;
2045
0
  int  stopCand                                 =   numMergeSatdCand;
2046
2047
0
  CHECK( numMergeSatdCand > 0 && m_mergeItemList.size() == 0, "Empty merge item list is not expected" );
2048
2049
0
  for( uint32_t noResidualPass = 0; noResidualPass < 2; noResidualPass++ )
2050
0
  {
2051
0
    const bool forceNoResidual = noResidualPass == 1;
2052
0
    for( uint32_t mrgHadIdx = 0; mrgHadIdx < stopCand; mrgHadIdx++ )
2053
0
    {
2054
0
      auto mergeItem = m_mergeItemList.getMergeItemInList( mrgHadIdx );
2055
0
      CHECK( mergeItem == nullptr, "Wrong merge item" );
2056
2057
0
      const bool isCiip = mergeItem->mergeItemType == MergeItem::MergeItemType::CIIP;
2058
0
      const bool isGeo  = mergeItem->mergeItemType == MergeItem::MergeItemType::GPM;
2059
0
      const bool isRglr = mergeItem->mergeItemType == MergeItem::MergeItemType::REGULAR;
2060
0
      const bool isMmvd = mergeItem->mergeItemType == MergeItem::MergeItemType::MMVD;
2061
2062
0
      if( noResidualPass != 0 && isCiip && isRegularTestedAsSkip[mergeItem->mergeIdx] )
2063
0
      {
2064
0
        continue;
2065
0
      }
2066
2067
0
      if( noResidualPass ? mergeItem->noResidual : bestIsSkip )
2068
0
      {
2069
0
        continue;
2070
0
      }
2071
2072
0
      if( isGeo )
2073
0
      {
2074
0
        if( m_pcEncCfg->m_Geo > 2 && geoWasTested && !bestCS->cus.empty() && !bestCS->getCU( partitioner.chType, partitioner.treeType )->geo )
2075
0
        {
2076
0
          continue;
2077
0
        }
2078
2079
0
        geoWasTested = true;
2080
0
      }
2081
2082
0
      cu = getCuForInterPrediction( tempCS, encTestMode );
2083
0
      partitioner.setCUData( *cu );
2084
0
      const bool resetCiip2Regular = mergeItem->exportMergeInfo( *cu, forceNoResidual );
2085
2086
0
      if( isRglr || resetCiip2Regular )
2087
0
      {
2088
0
        if( CU::checkDMVRCondition( *cu ) ) std::copy_n( m_subPuMvOffset[mergeItem->mergeIdx].data(), getDmvrMvdNum( *cu ), cu->mvdL0SubPu );
2089
0
      }
2090
2091
0
      if( isMmvd && mergeItem->noBdofRefine )
2092
0
      {
2093
        // no BDOF refinement was made for the luma prediction, need to have luma prediction again
2094
0
        mergeItem->lumaPredReady = false;
2095
0
      }
2096
2097
0
      PelUnitBuf *predBuf1   = nullptr, *predBuf2 = isCiip ? &ciipBuf : nullptr;
2098
0
      PelUnitBuf  dstPredBuf = tempCS->getPredBuf( *cu );
2099
2100
0
      if( isGeo )
2101
0
      {
2102
0
        predBuf1 = &geoBuffer[cu->geoMergeIdx[0]];
2103
0
        predBuf2 = &geoBuffer[cu->geoMergeIdx[1]];
2104
0
      }
2105
2106
0
      if( resetCiip2Regular )
2107
0
      {
2108
0
        dstPredBuf.copyFrom( mrgPredBufNoCiip[mergeItem->mergeIdx] );
2109
0
      }
2110
0
      else
2111
0
      {
2112
0
        if( isCiip && !resetCiip2Regular && isChromaEnabled( cu->chromaFormat ) && cu->chromaSize().width > 2 )
2113
0
        {
2114
0
          if( !ciipChromaDone )
2115
0
          {
2116
0
            cu->intraDir[0] = PLANAR_IDX;
2117
0
            cu->intraDir[1] = DM_CHROMA_IDX;
2118
2119
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cb() );
2120
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cb, ciipBuf.Cb(), *cu );
2121
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cr() );
2122
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cr, ciipBuf.Cr(), *cu );
2123
2124
0
            ciipChromaDone  = true;
2125
0
          }
2126
0
        }
2127
2128
0
        if(  mergeItem->lumaPredReady ||  mergeItem->chromaPredReady )
2129
0
          dstPredBuf.copyFrom( mergeItem->getPredBuf( localUnitArea ), mergeItem->lumaPredReady, mergeItem->chromaPredReady );
2130
0
        if( !mergeItem->lumaPredReady || !mergeItem->chromaPredReady )
2131
0
          generateMergePrediction( localUnitArea, mergeItem, *cu, !mergeItem->lumaPredReady, !mergeItem->chromaPredReady, dstPredBuf, true, forceNoResidual, predBuf1, predBuf2 );
2132
0
      }
2133
2134
0
      if( !cu->mmvdSkip && !cu->ciip && !cu->affine && !cu->geo && noResidualPass != 0 )
2135
0
      {
2136
0
        CHECK( mergeItem->mergeIdx >= mergeCtx.numValidMergeCand, "out of normal merge" );
2137
0
        isRegularTestedAsSkip[mergeItem->mergeIdx] = true;
2138
0
      }
2139
2140
0
      xEncodeInterResidual( tempCS, bestCS, partitioner, encTestMode, noResidualPass, noResidualPass == 0 ? &mergeItem->noResidual : nullptr );
2141
2142
0
      if( m_pcEncCfg->m_useFastMrg >= 2 )
2143
0
      {
2144
0
        if( cu->ciip && bestCS->cost == MAX_DOUBLE && mrgHadIdx + 1 == numMergeSatdCand )
2145
0
        {
2146
0
          numMergeSatdCand = ( unsigned ) m_mergeItemList.size();
2147
0
        }
2148
      
2149
0
        if( mrgHadIdx > 0 && tempCS->cost >= bestEndCost && !cu->ciip && !isGeo )
2150
0
        {
2151
0
          stopCand = mrgHadIdx + 1;
2152
0
        }
2153
      
2154
0
        if( noResidualPass == 0 )
2155
0
        {
2156
0
          bestEndCost = std::min( bestEndCost, tempCS->cost );
2157
0
        }
2158
0
      }
2159
2160
0
      if( m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip && !cu->ciip )
2161
0
      {
2162
0
        bestIsSkip = !bestCS->cus.empty() && bestCS->getCU( partitioner.chType, partitioner.treeType )->rootCbf == 0;
2163
0
      }
2164
2165
0
      tempCS->initStructData( encTestMode.qp );
2166
0
    }   // end loop mrgHadIdx
2167
0
  }
2168
0
}
2169
2170
unsigned int EncCu::updateRdCheckingNum( MergeItemList &mergeItemList, double threshold, unsigned int numMergeSatdCand )
2171
0
{
2172
0
  for( uint32_t i = 0; i < mergeItemList.size(); i++ )
2173
0
  {
2174
0
    const auto mergeItem = mergeItemList.getMergeItemInList( i );
2175
0
    if( mergeItem == nullptr || mergeItem->cost > threshold )
2176
0
    {
2177
0
      numMergeSatdCand = i;
2178
0
      break;
2179
0
    }
2180
0
  }
2181
0
  return std::min( numMergeSatdCand, ( unsigned ) mergeItemList.size() );
2182
0
}
2183
2184
void EncCu::generateMergePrediction( const UnitArea &unitArea, MergeItem *mergeItem, CodingUnit &pu, bool luma, bool chroma,
2185
                                     PelUnitBuf &dstBuf, bool finalRd, bool forceNoResidual, PelUnitBuf *predBuf1, PelUnitBuf *predBuf2 )
2186
0
{
2187
0
  CHECK( ( luma && mergeItem->lumaPredReady ) || ( chroma && mergeItem->chromaPredReady ), "Prediction has been avaiable" );
2188
2189
0
  pu.mcControl = ( !luma ? 4 : 0 ) | ( !chroma ? 2 : 0 );
2190
2191
0
  switch( mergeItem->mergeItemType )
2192
0
  {
2193
0
  case MergeItem::MergeItemType::REGULAR:
2194
    // here predBuf1 is predBufNoCiip
2195
0
    pu.mvRefine = true;
2196
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2197
0
    pu.mvRefine = false;
2198
0
    if( predBuf1 != nullptr )
2199
0
    {
2200
0
      predBuf1->copyFrom( dstBuf, luma, chroma );
2201
0
    }
2202
0
    break;
2203
2204
0
  case MergeItem::MergeItemType::CIIP:
2205
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2206
2207
0
    if( luma )
2208
0
    {
2209
      // generate intrainter Y prediction
2210
0
      dstBuf.Y().weightCiip( predBuf2->Y(), mergeItem->numCiipIntra );
2211
0
    }
2212
2213
0
    if( chroma )
2214
0
    {
2215
0
      if( pu.chromaSize().width > 2 )
2216
0
      {
2217
0
        dstBuf.Cb().weightCiip( predBuf2->Cb(), mergeItem->numCiipIntra );
2218
0
        dstBuf.Cr().weightCiip( predBuf2->Cr(), mergeItem->numCiipIntra );
2219
0
      }
2220
0
    }
2221
2222
0
    break;
2223
2224
0
  case MergeItem::MergeItemType::MMVD:
2225
0
    pu.mcControl           |= finalRd ? 0 : ( pu.mmvdMergeIdx.pos.step > 2 || m_pcEncCfg->m_MMVD > 1 ) ? 1 : 0;
2226
0
    mergeItem->noBdofRefine = pu.mccNoBdof() && pu.cs->sps->BDOF && !pu.cs->picHeader->disBdofFlag;
2227
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2228
0
    break;
2229
2230
0
  case MergeItem::MergeItemType::SBTMVP:
2231
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2232
0
    break;
2233
2234
0
  case MergeItem::MergeItemType::AFFINE:
2235
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2236
0
    break;
2237
2238
0
  case MergeItem::MergeItemType::GPM:
2239
    // here predBuf1 and predBuf2 point to geoBuffer[mergeCand0] and geoBuffer[mergeCand1], respectively
2240
0
    CHECK( predBuf1 == nullptr || predBuf2 == nullptr, "Invalid input buffer to GPM" );
2241
0
    m_cInterSearch.weightedGeoBlk( pu.slice->clpRngs, pu, pu.geoSplitDir, luma && chroma ? MAX_NUM_CH : luma ? CH_L : CH_C, dstBuf, *predBuf1, *predBuf2 );
2242
0
    break;
2243
2244
0
  default:
2245
0
    THROW("Wrong merge item type");
2246
0
  }
2247
2248
0
  auto mergeItemPredBuf = mergeItem->getPredBuf( unitArea );
2249
2250
0
  if( dstBuf.Y().buf == mergeItemPredBuf.Y().buf )
2251
0
  {
2252
    // dst is the internal buffer
2253
0
    mergeItem->lumaPredReady   |= luma;
2254
0
    mergeItem->chromaPredReady |= chroma;
2255
0
  }
2256
0
  else if( finalRd && !forceNoResidual )
2257
0
  {
2258
    // at final RD stage, with and without residuals are both checked
2259
    // it makes sense to buffer the prediction
2260
0
    mergeItemPredBuf.copyFrom( dstBuf, luma, chroma );
2261
0
    mergeItem->lumaPredReady   |= luma;
2262
0
    mergeItem->chromaPredReady |= chroma;
2263
0
  }
2264
0
}
2265
2266
void EncCu::addRegularCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart,
2267
                                          DistParam& distParam, CodingUnit& pu, bool* sameMv, MergeBufVector& regularPred )
2268
0
{
2269
0
  pu.geo = pu.affine
2270
0
         = pu.mmvdMergeFlag = pu.mmvdSkip
2271
0
         = pu.ciip
2272
0
         = false;
2273
2274
0
  for( uint32_t uiMergeCand = 0; uiMergeCand < mergeCtx.numValidMergeCand; uiMergeCand++ )
2275
0
  {
2276
0
    if( sameMv[uiMergeCand] ) continue;
2277
2278
0
    mergeCtx.setMergeInfo   ( pu, uiMergeCand );
2279
2280
0
    if( m_pcEncCfg->m_ifpLines && // what about DMVR?
2281
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2282
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2283
0
    {
2284
0
      continue;
2285
0
    }
2286
2287
0
    pu.interDir             = mergeCtx.interDirNeighbours[uiMergeCand];
2288
0
    pu.BcwIdx               = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2289
0
    pu.imv                  = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2290
0
    CU::spanMotionInfo      ( pu );
2291
2292
0
    MergeItem *regularMerge = m_mergeItemList.allocateNewMergeItem();
2293
0
    regularMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::REGULAR, pu );
2294
0
    auto dstBuf             = regularMerge->getPredBuf( localUnitArea );
2295
0
    generateMergePrediction ( localUnitArea, regularMerge, pu, true, true, dstBuf, false, false, &regularPred[uiMergeCand], nullptr );
2296
0
    regularMerge->cost      = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2297
0
    if( CU::checkDMVRCondition( pu ) ) std::copy_n( pu.mvdL0SubPu, getDmvrMvdNum( pu ), m_subPuMvOffset[uiMergeCand].data() );
2298
0
    m_mergeItemList         . insertMergeItemToList( regularMerge );
2299
0
  }
2300
0
}
2301
2302
void EncCu::addCiipCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart, DistParam &distParam, CodingUnit &pu, bool* sameMv )
2303
0
{
2304
0
  int                numCiipIntra = -1;
2305
0
  PelUnitBuf         ciipBuf      = m_aTmpStorageLCU[1].getCompactBuf( pu );
2306
2307
0
  pu.ciip        = true;
2308
0
  pu.intraDir[0] = PLANAR_IDX;
2309
0
  pu.geo         = pu.affine
2310
0
                 = pu.mmvdMergeFlag = pu.mmvdSkip
2311
0
                 = false;
2312
0
  m_cIntraSearch . initIntraPatternChType        ( pu, pu.Y() );
2313
0
  m_cIntraSearch . predIntraAng                  ( COMP_Y, ciipBuf.Y(), pu );
2314
0
  numCiipIntra   = m_cIntraSearch.getNumIntraCiip( pu );
2315
2316
0
  int nonCiipMrgCnds[MRG_MAX_NUM_CANDS] = { 0, };
2317
0
  int numNonCiipCnds                    =   0;
2318
0
  for( ; numNonCiipCnds < m_mergeItemList.size(); numNonCiipCnds++ ) nonCiipMrgCnds[numNonCiipCnds] = m_mergeItemList.getMergeItemInList( numNonCiipCnds )->mergeIdx;
2319
2320
0
  for( int i = 0; i < numNonCiipCnds; i++ )
2321
0
  {
2322
0
    const unsigned int uiMergeCand = nonCiipMrgCnds[i];
2323
2324
0
    if( sameMv[uiMergeCand] ) continue;
2325
2326
0
    mergeCtx.setMergeInfo     ( pu, uiMergeCand );
2327
2328
0
    if( m_pcEncCfg->m_ifpLines && 
2329
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2330
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2331
0
    {
2332
0
      continue;
2333
0
    }
2334
2335
0
    pu.interDir               = mergeCtx.interDirNeighbours[uiMergeCand];
2336
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2337
0
    pu.imv                    = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2338
0
    CU::spanMotionInfo        ( pu );
2339
2340
0
    MergeItem* ciipMerge      = m_mergeItemList.allocateNewMergeItem();
2341
0
    ciipMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::CIIP, pu );
2342
0
    ciipMerge->numCiipIntra   = numCiipIntra;
2343
0
    auto dstBuf               = ciipMerge->getPredBuf( localUnitArea );
2344
0
    generateMergePrediction   ( localUnitArea, ciipMerge, pu, true, false, dstBuf, false, false, nullptr, &ciipBuf );
2345
0
      ciipMerge->cost         = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2346
0
    if( !m_mergeItemList      . insertMergeItemToList( ciipMerge ) && m_pcEncCfg->m_CIIP > 1 )
2347
0
    {
2348
0
      break;
2349
0
    }
2350
0
  }
2351
0
}
2352
2353
void EncCu::addMmvdCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx& ctxStart,
2354
                                       DistParam& distParam, CodingUnit& pu )
2355
0
{
2356
0
  pu.mmvdSkip              = true;
2357
0
  pu.affine                = pu.geo
2358
0
                           = pu.ciip
2359
0
                           = false;
2360
2361
0
  int       mmvdTestNum    = mergeCtx.numValidMergeCand > 1 ? MmvdIdx::ADD_NUM : MmvdIdx::ADD_NUM >> 1;
2362
0
  int       bestDir        = 0;
2363
0
  size_t    curListSize    = m_mergeItemList.size();
2364
0
  double    bestCostMerge  = m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost;
2365
0
  double    bestCostOffset = MAX_DOUBLE;
2366
0
  int       shiftCandStart = 0;
2367
2368
0
  if( m_pcEncCfg->m_MMVD == 4 )
2369
0
  {
2370
0
    const int cnd1idx = m_mergeItemList.size() == 1 ? 0 : 1;
2371
0
    const int mrgCnd0 = m_mergeItemList.getMergeItemInList(       0 )->mergeIdx;
2372
0
    const int mrgCnd1 = m_mergeItemList.getMergeItemInList( cnd1idx )->mergeIdx;
2373
2374
0
    if( mrgCnd0 > 1 && mrgCnd1 > 1 )
2375
0
    {
2376
0
      mmvdTestNum = 0;
2377
0
    }
2378
0
    else if( mrgCnd0 > 1 || mrgCnd1 > 1 )
2379
0
    {
2380
0
      int shiftCand = mrgCnd0 < 2 ? mrgCnd0 : mrgCnd1;
2381
2382
0
      if( shiftCand )
2383
0
      {
2384
0
        shiftCandStart = MMVD_MAX_REFINE_NUM;
2385
0
      }
2386
0
      else
2387
0
      {
2388
0
        mmvdTestNum    = MMVD_MAX_REFINE_NUM;
2389
0
      }
2390
0
    }
2391
0
  }
2392
2393
0
  for( int mmvdMergeCand = shiftCandStart; mmvdMergeCand < mmvdTestNum; mmvdMergeCand++ )
2394
0
  {
2395
0
    MmvdIdx mmvdIdx;
2396
0
    mmvdIdx.val = mmvdMergeCand;
2397
2398
0
    if( mmvdIdx.pos.step >= m_pcEncCfg->m_MmvdDisNum )
2399
0
    {
2400
0
      continue;
2401
0
    }
2402
2403
0
    if( m_pcEncCfg->m_MMVD > 1 )
2404
0
    {
2405
0
      int checkMMVD = xCheckMMVDCand( mmvdIdx, bestDir, mmvdTestNum, bestCostOffset, bestCostMerge, m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost );
2406
0
      mmvdMergeCand = mmvdIdx.val;
2407
2408
0
      if( checkMMVD )
2409
0
      {
2410
0
        if( checkMMVD == 2 )
2411
0
        {
2412
0
          break;
2413
0
        }
2414
0
        continue;
2415
0
      }
2416
0
    }
2417
2418
0
    mergeCtx.setMmvdMergeCandiInfo( pu, mmvdIdx );
2419
2420
0
    if( m_pcEncCfg->m_ifpLines &&
2421
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2422
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2423
0
    {
2424
      // skip candidate
2425
0
      continue;
2426
0
    }
2427
2428
0
    pu.interDir               = mergeCtx.interDirNeighbours[mmvdIdx.pos.baseIdx];
2429
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[mmvdIdx.pos.baseIdx] : BCW_DEFAULT;
2430
0
    pu.imv                    = mergeCtx.useAltHpelIf[mmvdIdx.pos.baseIdx] ? IMV_HPEL : IMV_OFF;
2431
0
    CU::spanMotionInfo        ( pu );
2432
2433
0
    MergeItem *mmvdMerge      = m_mergeItemList.allocateNewMergeItem();
2434
0
    mmvdMerge->importMergeInfo( mergeCtx, mmvdIdx.val, MergeItem::MergeItemType::MMVD, pu );
2435
0
    auto dstBuf               = mmvdMerge->getPredBuf( localUnitArea );
2436
0
    generateMergePrediction   ( localUnitArea, mmvdMerge, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2437
0
    mmvdMerge->cost           = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2438
0
    m_mergeItemList           . insertMergeItemToList( mmvdMerge );
2439
2440
0
    if( m_pcEncCfg->m_MMVD > 1 && mmvdMerge->cost < bestCostOffset )
2441
0
    {
2442
0
      bestCostOffset          = mmvdMerge->cost;
2443
0
      int CandCur             = mmvdIdx.val - MMVD_MAX_REFINE_NUM * mmvdIdx.pos.baseIdx;
2444
0
      if( CandCur < 4 )
2445
0
        bestDir               = CandCur;
2446
0
    }
2447
0
  }
2448
2449
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2450
0
  {
2451
0
    m_mergeItemList           . shrinkList( curListSize );
2452
0
  }
2453
0
}
2454
2455
void EncCu::addAffineCandsToPruningList( AffineMergeCtx &affineMergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2456
                                         const TempCtx& ctxStart, DistParam& distParam, CodingUnit& pu)
2457
0
{
2458
0
  bool sameMV[AFFINE_MRG_MAX_NUM_CANDS + 1]
2459
0
                      = { false, };
2460
0
  size_t curListSize  = m_mergeItemList.size();
2461
2462
0
  pu.mergeFlag = true;
2463
0
  pu.affine    = true;
2464
0
  pu.imv       = IMV_OFF;
2465
0
  pu.geo       = pu.mmvdMergeFlag = pu.mmvdSkip
2466
0
               = pu.ciip
2467
0
               = false;
2468
2469
0
  if( m_pcEncCfg->m_Affine > 1 )
2470
0
  {
2471
0
    for( int m = 0; m < affineMergeCtx.numValidMergeCand; m++ )
2472
0
    {
2473
0
      if( pu.cs->slice->TLayer > 3 && affineMergeCtx.mergeType[m] != MRG_TYPE_SUBPU_ATMVP )
2474
0
      {
2475
0
        sameMV[m] = m != 0;
2476
0
      }
2477
0
      else if( !sameMV[m + 1] )
2478
0
      {
2479
0
        for( int n = m + 1; n < affineMergeCtx.numValidMergeCand; n++ )
2480
0
        {
2481
0
          sameMV[n] |= affineMergeCtx.mvFieldNeighbours[m][0][0] == affineMergeCtx.mvFieldNeighbours[n][0][0]
2482
0
                    && affineMergeCtx.mvFieldNeighbours[m][1][0] == affineMergeCtx.mvFieldNeighbours[n][1][0];
2483
0
        }
2484
0
      }
2485
0
    }
2486
0
  }
2487
2488
0
  for( uint32_t mergeIdx = 0; mergeIdx < affineMergeCtx.numValidMergeCand; mergeIdx++ )
2489
0
  {
2490
0
    if( ( affineMergeCtx.mergeType[mergeIdx] != MRG_TYPE_SUBPU_ATMVP && m_pcEncCfg->m_Affine == 0 ) || sameMV[mergeIdx] )
2491
0
    {
2492
0
      continue;
2493
0
    }
2494
2495
0
    pu.mergeType              = affineMergeCtx.mergeType[mergeIdx];
2496
0
    pu.affineType             = affineMergeCtx.affineType[mergeIdx];
2497
0
    pu.interDir               = affineMergeCtx.interDirNeighbours[mergeIdx];
2498
0
    pu.BcwIdx                 = pu.interDir == 3 ? affineMergeCtx.BcwIdx[mergeIdx] : BCW_DEFAULT;
2499
2500
    // generate motion buf for IFP
2501
0
    if( affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP )
2502
0
    {
2503
0
      pu.refIdx[L0]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].refIdx;
2504
0
      pu.refIdx[L1]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].refIdx;
2505
0
      pu.mv    [L0][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].mv;
2506
0
      pu.mv    [L1][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].mv;
2507
0
      CU::spanMotionInfo      ( pu, &affineMergeCtx );
2508
0
    }
2509
0
    else
2510
0
    {
2511
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L0], L0 );
2512
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L1], L1 );
2513
0
      CU::spanMotionInfo      ( pu );
2514
0
    }
2515
2516
0
    if( m_pcEncCfg->m_ifpLines && !CU::isMotionBufInRangeFPP( pu, m_pcEncCfg->m_ifpLines ) )
2517
0
    {
2518
0
      continue;
2519
0
    }
2520
2521
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2522
0
    mergeItem->importMergeInfo( affineMergeCtx, mergeIdx, affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP ? MergeItem::MergeItemType::SBTMVP : MergeItem::MergeItemType::AFFINE, pu );
2523
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2524
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2525
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2526
0
    m_mergeItemList        . insertMergeItemToList( mergeItem );
2527
0
  }
2528
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2529
0
  {
2530
0
    m_mergeItemList        . shrinkList( curListSize );
2531
0
  }
2532
0
}
2533
2534
void EncCu::addGpmCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2535
                                      const TempCtx& ctxStart, const GeoComboCostList& comboList, MergeBufVector& geoBuffer, DistParam& distParam, CodingUnit& pu)
2536
0
{
2537
0
  int geoNumMrgSadCand    = std::min( GEO_MAX_TRY_WEIGHTED_SAD, ( int ) comboList.list.size() );
2538
0
  geoNumMrgSadCand        = std::min( geoNumMrgSadCand, m_pcEncCfg->m_Geo > 2 ? 10 : GEO_MAX_TRY_WEIGHTED_SAD );
2539
0
  double bestGeoCost      = MAX_DOUBLE / 2.0;
2540
0
  MergeItem* best2geo[2]  = { nullptr, nullptr };
2541
2542
0
  pu.mergeFlag = true;
2543
0
  pu.geo       = true;
2544
0
  pu.mergeType = MRG_TYPE_DEFAULT_N;
2545
0
  pu.BcwIdx    = BCW_DEFAULT;
2546
0
  pu.interDir  = 3;
2547
0
  pu.imv       = IMV_OFF;
2548
0
  pu.affine    = pu.mmvdMergeFlag = pu.mmvdSkip
2549
0
               = pu.ciip
2550
0
               = false;
2551
2552
0
  for( int candidateIdx = 0; candidateIdx < geoNumMrgSadCand; candidateIdx++ )
2553
0
  {
2554
0
    const int          splitDir     = comboList.list[candidateIdx].splitDir;
2555
0
    const MergeIdxPair mergeIdxPair { comboList.list[candidateIdx].mergeIdx0, comboList.list[candidateIdx].mergeIdx1 };
2556
0
    const int          gpmIndex     = MergeItem::getGpmUnfiedIndex( splitDir, mergeIdxPair );
2557
2558
0
    pu.mergeIdx            = gpmIndex;
2559
0
    pu.geoMergeIdx         = mergeIdxPair;
2560
0
    pu.geoSplitDir         = splitDir;
2561
0
    CU::spanGeoMotionInfo  ( pu, mergeCtx, pu.geoSplitDir, pu.geoMergeIdx[0], pu.geoMergeIdx[1] );
2562
2563
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2564
0
    mergeItem->importMergeInfo( mergeCtx, gpmIndex, MergeItem::MergeItemType::GPM, pu );
2565
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2566
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, &geoBuffer[mergeIdxPair[0]], &geoBuffer[mergeIdxPair[1]] );
2567
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2568
0
    bestGeoCost            = std::min( mergeItem->cost, bestGeoCost );
2569
2570
0
    if( mergeItem->cost > MRG_FAST_RATIO[0] * bestGeoCost || mergeItem->cost > m_mergeBestSATDCost )
2571
0
    {
2572
0
      m_mergeItemList      . giveBackMergeItem( mergeItem );
2573
2574
0
      if( m_pcEncCfg->m_Geo > 2 ) break;
2575
0
    }
2576
0
    else if( m_pcEncCfg->m_Geo < 2 )
2577
0
    {
2578
0
      m_mergeItemList      . insertMergeItemToList( mergeItem );
2579
0
    }
2580
0
    else
2581
0
    {
2582
0
      if( ( m_mergeItemList.size() > 0 && m_mergeItemList.getMergeItemInList( m_mergeItemList.size() - 1 )->cost <= mergeItem->cost ) ||
2583
0
        ( best2geo[1] && best2geo[1]->cost <= mergeItem->cost ) )
2584
0
      {
2585
0
        m_mergeItemList    . giveBackMergeItem( mergeItem );
2586
0
      }
2587
0
      else
2588
0
      {
2589
0
        if( !best2geo[0] || mergeItem->cost < best2geo[0]->cost )
2590
0
        {
2591
0
          if( best2geo[1] )
2592
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2593
2594
0
          best2geo[1] = best2geo[0]; best2geo[0] = mergeItem;
2595
0
        }
2596
0
        else
2597
0
        {
2598
0
          if( best2geo[1] ) 
2599
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2600
2601
0
          best2geo[1] = mergeItem;
2602
0
        }
2603
0
      }
2604
0
    }
2605
0
  }
2606
2607
0
  if( best2geo[0] )
2608
0
    m_mergeItemList        . insertMergeItemToList( best2geo[0] );
2609
0
  if( best2geo[1] )
2610
0
    m_mergeItemList        . insertMergeItemToList( best2geo[1] );
2611
0
}
2612
2613
bool EncCu::prepareGpmComboList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2614
                                 GeoComboCostList& comboList, MergeBufVector& geoBuffer, CodingUnit& pu )
2615
0
{
2616
0
          sqrtLambdaForFirstPass /= FRAC_BITS_SCALE;
2617
0
  const int bitsForPartitionIdx   = floorLog2(GEO_NUM_PARTITION_MODE);
2618
0
  const int maxNumMergeCandidates = std::min( ( int ) pu.cs->sps->maxNumGeoCand, MRG_MAX_NUM_CANDS );
2619
0
  DistParam distParam;
2620
  // the second arguments to setDistParam is dummy and will be updated before being used
2621
0
  DistParam  distParamWholeBlk     = m_cRdCost.setDistParam( pu.cs->getOrgBuf().Y(), pu.cs->getOrgBuf().Y(), pu.cs->sps->bitDepths[ CH_L ], DF_SAD );
2622
0
  Distortion bestWholeBlkSad       = MAX_UINT64;
2623
0
  double     bestWholeBlkCost      = MAX_DOUBLE;
2624
0
  const ClpRng&  lclpRng           = pu.slice->clpRngs[COMP_Y];
2625
0
  const unsigned rshift            = std::max<int>( 2, ( IF_INTERNAL_PREC - lclpRng.bd ) );
2626
0
  const int      offset            = ( 1 << ( rshift - 1 ) ) + IF_INTERNAL_OFFS;
2627
0
  const int      numSamples        = pu.Y().area();
2628
0
  Distortion sadWholeBlk            [GEO_MAX_NUM_UNI_CANDS];
2629
0
  int        pocMrg                 [GEO_MAX_NUM_UNI_CANDS];
2630
0
  Mv         mergeMv                [GEO_MAX_NUM_UNI_CANDS];
2631
0
  bool       isSkipThisCand         [GEO_MAX_NUM_UNI_CANDS]
2632
0
                                   = { false, };
2633
0
  bool       sameMV                 [MRG_MAX_NUM_CANDS]
2634
0
                                   = { false, };
2635
0
  MergeBufVector geoTempBuf;
2636
2637
0
  if( m_pcEncCfg->m_Geo > 2 )
2638
0
  {
2639
0
    for( int m = 0; m < maxNumMergeCandidates; m++ )
2640
0
    {
2641
0
      if( !sameMV[m] )
2642
0
      {
2643
0
        for( int n = m + 1; n < maxNumMergeCandidates; n++ )
2644
0
        {
2645
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
2646
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
2647
0
        }
2648
0
      }
2649
0
    }
2650
0
  }
2651
2652
0
  for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2653
0
  {
2654
0
    geoBuffer .push_back ( m_aTmpStorageLCU[2                         + mergeCand].getCompactBuf( localUnitArea ) );
2655
0
    geoTempBuf.push_back ( m_aTmpStorageLCU[2 + GEO_MAX_NUM_UNI_CANDS + mergeCand].getCompactBuf( localUnitArea ) );
2656
2657
0
    const int  listIdx    = mergeCtx.mvFieldNeighbours[mergeCand][0]      .refIdx == -1 ? 1 : 0;
2658
0
    const auto refPicList = RefPicList(listIdx);
2659
0
    const int  refIdx     = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].refIdx;
2660
2661
0
    pocMrg [mergeCand]    = pu.cs->slice->getRefPic( refPicList, refIdx )->poc;
2662
0
    mergeMv[mergeCand]    = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].mv;
2663
2664
0
    for( int i = 0; i < mergeCand; i++ )
2665
0
    {
2666
0
      if( pocMrg[mergeCand] == pocMrg[i] && mergeMv[mergeCand] == mergeMv[i] )
2667
0
      {
2668
0
        isSkipThisCand[mergeCand] = true;
2669
0
        break;
2670
0
      }
2671
0
    }
2672
2673
0
    if( sameMV[mergeCand] )
2674
0
    {
2675
0
      continue;
2676
0
    }
2677
2678
0
    if( m_pcEncCfg->m_ifpLines ) 
2679
0
    {
2680
0
      bool isOutOfRange  = !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][0].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2681
0
           isOutOfRange |= !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][1].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2682
2683
      // use sameMV to surpress processing of this cand later on...
2684
0
      sameMV[mergeCand] |= isOutOfRange;
2685
2686
0
      if( isOutOfRange )
2687
0
        continue;
2688
0
    }
2689
2690
0
    mergeCtx.setMergeInfo            ( pu, mergeCand );
2691
0
    CU::spanMotionInfo               ( pu );
2692
0
    m_cInterSearch.motionCompensation( pu, geoBuffer[mergeCand], REF_PIC_LIST_X );
2693
2694
0
    g_pelBufOP.roundGeo( geoBuffer[mergeCand].Y().buf, geoTempBuf[mergeCand].Y().buf, numSamples, rshift, offset, lclpRng );
2695
2696
0
    distParamWholeBlk.cur  = geoTempBuf[mergeCand].Y();
2697
0
    sadWholeBlk[mergeCand] = distParamWholeBlk.distFunc( distParamWholeBlk );
2698
2699
0
    if( sadWholeBlk[mergeCand] < bestWholeBlkSad )
2700
0
    {
2701
0
      bestWholeBlkSad  = sadWholeBlk[mergeCand];
2702
0
      int bitsCand     = mergeCand + 1;
2703
0
      bestWholeBlkCost = ( double ) bestWholeBlkSad + ( double ) bitsCand * sqrtLambdaForFirstPass;
2704
0
    }
2705
0
  }
2706
2707
0
  bool allCandsAreSame = true;
2708
0
  for( uint8_t mergeCand = 1; mergeCand < maxNumMergeCandidates; mergeCand++ )
2709
0
  {
2710
0
    allCandsAreSame &= isSkipThisCand[mergeCand];
2711
0
  }
2712
0
  if( allCandsAreSame )
2713
0
  {
2714
0
    return false;
2715
0
  }
2716
2717
0
  const int wIdx = floorLog2( pu.lwidth() )  - GEO_MIN_CU_LOG2;
2718
0
  const int hIdx = floorLog2( pu.lheight() ) - GEO_MIN_CU_LOG2;
2719
2720
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2721
0
  {
2722
0
    int maskStride = 0, maskStride2 = 0;
2723
0
    int stepX = 1;
2724
0
    Pel *sadMask;
2725
0
    int16_t angle = g_GeoParams[splitDir][0];
2726
    
2727
0
    if( g_angle2mirror[angle] == 2 )
2728
0
    {
2729
0
      maskStride  = -GEO_WEIGHT_MASK_SIZE;
2730
0
      maskStride2 = -( int ) pu.lwidth();
2731
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2732
0
                      [( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][1] ) * GEO_WEIGHT_MASK_SIZE
2733
0
                                                  + g_weightOffset[hIdx][wIdx][splitDir][0]
2734
0
                      ];
2735
0
    }
2736
0
    else if( g_angle2mirror[angle] == 1 )
2737
0
    {
2738
0
      stepX       = -1;
2739
0
      maskStride2 = pu.lwidth();
2740
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2741
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2742
0
                      [     GEO_WEIGHT_MASK_SIZE *     g_weightOffset[hIdx][wIdx][splitDir][1]
2743
0
                        + ( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][0] )
2744
0
                      ];
2745
0
    }
2746
0
    else
2747
0
    {
2748
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2749
0
      maskStride2 = -( int ) pu.lwidth();
2750
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2751
0
                      [   g_weightOffset[hIdx][wIdx][splitDir][1] * GEO_WEIGHT_MASK_SIZE
2752
0
                        + g_weightOffset[hIdx][wIdx][splitDir][0]
2753
0
                      ];
2754
0
    }
2755
2756
0
    m_cRdCost.setDistParamGeo ( distParam, pu.cs->getOrgBuf().Y(),
2757
0
                                nullptr, 0,
2758
0
                                sadMask, maskStride, stepX, maskStride2,
2759
0
                                pu.cs->sps->bitDepths[CH_L], COMP_Y );
2760
2761
0
    for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2762
0
    {
2763
0
      if( sameMV[mergeCand] )
2764
0
      {
2765
0
        continue;
2766
0
      }
2767
2768
0
      distParam.cur.buf         = geoTempBuf[mergeCand].Y().buf;
2769
0
      distParam.cur.stride      = geoTempBuf[mergeCand].Y().stride;
2770
0
      const Distortion sadLarge = distParam.distFunc( distParam );
2771
0
      const Distortion sadSmall = sadWholeBlk[mergeCand] - sadLarge;
2772
2773
0
      const int bitsCand        = mergeCand + 1;
2774
2775
0
      const double cost0        = ( double ) sadLarge + ( double ) bitsCand * sqrtLambdaForFirstPass;
2776
0
      const double cost1        = ( double ) sadSmall + ( double ) bitsCand * sqrtLambdaForFirstPass;
2777
2778
0
      m_GeoCostList.insert( splitDir, 0, mergeCand, cost0 );
2779
0
      m_GeoCostList.insert( splitDir, 1, mergeCand, cost1 );
2780
0
    }
2781
2782
0
    if( m_pcEncCfg->m_Geo == 4 )
2783
0
    {
2784
0
      if( splitDir == 1 )
2785
0
      {
2786
0
        splitDir += 7;
2787
0
      }
2788
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2789
0
      {
2790
0
        splitDir++;
2791
0
      }
2792
0
      else
2793
0
      {
2794
0
        splitDir += 5;
2795
0
      }
2796
0
    }
2797
0
    else
2798
0
    {
2799
0
      splitDir++;
2800
0
    }
2801
0
  }
2802
2803
0
  comboList.list.clear();
2804
2805
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2806
0
  {
2807
0
    for( int geoMotionIdx = 0; geoMotionIdx < maxNumMergeCandidates * ( maxNumMergeCandidates - 1 ); geoMotionIdx++ )
2808
0
    {
2809
0
      const MergeIdxPair mergeIdxPair = m_GeoModeTest[geoMotionIdx];
2810
2811
0
      if( sameMV[mergeIdxPair[0]] || sameMV[mergeIdxPair[1]] )
2812
0
      {
2813
0
        continue;
2814
0
      }
2815
2816
0
      double tempCost = m_GeoCostList.getCost( splitDir, mergeIdxPair[0], mergeIdxPair[1] );
2817
2818
0
      if( tempCost > bestWholeBlkCost )
2819
0
      {
2820
0
        continue;
2821
0
      }
2822
2823
0
      tempCost = tempCost + ( double ) bitsForPartitionIdx * sqrtLambdaForFirstPass;
2824
0
      comboList.list.push_back( GeoMergeCombo{ splitDir, mergeIdxPair[0], mergeIdxPair[1], tempCost } );
2825
0
    }
2826
2827
0
    if( m_pcEncCfg->m_Geo == 4 )
2828
0
    {
2829
0
      if( splitDir == 1 )
2830
0
      {
2831
0
        splitDir += 7;
2832
0
      }
2833
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2834
0
      {
2835
0
        splitDir++;
2836
0
      }
2837
0
      else
2838
0
      {
2839
0
        splitDir += 5;
2840
0
      }
2841
0
    }
2842
0
    else
2843
0
    {
2844
0
      splitDir++;
2845
0
    }
2846
0
  }
2847
2848
0
  if( comboList.list.empty() )
2849
0
  {
2850
0
    return false;
2851
0
  }
2852
2853
0
  comboList.sortByCost();
2854
0
  return true;
2855
0
}
2856
2857
double EncCu::calcLumaCost4MergePrediction( const TempCtx &ctxStart, const PelUnitBuf &predBuf, double lambda, CodingUnit &cu, DistParam &distParam )
2858
0
{
2859
0
  distParam.cur = predBuf.Y();
2860
0
  auto dist     = distParam.distFunc(distParam);
2861
2862
0
  m_CABACEstimator->getCtx() = ctxStart;
2863
0
  auto fracBits = xCalcPuMeBits( cu );
2864
2865
0
  double cost   = ( double ) dist + ( double ) fracBits * lambda;
2866
2867
0
  m_uiSadBestForQPA = std::min( dist, m_uiSadBestForQPA );
2868
2869
0
  return cost;
2870
0
}
2871
2872
//////////////////////////////////////////////////////////////////////////////////////////////
2873
// ibc merge/skip mode check
2874
void EncCu::xCheckRDCostIBCModeMerge2Nx2N(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner, const EncTestMode& encTestMode)
2875
23.9k
{
2876
23.9k
  assert(partitioner.chType != CH_C); // chroma IBC is derived
2877
23.9k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128) // disable IBC mode larger than 64x64
2878
0
  {
2879
0
    return;
2880
0
  }
2881
2882
23.9k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
2883
0
  {
2884
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
2885
0
    {
2886
0
      return;
2887
0
    }
2888
0
  }
2889
2890
23.9k
  const SPS& sps = *tempCS->sps;
2891
2892
23.9k
  tempCS->initStructData(encTestMode.qp);
2893
23.9k
  MergeCtx mergeCtx;
2894
2895
23.9k
  {
2896
    // first get merge candidates
2897
23.9k
    CodingUnit cu(tempCS->area);
2898
23.9k
    cu.cs = tempCS;
2899
23.9k
    cu.predMode = MODE_IBC;
2900
23.9k
    cu.slice = tempCS->slice;
2901
23.9k
    cu.tileIdx = m_tileIdx;
2902
23.9k
    cu.initPuData();
2903
23.9k
    cu.cs = tempCS;
2904
23.9k
    cu.mmvdSkip = false;
2905
23.9k
    cu.mmvdMergeFlag = false;
2906
23.9k
    cu.geo = false;
2907
23.9k
    CU::getIBCMergeCandidates(cu, mergeCtx);
2908
23.9k
  }
2909
23.9k
  int candHasNoResidual[MRG_MAX_NUM_CANDS];
2910
167k
  for (unsigned int ui = 0; ui < mergeCtx.numValidMergeCand; ui++)
2911
143k
  {
2912
143k
    candHasNoResidual[ui] = 0;
2913
143k
  }
2914
2915
23.9k
  bool                                        bestIsSkip = false;
2916
23.9k
  unsigned                                    numMrgSATDCand = mergeCtx.numValidMergeCand;
2917
23.9k
  static_vector<unsigned, MRG_MAX_NUM_CANDS>  RdModeList(MRG_MAX_NUM_CANDS);
2918
167k
  for (unsigned i = 0; i < MRG_MAX_NUM_CANDS; i++)
2919
143k
  {
2920
143k
    RdModeList[i] = i;
2921
143k
  }
2922
2923
  //{
2924
23.9k
  static_vector<double, MRG_MAX_NUM_CANDS>  candCostList(MRG_MAX_NUM_CANDS, MAX_DOUBLE);
2925
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
2926
23.9k
  {
2927
23.9k
    const double sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda();
2928
2929
23.9k
    CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType,partitioner.treeType), partitioner.chType);
2930
2931
23.9k
    partitioner.setCUData(cu);
2932
23.9k
    cu.slice = tempCS->slice;
2933
23.9k
    cu.tileIdx = m_tileIdx;
2934
23.9k
    cu.skip = false;
2935
23.9k
    cu.predMode = MODE_IBC;
2936
23.9k
    cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
2937
23.9k
    cu.qp = encTestMode.qp;
2938
23.9k
    cu.mmvdSkip = false;
2939
23.9k
    cu.geo = false;
2940
23.9k
    DistParam distParam;
2941
23.9k
    cu.initPuData();
2942
23.9k
    cu.mmvdMergeFlag = false;
2943
23.9k
    Picture* refPic = cu.slice->pic;
2944
23.9k
    const UnitArea localUnitArea(tempCS->area.chromaFormat, Area(cu.blocks[COMP_Y].x, cu.blocks[COMP_Y].y, tempCS->area.Y().width, tempCS->area.Y().height));
2945
23.9k
    const CompArea& compArea = localUnitArea.block(COMP_Y);
2946
23.9k
    const CPelBuf refBuf = refPic->getRecoBuf(compArea);
2947
23.9k
    const Pel* piRefSrch = refBuf.buf;
2948
2949
23.9k
      distParam = m_cRdCost.setDistParam(tempCS->getOrgBuf(COMP_Y), refBuf, sps.bitDepths[CH_L], DF_HAD);
2950
23.9k
    int refStride = refBuf.stride;
2951
2952
23.9k
    int numValidBv = mergeCtx.numValidMergeCand;
2953
167k
    for (unsigned int mergeCand = 0; mergeCand < mergeCtx.numValidMergeCand; mergeCand++)
2954
143k
    {
2955
143k
      mergeCtx.setMergeInfo(cu, mergeCand); // set bv info in merge mode
2956
143k
      const int cuPelX = cu.Y().x;
2957
143k
      const int cuPelY = cu.Y().y;
2958
143k
      int roiWidth     = cu.lwidth();
2959
143k
      int roiHeight    = cu.lheight();
2960
143k
      const int picWidth  = cu.cs->slice->pps->picWidthInLumaSamples;
2961
143k
      const int picHeight = cu.cs->slice->pps->picHeightInLumaSamples;
2962
143k
      const unsigned int lcuWidth = cu.cs->slice->sps->CTUSize;
2963
2964
143k
      Mv bv = cu.mv[0][0];
2965
143k
      bv.changePrecision( MV_PRECISION_INTERNAL, MV_PRECISION_INT);
2966
143k
      int xPred = bv.hor;
2967
143k
      int yPred = bv.ver;
2968
      
2969
143k
      if( !m_cInterSearch.searchBvIBC( cu, cuPelX, cuPelY, roiWidth, roiHeight, picWidth, picHeight, xPred, yPred, lcuWidth ) ) // not valid bv derived
2970
143k
      {
2971
143k
        numValidBv--;
2972
143k
        continue;
2973
143k
      }
2974
0
      CU::spanMotionInfo(cu);
2975
0
      distParam.cur.buf = piRefSrch + refStride * yPred + xPred;
2976
2977
0
      Distortion sad = distParam.distFunc(distParam);
2978
0
      unsigned int bitsCand = mergeCand + 1;
2979
0
      if (mergeCand == tempCS->sps->maxNumIBCMergeCand - 1)
2980
0
      {
2981
0
        bitsCand--;
2982
0
      }
2983
0
      double cost = (double)sad + (double)bitsCand * sqrtLambdaForFirstPass;
2984
2985
0
      updateCandList( mergeCand, cost, RdModeList, candCostList, numMrgSATDCand );
2986
0
    }
2987
2988
    // Try to limit number of candidates using SATD-costs
2989
23.9k
    if (numValidBv)
2990
0
    {
2991
0
      numMrgSATDCand = numValidBv;
2992
0
      for (unsigned int i = 1; i < numValidBv; i++)
2993
0
      {
2994
0
        if (candCostList[i] > MRG_FAST_RATIO[0] * candCostList[0])
2995
0
        {
2996
0
          numMrgSATDCand = i;
2997
0
          break;
2998
0
        }
2999
0
      }
3000
0
    }
3001
23.9k
    else
3002
23.9k
    {
3003
23.9k
      tempCS->dist = 0;
3004
23.9k
      tempCS->fracBits = 0;
3005
23.9k
      tempCS->cost = MAX_DOUBLE;
3006
23.9k
      tempCS->costDbOffset = 0;
3007
23.9k
      tempCS->initStructData(encTestMode.qp);
3008
23.9k
      return;
3009
23.9k
    }
3010
3011
0
    tempCS->initStructData(encTestMode.qp);
3012
0
  }
3013
  //}
3014
3015
3016
0
  const unsigned int iteration = 2;
3017
 // m_bestModeUpdated = tempCS->cost = bestCS->cost = false;
3018
  // 2. Pass: check candidates using full RD test
3019
0
  for (unsigned int numResidualPass = 0; numResidualPass < iteration; numResidualPass++)
3020
0
  {
3021
0
    for (unsigned int mrgHADIdx = 0; mrgHADIdx < numMrgSATDCand; mrgHADIdx++)
3022
0
    {
3023
0
      unsigned int mergeCand = RdModeList[mrgHADIdx];
3024
0
      if (!(numResidualPass == 1 && candHasNoResidual[mergeCand] == 1))
3025
0
      {
3026
0
        if (!(bestIsSkip && (numResidualPass == 0)))
3027
0
        {
3028
0
          {
3029
3030
            // first get merge candidates
3031
0
            CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, (const ChannelType)partitioner.chType,partitioner.treeType), (const ChannelType)partitioner.chType);
3032
3033
0
            partitioner.setCUData(cu);
3034
0
            cu.slice = tempCS->slice;
3035
0
            cu.tileIdx = m_tileIdx;
3036
0
            cu.skip = false;
3037
0
            cu.predMode = MODE_IBC;
3038
0
            cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3039
0
            cu.qp = encTestMode.qp;
3040
0
            cu.sbtInfo = 0;
3041
0
            cu.initPuData();
3042
0
            cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3043
0
            cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3044
0
            cu.mmvdSkip = false;
3045
0
            cu.mmvdMergeFlag = false;
3046
0
            cu.geo = false;
3047
0
            mergeCtx.setMergeInfo(cu, mergeCand);
3048
0
            CU::spanMotionInfo(cu);
3049
3050
0
            assert(mergeCtx.mrgTypeNeighbours[mergeCand] == MRG_TYPE_IBC);
3051
0
            const bool chroma = !CU::isSepTree(cu);
3052
3053
            //  MC
3054
0
            cu.mcControl = chroma ? 0: 2;
3055
0
            m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3056
0
            m_CABACEstimator->getCtx() = m_CurrCtx->start;
3057
3058
0
            m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, (numResidualPass != 0));
3059
0
            cu.mcControl = 0;
3060
0
            xEncodeDontSplit(*tempCS, partitioner);
3061
0
            xCheckDQP(*tempCS, partitioner);
3062
0
            xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3063
3064
0
            tempCS->initStructData(encTestMode.qp);
3065
0
          }
3066
3067
0
          if (m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip)
3068
0
          {
3069
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType) == NULL)
3070
0
              bestIsSkip = 0;
3071
0
            else
3072
0
              bestIsSkip = bestCS->getCU(partitioner.chType, partitioner.treeType)->rootCbf == 0;
3073
0
          }
3074
0
        }
3075
0
      }
3076
0
    }
3077
0
  }
3078
0
}
3079
3080
void EncCu::xCheckRDCostIBCMode(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner,
3081
  const EncTestMode& encTestMode)
3082
23.9k
{
3083
23.9k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128)   // disable IBC mode larger than 64x64
3084
0
  {
3085
0
    return;
3086
0
  }
3087
23.9k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
3088
0
  {
3089
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3090
0
    {
3091
0
      return;
3092
0
    }
3093
0
  }
3094
3095
23.9k
  tempCS->initStructData(encTestMode.qp);
3096
3097
23.9k
  CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType, partitioner.treeType), partitioner.chType);
3098
3099
23.9k
  partitioner.setCUData(cu);
3100
23.9k
  cu.slice = tempCS->slice;
3101
23.9k
  cu.tileIdx = m_tileIdx;
3102
23.9k
  cu.skip = false;
3103
23.9k
  cu.predMode = MODE_IBC;
3104
23.9k
  cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3105
23.9k
  cu.qp = encTestMode.qp;
3106
23.9k
  cu.initPuData();
3107
23.9k
  cu.imv = IMV_OFF;
3108
23.9k
  cu.sbtInfo = 0;
3109
23.9k
  cu.mmvdSkip = false;
3110
23.9k
  cu.mmvdMergeFlag = false;
3111
3112
23.9k
  cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3113
23.9k
  cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3114
3115
23.9k
  cu.interDir = 1; // use list 0 for IBC mode
3116
23.9k
  cu.refIdx[REF_PIC_LIST_0] = MAX_NUM_REF; // last idx in the list
3117
23.9k
  bool bValid = m_cInterSearch.predIBCSearch(cu, partitioner);
3118
3119
23.9k
  if (bValid)
3120
21.3k
  {
3121
21.3k
    CU::spanMotionInfo(cu);
3122
21.3k
    const bool chroma = !CU::isSepTree(cu);
3123
    //  MC
3124
21.3k
    cu.mcControl = chroma ? 0 : 2;
3125
21.3k
    m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3126
3127
21.3k
    m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, false);
3128
21.3k
    cu.mcControl = 0;
3129
3130
21.3k
    xEncodeDontSplit(*tempCS, partitioner);
3131
21.3k
    xCheckDQP(*tempCS, partitioner);
3132
21.3k
    xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3133
21.3k
  } // bValid
3134
2.54k
  else
3135
2.54k
  {
3136
2.54k
    tempCS->dist = 0;
3137
2.54k
    tempCS->fracBits = 0;
3138
2.54k
    tempCS->cost = MAX_DOUBLE;
3139
2.54k
    tempCS->costDbOffset = 0;
3140
2.54k
  }
3141
23.9k
}
3142
3143
void EncCu::xCheckRDCostInter( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
3144
0
{
3145
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD, tempCS, partitioner.chType );
3146
0
  tempCS->initStructData( encTestMode.qp );
3147
3148
0
  m_cInterSearch.setAffineModeSelected( false );
3149
3150
0
  m_cInterSearch.resetBufferedUniMotions();
3151
3152
0
  int bcwLoopNum = BCW_NUM;
3153
3154
0
  if( tempCS->area.Y().area() < BCW_SIZE_CONSTRAINT || !tempCS->slice->isInterB() || !tempCS->sps->BCW )
3155
0
  {
3156
0
    bcwLoopNum = 1;
3157
0
  }
3158
  
3159
0
  double curBestCost = bestCS->cost;
3160
0
  double equBcwCost = MAX_DOUBLE;
3161
3162
0
  for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3163
0
  {
3164
0
    if( m_pcEncCfg->m_BCW == 2 )
3165
0
    {
3166
0
      bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3167
0
      uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3168
3169
0
      if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3170
0
      {
3171
0
        continue;
3172
0
      }
3173
0
    }
3174
    
3175
0
    if( !tempCS->slice->checkLDC )
3176
0
    {
3177
0
      if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3178
0
      {
3179
0
        continue;
3180
0
      }
3181
0
    }
3182
  
3183
0
    CodingUnit &cu      = tempCS->addCU( tempCS->area, partitioner.chType );
3184
3185
0
    partitioner.setCUData( cu );
3186
0
    cu.slice            = tempCS->slice;
3187
0
    cu.tileIdx          = m_tileIdx;
3188
0
    cu.skip             = false;
3189
0
    cu.mmvdSkip         = false;
3190
0
    cu.predMode         = MODE_INTER;
3191
0
    cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
3192
0
    cu.qp               = encTestMode.qp;
3193
0
    cu.initPuData();
3194
3195
0
    cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3196
0
    uint8_t bcwIdx = cu.BcwIdx;
3197
0
    bool testBcw = (bcwIdx != BCW_DEFAULT);
3198
3199
0
    bool StopInterRes = (m_pcEncCfg->m_FastInferMerge >> 3) & 1;
3200
0
    StopInterRes &= bestCS->slice->TLayer > (m_pcEncCfg->m_maxTLayer - (m_pcEncCfg->m_FastInferMerge & 7));
3201
0
    double bestCostInter = StopInterRes ? m_mergeBestSATDCost : MAX_DOUBLE;
3202
3203
0
    bool stopTest = m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3204
3205
0
    if (StopInterRes && (bestCostInter != m_mergeBestSATDCost))
3206
0
    {
3207
0
      int L = (cu.slice->TLayer <= 2) ? 0 : (cu.slice->TLayer - 2);
3208
0
      if ((bestCostInter > MRG_FAST_RATIOMYV[L] * m_mergeBestSATDCost))
3209
0
      {
3210
0
        stopTest = true;
3211
0
      }
3212
0
    }
3213
3214
0
    if( !stopTest )
3215
0
    {
3216
0
      bcwIdx   = CU::getValidBcwIdx(cu);
3217
0
      stopTest = testBcw && bcwIdx == BCW_DEFAULT;
3218
0
    }
3219
    
3220
0
    if( stopTest )
3221
0
    {
3222
0
      tempCS->initStructData(encTestMode.qp);
3223
0
      continue;
3224
0
    }
3225
3226
0
    CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3227
        
3228
0
    xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3229
    
3230
0
    if( bcwIdx == BCW_DEFAULT )
3231
0
    {
3232
0
      m_cInterSearch.setAffineModeSelected( bestCS->cus.front()->affine && !bestCS->cus.front()->mergeFlag );
3233
0
    }
3234
3235
0
    tempCS->initStructData(encTestMode.qp);
3236
  
3237
0
    double skipTH = MAX_DOUBLE;
3238
0
    skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3239
0
    if( equBcwCost > curBestCost * skipTH )
3240
0
    {
3241
0
      break;
3242
0
    }
3243
3244
0
    if( m_pcEncCfg->m_BCW == 2 )
3245
0
    {
3246
0
      if( ( cu.interDir != 3 && testBcw == 0 && ! m_pcEncCfg->m_picReordering )
3247
0
         || ( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) ) )
3248
0
      {
3249
0
        break;
3250
0
      }
3251
0
    }
3252
0
  }
3253
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3254
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3255
0
}
3256
3257
void EncCu::xCheckRDCostInterIMV(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode)
3258
0
{
3259
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD_IMV, tempCS, partitioner.chType );
3260
0
  bool Test_AMVR = m_pcEncCfg->m_AMVRspeed ? true: false;
3261
0
  if (m_pcEncCfg->m_AMVRspeed > 2 && m_pcEncCfg->m_AMVRspeed < 5 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3262
0
  {
3263
0
    Test_AMVR = false;
3264
0
  }
3265
0
  else if (m_pcEncCfg->m_AMVRspeed > 4 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->mergeFlag && !bestCS->getCU(partitioner.chType, partitioner.treeType)->ciip)
3266
0
  {
3267
0
    Test_AMVR = false;
3268
0
  }
3269
0
  bool Do_Limit = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 4 || m_pcEncCfg->m_AMVRspeed == 6) ? true : false;
3270
0
  bool Do_OnceRes = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 7) ? true : false;
3271
3272
0
  if( Test_AMVR )
3273
0
  {
3274
0
    double Fpel_cost    = m_pcEncCfg->m_AMVRspeed == 1 ? MAX_DOUBLE*0.5 : MAX_DOUBLE;
3275
0
    double costCurStart = m_pcEncCfg->m_AMVRspeed == 1 ? m_modeCtrl.comprCUCtx->bestCostNoImv : bestCS->cost;
3276
0
    double costCur      = MAX_DOUBLE;
3277
0
    double bestCostIMV  = MAX_DOUBLE;
3278
3279
0
    if (Do_OnceRes)
3280
0
    {
3281
0
      costCurStart = xCalcDistortion(bestCS, partitioner.chType, bestCS->sps->bitDepths[CH_L], 0);
3282
0
      Fpel_cost = costCurStart;
3283
0
      tempCS->initSubStructure(*m_pTempCS2, partitioner.chType, partitioner.currArea(), false);
3284
0
    }
3285
3286
0
    CodingStructure *tempCSbest = m_pTempCS2;
3287
3288
0
    m_cInterSearch.setAffineModeSelected( false );
3289
3290
0
    m_cInterSearch.resetBufferedUniMotions();
3291
3292
0
    int bcwLoopNum = (tempCS->slice->isInterB() ? BCW_NUM : 1);
3293
0
    bcwLoopNum = (tempCS->sps->BCW ? bcwLoopNum : 1);
3294
3295
0
    if( tempCS->area.lwidth() * tempCS->area.lheight() < BCW_SIZE_CONSTRAINT )
3296
0
    {
3297
0
      bcwLoopNum = 1;
3298
0
    }
3299
3300
0
    for (int i = 1; i <= IMV_HPEL; i++)
3301
0
    {
3302
0
      double curBestCost = bestCS->cost;
3303
0
      double equBcwCost  = MAX_DOUBLE;
3304
3305
0
      for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3306
0
      {
3307
0
        if( m_pcEncCfg->m_BCW == 2 )
3308
0
        {
3309
0
          bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3310
0
          uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3311
3312
0
          if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3313
0
          {
3314
0
            continue;
3315
0
          }
3316
          
3317
0
          if( tempCS->slice->checkLDC && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT
3318
0
            && (m_bestBcwIdx[0] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[0])
3319
0
            && (m_bestBcwIdx[1] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[1]))
3320
0
          {
3321
0
            continue;
3322
0
          }
3323
0
        }
3324
3325
0
        if( !tempCS->slice->checkLDC )
3326
0
        {
3327
0
          if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3328
0
          {
3329
0
            continue;
3330
0
          }
3331
0
        }
3332
3333
0
        bool testBcw;
3334
0
        uint8_t bcwIdx;
3335
0
        bool isEqualUni = false;
3336
3337
0
        if (i > IMV_FPEL)
3338
0
        {
3339
0
          bool nextimv = false;
3340
0
          double stopCost = i == IMV_HPEL ? 1.25 : 1.06;
3341
0
          if (Fpel_cost > stopCost * costCurStart)
3342
0
          {
3343
0
            nextimv = true;
3344
0
          }
3345
0
          if ( m_pcEncCfg->m_AMVRspeed == 1 )
3346
0
          {
3347
0
            costCurStart = bestCS->cost;
3348
0
          }
3349
0
          if (nextimv)
3350
0
          {
3351
0
            continue;
3352
0
          }
3353
0
        }
3354
3355
0
        bool Do_Search = Do_OnceRes ? false : true;
3356
3357
0
        if (Do_Limit)
3358
0
        {
3359
0
          Do_Search = i == IMV_FPEL ? true : false;
3360
3361
0
          if (i == IMV_HPEL)
3362
0
          {
3363
0
            if (bestCS->slice->TLayer > 3)
3364
0
            {
3365
0
              continue;
3366
0
            }
3367
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType)->imv != 0)
3368
0
            {
3369
0
              Do_Search = true; //do_est
3370
0
            }
3371
0
          }
3372
0
          if (bestCS->getCU(partitioner.chType, partitioner.treeType)->mmvdMergeFlag || bestCS->getCU(partitioner.chType, partitioner.treeType)->geo)
3373
0
          {
3374
0
            Do_Search = true;
3375
0
          }
3376
0
        }
3377
0
        tempCS->initStructData(encTestMode.qp);
3378
3379
0
        if (!Do_Search)
3380
0
        {
3381
0
          tempCS->copyStructure(*bestCS, partitioner.chType, TREE_D);
3382
0
        }
3383
0
        tempCS->dist = 0;
3384
0
        tempCS->fracBits = 0;
3385
0
        tempCS->cost = MAX_DOUBLE;
3386
0
        CodingUnit &cu = (Do_Search) ? tempCS->addCU(tempCS->area, partitioner.chType) : *tempCS->getCU(partitioner.chType, partitioner.treeType);
3387
0
        if (Do_Search)
3388
0
        {
3389
0
          partitioner.setCUData(cu);
3390
0
          cu.slice = tempCS->slice;
3391
0
          cu.tileIdx = m_tileIdx;
3392
0
          cu.skip = false;
3393
0
          cu.mmvdSkip = false;
3394
0
          cu.predMode = MODE_INTER;
3395
0
          cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3396
0
          cu.qp = encTestMode.qp;
3397
3398
0
          cu.initPuData();
3399
3400
0
          cu.imv = i;
3401
3402
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3403
0
          bcwIdx    = cu.BcwIdx;
3404
0
          testBcw   = (bcwIdx != BCW_DEFAULT);
3405
3406
0
          cu.interDir = 10;
3407
          
3408
0
          double bestCostInter = MAX_DOUBLE;
3409
0
          m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3410
          
3411
0
          if ( cu.interDir <= 3 )
3412
0
          {
3413
0
            bcwIdx = CU::getValidBcwIdx(cu);
3414
0
          }
3415
0
          else
3416
0
          {
3417
0
            continue;
3418
0
          }
3419
          
3420
0
          if( testBcw && bcwIdx == BCW_DEFAULT ) // Enabled Bcw but the search results is uni.
3421
0
          {
3422
0
            continue;
3423
0
          }
3424
0
          CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3425
3426
0
          if( m_pcEncCfg->m_BCW == 2 )
3427
0
          {
3428
0
            if( cu.interDir != 3 && testBcw == 0 )
3429
0
            {
3430
0
              isEqualUni = true;
3431
0
            }
3432
0
          }
3433
3434
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3435
0
          {
3436
0
            continue;
3437
0
          }
3438
0
        }
3439
0
        else
3440
0
        {
3441
0
          cu.smvdMode = 0;
3442
0
          cu.affine = false;
3443
0
          cu.imv = i ;
3444
0
          CU::resetMVDandMV2Int(cu);
3445
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3446
0
          {
3447
0
            continue;
3448
0
          }
3449
3450
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3451
3452
0
          cu.mvRefine = true;
3453
0
          m_cInterSearch.motionCompensation(cu, tempCS->getPredBuf() );
3454
0
          cu.mvRefine = false;
3455
0
        }
3456
3457
0
        if( Do_OnceRes )
3458
0
        {
3459
0
          costCur = xCalcDistortion(tempCS, partitioner.chType, tempCS->sps->bitDepths[CH_L], cu.imv );
3460
0
          if (costCur < bestCostIMV)
3461
0
          {
3462
0
            bestCostIMV = costCur;
3463
0
            tempCSbest->getPredBuf().copyFrom(tempCS->getPredBuf());
3464
0
            tempCSbest->clearCUs();
3465
0
            tempCSbest->clearTUs();
3466
0
            tempCSbest->copyStructure(*tempCS, partitioner.chType, TREE_D);
3467
0
          }
3468
0
          if (i > IMV_FPEL)
3469
0
          {
3470
0
            costCurStart = costCurStart > costCur ? costCur : costCurStart;
3471
0
          }
3472
0
        }
3473
0
        else
3474
0
        {
3475
0
          xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3476
0
          costCur = tempCS->cost;
3477
3478
0
          if (i > IMV_FPEL)
3479
0
          {
3480
0
            costCurStart = bestCS->cost;
3481
0
          }
3482
0
        }
3483
3484
0
        if (i == IMV_FPEL)
3485
0
        {
3486
0
           Fpel_cost = costCur;
3487
0
        }
3488
3489
0
        double skipTH = MAX_DOUBLE;
3490
0
        skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3491
0
        if( equBcwCost > curBestCost * skipTH )
3492
0
        {
3493
0
          break;
3494
0
        }
3495
3496
0
        if( m_pcEncCfg->m_BCW == 2 )
3497
0
        {
3498
0
          if( isEqualUni == true && ! m_pcEncCfg->m_picReordering )
3499
0
          {
3500
0
            break;
3501
0
          }
3502
0
          if( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) )
3503
0
          {
3504
0
            break;
3505
0
          }
3506
0
        }
3507
0
      }
3508
0
    }
3509
3510
0
    if (Do_OnceRes && (bestCostIMV != MAX_DOUBLE))
3511
0
    {
3512
0
      CodingStructure* CSCandBest = tempCSbest;
3513
0
      tempCS->initStructData(bestCS->currQP[partitioner.chType]);
3514
0
      tempCS->copyStructure(*CSCandBest, partitioner.chType, TREE_D);
3515
0
      tempCS->getPredBuf().copyFrom(tempCSbest->getPredBuf());
3516
0
      tempCS->dist = 0;
3517
0
      tempCS->fracBits = 0;
3518
0
      tempCS->cost = MAX_DOUBLE;
3519
3520
0
      xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, NULL);
3521
0
    }
3522
3523
0
    tempCS->initStructData(encTestMode.qp);
3524
0
  }
3525
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3526
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3527
0
}
3528
3529
void EncCu::xCalDebCost( CodingStructure &cs, Partitioner &partitioner )
3530
99.3k
{
3531
99.3k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_DEBLOCK_FILTER, &cs, partitioner.chType );
3532
99.3k
  if ( cs.slice->deblockingFilterDisable )
3533
0
  {
3534
0
    return;
3535
0
  }
3536
3537
99.3k
  const ChromaFormat format = cs.area.chromaFormat;
3538
99.3k
  CodingUnit*            cu = cs.getCU(partitioner.chType, partitioner.treeType);
3539
99.3k
  const Position    lumaPos = cu->Y().valid() ? cu->Y().pos() : recalcPosition( format, cu->chType, CH_L, cu->blocks[cu->chType].pos() );
3540
99.3k
  bool    topEdgeAvai = lumaPos.y > 0 && ((lumaPos.y % 4) == 0);
3541
99.3k
  bool   leftEdgeAvai = lumaPos.x > 0 && ((lumaPos.x % 4) == 0);
3542
3543
99.3k
  if( ! ( topEdgeAvai || leftEdgeAvai ))
3544
20.6k
  {
3545
20.6k
    return;
3546
20.6k
  }
3547
3548
78.7k
  ComponentID compStr = ( CU::isSepTree(*cu) && !isLuma( partitioner.chType ) ) ? COMP_Cb : COMP_Y;
3549
78.7k
  ComponentID compEnd = (( CU::isSepTree(*cu) && isLuma( partitioner.chType )) || cu->chromaFormat == VVENC_CHROMA_400 ) ? COMP_Y : COMP_Cr;
3550
78.7k
  const UnitArea currCsArea = clipArea( CS::getArea( cs, cs.area, partitioner.chType, partitioner.treeType ), *cs.picture );
3551
3552
78.7k
  PelStorage&  picDbBuf = m_dbBuffer; //th we could reduce the buffer size and do some relocate
3553
3554
  //deblock neighbour pixels
3555
78.7k
  const Size     lumaSize = cu->Y().valid() ? cu->Y().size() : recalcSize( format, cu->chType, CH_L, cu->blocks[cu->chType].size() );
3556
3557
78.7k
  int verOffset = lumaPos.y > 7 ? 8 : 4;
3558
78.7k
  int horOffset = lumaPos.x > 7 ? 8 : 4;
3559
3560
78.7k
  LoopFilter::calcFilterStrengths( *cu, true );
3561
3562
78.7k
  if( m_EDO == 2 && CS::isDualITree( cs ) && isLuma( partitioner.chType ) )
3563
43.6k
  {
3564
43.6k
    m_cLoopFilter.getMaxFilterLength( *cu, verOffset, horOffset );
3565
3566
43.6k
    if( 0== (verOffset + horOffset) )
3567
0
    {
3568
0
      return;
3569
0
    }
3570
3571
43.6k
    topEdgeAvai  &= verOffset != 0;
3572
43.6k
    leftEdgeAvai &= horOffset != 0;
3573
43.6k
  }
3574
3575
78.7k
  const UnitArea  areaTop  = UnitArea( format, Area( lumaPos.x,             lumaPos.y - verOffset, lumaSize.width, verOffset       ) );
3576
78.7k
  const UnitArea  areaLeft = UnitArea( format, Area( lumaPos.x - horOffset, lumaPos.y,             horOffset,      lumaSize.height ) );
3577
3578
192k
  for ( int compIdx = compStr; compIdx <= compEnd; compIdx++ )
3579
113k
  {
3580
113k
    ComponentID compId = (ComponentID)compIdx;
3581
3582
    //Copy current CU's reco to Deblock Pic Buffer
3583
113k
    const CompArea&  compArea = currCsArea.block( compId );
3584
113k
    CompArea         locArea  = compArea;
3585
113k
    locArea.x -= cu->blocks[compIdx].x;
3586
113k
    locArea.y -= cu->blocks[compIdx].y;
3587
113k
    PelBuf dbReco = picDbBuf.getBuf( locArea );
3588
113k
      PelBuf reco = cs.getRecoBuf( compId );
3589
113k
      dbReco.copyFrom( reco );
3590
    //left neighbour
3591
113k
    if ( leftEdgeAvai )
3592
79.3k
    {
3593
79.3k
      const CompArea&  compArea = areaLeft.block(compId);
3594
79.3k
      CompArea         locArea = compArea;
3595
79.3k
      locArea.x -= cu->blocks[compIdx].x;
3596
79.3k
      locArea.y -= cu->blocks[compIdx].y;
3597
79.3k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3598
79.3k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3599
79.3k
      }
3600
    //top neighbour
3601
113k
    if ( topEdgeAvai )
3602
82.6k
    {
3603
82.6k
      const CompArea&  compArea = areaTop.block( compId );
3604
82.6k
      CompArea         locArea = compArea;
3605
82.6k
      locArea.x -= cu->blocks[compIdx].x;
3606
82.6k
      locArea.y -= cu->blocks[compIdx].y;
3607
82.6k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3608
82.6k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3609
82.6k
      }
3610
113k
    }
3611
3612
78.7k
  ChannelType dbChType = CU::isSepTree(*cu) ? partitioner.chType : MAX_NUM_CH;
3613
3614
78.7k
  CHECK( CU::isSepTree(*cu) && !cu->Y().valid() && partitioner.chType == CH_L, "xxx" );
3615
3616
78.7k
  if( cu->Y() .valid() ) m_cLoopFilter.setOrigin( CH_L, cu->lumaPos() );
3617
78.7k
  if( cu->chromaFormat != VVENC_CHROMA_400 && cu->Cb().valid() ) m_cLoopFilter.setOrigin( CH_C, cu->chromaPos() );
3618
3619
  //deblock
3620
78.7k
  if( leftEdgeAvai )
3621
57.8k
  {
3622
57.8k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_VER, m_dbBuffer );
3623
57.8k
  }
3624
3625
78.7k
  if( topEdgeAvai )
3626
59.4k
  {
3627
59.4k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_HOR, m_dbBuffer );
3628
59.4k
  }
3629
3630
  //calculate difference between DB_before_SSE and DB_after_SSE for neighbouring CUs
3631
78.7k
  Distortion distBeforeDb = 0, distAfterDb = 0, distCur = 0;
3632
192k
  for (int compIdx = compStr; compIdx <= compEnd; compIdx++)
3633
113k
  {
3634
113k
    ComponentID compId = (ComponentID)compIdx;
3635
113k
    {
3636
113k
      CompArea compArea = currCsArea.block( compId );
3637
113k
      CompArea         locArea  = compArea;
3638
113k
      locArea.x -= cu->blocks[compIdx].x;
3639
113k
      locArea.y -= cu->blocks[compIdx].y;
3640
113k
      CPelBuf reco      = picDbBuf.getBuf( locArea );
3641
113k
      CPelBuf org       = cs.getOrgBuf( compId );
3642
113k
      distCur += xGetDistortionDb( cs, org, reco, compArea, false );
3643
113k
    }
3644
3645
113k
    if ( leftEdgeAvai )
3646
79.3k
    {
3647
79.3k
      const CompArea&  compArea = areaLeft.block( compId );
3648
79.3k
      CompArea         locArea  = compArea;
3649
79.3k
      locArea.x -= cu->blocks[compIdx].x;
3650
79.3k
      locArea.y -= cu->blocks[compIdx].y;
3651
79.3k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3652
79.3k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3653
0
      {
3654
0
        org = cs.picture->getFiltOrigBuf( compArea );
3655
0
      }
3656
79.3k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3657
79.3k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3658
79.3k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3659
79.3k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3660
79.3k
    }
3661
3662
113k
    if ( topEdgeAvai )
3663
82.6k
    {
3664
82.6k
      const CompArea&  compArea = areaTop.block( compId );
3665
82.6k
      CompArea         locArea  = compArea;
3666
82.6k
      locArea.x -= cu->blocks[compIdx].x;
3667
82.6k
      locArea.y -= cu->blocks[compIdx].y;
3668
82.6k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3669
82.6k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3670
0
      {
3671
0
        org = cs.picture->getFiltOrigBuf( compArea );
3672
0
      }
3673
82.6k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3674
82.6k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3675
82.6k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3676
82.6k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3677
82.6k
    }
3678
113k
  }
3679
3680
  //updated cost
3681
78.7k
  int64_t distTmp = distCur - cs.dist + distAfterDb - distBeforeDb;
3682
78.7k
  cs.costDbOffset = distTmp < 0 ? -m_cRdCost.calcRdCost( 0, -distTmp ) : m_cRdCost.calcRdCost( 0, distTmp );
3683
78.7k
}
3684
3685
Distortion EncCu::xGetDistortionDb(CodingStructure &cs, CPelBuf& org, CPelBuf& reco, const CompArea& compArea, bool beforeDb)
3686
437k
{
3687
437k
  Distortion dist;
3688
437k
  const ComponentID compID = compArea.compID;
3689
3690
437k
  dist = m_cRdCost.getDistPart(org, reco, cs.sps->bitDepths[toChannelType(compID)], compID, DF_SSE);
3691
437k
  return dist;
3692
437k
}
3693
3694
bool checkValidMvs( const CodingUnit& cu)
3695
0
{
3696
  // clang-format off
3697
0
  const int affineShiftTab[3] =
3698
0
  {
3699
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3700
0
    MV_PRECISION_INTERNAL - MV_PRECISION_SIXTEENTH,
3701
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT
3702
0
  };
3703
3704
0
  const int normalShiftTab[NUM_IMV_MODES] =
3705
0
  {
3706
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3707
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT,
3708
0
    MV_PRECISION_INTERNAL - MV_PRECISION_4PEL,
3709
0
    MV_PRECISION_INTERNAL - MV_PRECISION_HALF,
3710
0
  };
3711
  // clang-format on
3712
3713
0
  int mvShift;
3714
3715
0
  for (int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++)
3716
0
  {
3717
0
    if (cu.refIdx[refList] >= 0)
3718
0
    {
3719
0
      if (!cu.affine)
3720
0
      {
3721
0
        mvShift = normalShiftTab[cu.imv];
3722
0
        Mv signaledmvd(cu.mvd[refList][0].hor >> mvShift, cu.mvd[refList][0].ver >> mvShift);
3723
0
        if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3724
0
          return false;
3725
0
      }
3726
0
      else
3727
0
      {
3728
0
        for (int ctrlP = 1 + (cu.affineType == AFFINEMODEL_6PARAM); ctrlP >= 0; ctrlP--)
3729
0
        {
3730
0
          mvShift = affineShiftTab[cu.imv];
3731
0
          Mv signaledmvd(cu.mvd[refList][ctrlP].hor >> mvShift, cu.mvd[refList][ctrlP].ver >> mvShift);
3732
0
          if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3733
0
            return false;;
3734
0
        }
3735
0
      }
3736
0
    }
3737
0
  }
3738
  // avoid MV exceeding 18-bit dynamic range
3739
0
  const int maxMv = 1 << 17;
3740
0
  if (!cu.affine && !cu.mergeFlag)
3741
0
  {
3742
0
    if(    ( cu.refIdx[ 0 ] >= 0 && ( cu.mv[ 0 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 0 ][ 0 ].getAbsVer() >= maxMv ) )
3743
0
        || ( cu.refIdx[ 1 ] >= 0 && ( cu.mv[ 1 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 1 ][ 0 ].getAbsVer() >= maxMv ) ) )
3744
0
    {
3745
0
      return false;
3746
0
    }
3747
0
  }
3748
0
  if( cu.affine && !cu.mergeFlag )
3749
0
  {
3750
0
    for( int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++ )
3751
0
    {
3752
0
      if( cu.refIdx[ refList ] >= 0 )
3753
0
      {
3754
0
        for( int ctrlP = 1 + ( cu.affineType == AFFINEMODEL_6PARAM ); ctrlP >= 0; ctrlP-- )
3755
0
        {
3756
0
          if( cu.mv[ refList ][ ctrlP ].getAbsHor() >= maxMv || cu.mv[ refList ][ ctrlP ].getAbsVer() >= maxMv )
3757
0
          {
3758
0
            return false;
3759
0
          }
3760
0
        }
3761
0
      }
3762
0
    }
3763
0
  }
3764
0
  return true;
3765
0
}
3766
3767
3768
void EncCu::xEncodeInterResidual( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, int residualPass, bool* bestHasNonResi, double* equBcwCost )
3769
0
{
3770
0
  if( residualPass == 1 && encTestMode.lossless )
3771
0
  {
3772
0
    return;
3773
0
  }
3774
3775
0
  CodingUnit*            cu        = tempCS->getCU( partitioner.chType, partitioner.treeType );
3776
0
  double   bestCostInternal        = MAX_DOUBLE;
3777
3778
0
  if( !checkValidMvs( *cu ) )
3779
0
    return;
3780
3781
0
  double  currBestCost = MAX_DOUBLE;
3782
3783
  // For SBT
3784
0
  double     bestCost          = bestCS->cost;
3785
0
  double     bestCostBegin     = bestCS->cost;
3786
0
  const CodingUnit* prevBestCU = bestCS->getCU( partitioner.chType, partitioner.treeType );
3787
0
  uint8_t    prevBestSbt       = ( prevBestCU == nullptr ) ? 0 : prevBestCU->sbtInfo;
3788
0
  Distortion sbtOffDist        = 0;
3789
0
  bool       sbtOffRootCbf     = 0;
3790
0
  double     sbtOffCost        = MAX_DOUBLE;
3791
0
  uint8_t    currBestSbt       = 0;
3792
0
  uint8_t    histBestSbt       = MAX_UCHAR;
3793
0
  Distortion curPuSse          = MAX_DISTORTION;
3794
0
  uint8_t    numRDOTried       = 0;
3795
0
  bool       doPreAnalyzeResi  = false;
3796
0
  const bool mtsAllowed        =   tempCS->sps->MTSInter && cu->Y().maxDim() <= MTS_INTER_MAX_CU_SIZE;
3797
0
  const uint8_t sbtAllowed     = ( tempCS->pps->picWidthInLumaSamples < SBT_FAST64_WIDTH_THRESHOLD || m_pcEncCfg->m_SBT > 1 ) && cu->Y().maxDim() > 32 ? 0 : CU::checkAllowedSbt(*cu);
3798
3799
0
  if( sbtAllowed )
3800
0
  {
3801
    //SBT resolution-dependent fast algorithm: not try size-64 SBT in RDO for low-resolution sequences (now resolution below HD)
3802
0
    doPreAnalyzeResi = ( sbtAllowed || mtsAllowed ) && residualPass == 0;
3803
0
    m_cInterSearch.getBestSbt( tempCS, cu, histBestSbt, curPuSse, sbtAllowed, doPreAnalyzeResi, mtsAllowed );
3804
0
  }
3805
3806
0
  cu->skip    = false;
3807
0
  cu->sbtInfo = 0;
3808
3809
0
  const bool skipResidual = residualPass == 1;
3810
0
  if( skipResidual || histBestSbt == MAX_UCHAR || !CU::isSbtMode( histBestSbt ) )
3811
0
  {
3812
0
    m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3813
0
    xEncodeDontSplit( *tempCS, partitioner );
3814
0
    xCheckDQP       ( *tempCS, partitioner );
3815
3816
0
    if( NULL != bestHasNonResi && (bestCostInternal > tempCS->cost) )
3817
0
    {
3818
0
      bestCostInternal = tempCS->cost;
3819
0
      if( !cu->ciip )
3820
0
        *bestHasNonResi = !cu->rootCbf;
3821
0
    }
3822
3823
0
    if( cu->rootCbf == false )
3824
0
    {
3825
0
      if( cu->ciip )
3826
0
      {
3827
0
        tempCS->cost = MAX_DOUBLE;
3828
0
        tempCS->costDbOffset = 0;
3829
0
        return;
3830
0
      }
3831
0
    }
3832
0
    currBestCost = tempCS->cost;
3833
0
    if( sbtAllowed )
3834
0
    {
3835
0
      sbtOffCost    = tempCS->cost;
3836
0
      sbtOffDist    = tempCS->dist;
3837
0
      sbtOffRootCbf = cu->rootCbf;
3838
0
      currBestSbt   = cu->firstTU->mtsIdx[COMP_Y] > MTS_SKIP ? SBT_OFF_MTS : SBT_OFF_DCT;
3839
0
      numRDOTried  += mtsAllowed ? 2 : 1;
3840
0
    }
3841
3842
0
    DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3843
0
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3844
3845
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3846
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3847
0
  }
3848
3849
0
  if( sbtAllowed && ( m_pcEncCfg->m_SBT == 1 || sbtOffRootCbf ) )
3850
0
  {
3851
0
    bool swapped = false; // avoid unwanted data copy
3852
0
    uint8_t numSbtRdo = CU::numSbtModeRdo( sbtAllowed );
3853
    //early termination if all SBT modes are not allowed
3854
    //normative
3855
0
    if( !sbtAllowed || skipResidual )
3856
0
    {
3857
0
      numSbtRdo = 0;
3858
0
    }
3859
    //fast algorithm
3860
0
    if( ( histBestSbt != MAX_UCHAR && !CU::isSbtMode( histBestSbt ) ) || m_cInterSearch.getSkipSbtAll() )
3861
0
    {
3862
0
      numSbtRdo = 0;
3863
0
    }
3864
0
    if( bestCost != MAX_DOUBLE && sbtOffCost != MAX_DOUBLE )
3865
0
    {
3866
0
      double th = 1.07;
3867
0
      if( !( prevBestSbt == 0 || m_sbtCostSave[0] == MAX_DOUBLE ) )
3868
0
      {
3869
0
        assert( m_sbtCostSave[1] <= m_sbtCostSave[0] );
3870
0
        th *= ( m_sbtCostSave[0] / m_sbtCostSave[1] );
3871
0
      }
3872
0
      if( sbtOffCost > bestCost * th )
3873
0
      {
3874
0
        numSbtRdo = 0;
3875
0
      }
3876
0
    }
3877
0
    if( !sbtOffRootCbf && sbtOffCost != MAX_DOUBLE )
3878
0
    {
3879
0
      double th = Clip3( 0.05, 0.55, ( 27 - cu->qp ) * 0.02 + 0.35 );
3880
0
      if( sbtOffCost < m_cRdCost.calcRdCost( ( cu->lwidth() * cu->lheight() ) << SCALE_BITS, 0 ) * th )
3881
0
      {
3882
0
        numSbtRdo = 0;
3883
0
      }
3884
0
    }
3885
3886
0
    if( histBestSbt != MAX_UCHAR && numSbtRdo != 0 )
3887
0
    {
3888
0
      numSbtRdo = 1;
3889
0
      m_cInterSearch.initSbtRdoOrder( CU::getSbtMode( CU::getSbtIdx( histBestSbt ), CU::getSbtPos( histBestSbt ) ) );
3890
0
    }
3891
3892
0
    for( int sbtModeIdx = 0; sbtModeIdx < numSbtRdo; sbtModeIdx++ )
3893
0
    {
3894
0
      uint8_t sbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx );
3895
0
      uint8_t sbtIdx = CU::getSbtIdxFromSbtMode( sbtMode );
3896
0
      uint8_t sbtPos = CU::getSbtPosFromSbtMode( sbtMode );
3897
3898
      //fast algorithm (early skip, save & load)
3899
0
      if( histBestSbt == MAX_UCHAR )
3900
0
      {
3901
0
        uint8_t skipCode = m_cInterSearch.skipSbtByRDCost( cu->lwidth(), cu->lheight(), cu->mtDepth, sbtIdx, sbtPos, bestCS->cost, sbtOffDist, sbtOffCost, sbtOffRootCbf );
3902
0
        if( skipCode != MAX_UCHAR )
3903
0
        {
3904
0
          continue;
3905
0
        }
3906
3907
0
        if( sbtModeIdx > 0 )
3908
0
        {
3909
0
          uint8_t prevSbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx - 1 );
3910
          //make sure the prevSbtMode is the same size as the current SBT mode (otherwise the estimated dist may not be comparable)
3911
0
          if( CU::isSameSbtSize( prevSbtMode, sbtMode ) )
3912
0
          {
3913
0
            Distortion currEstDist = m_cInterSearch.getEstDistSbt( sbtMode );
3914
0
            Distortion prevEstDist = m_cInterSearch.getEstDistSbt( prevSbtMode );
3915
0
            if( currEstDist > prevEstDist * 1.15 )
3916
0
            {
3917
0
              continue;
3918
0
            }
3919
0
          }
3920
0
        }
3921
0
      }
3922
3923
      //init tempCS and TU
3924
0
      if( bestCost == bestCS->cost ) //The first EMT pass didn't become the bestCS, so we clear the TUs generated
3925
0
      {
3926
0
        tempCS->clearTUs();
3927
0
      }
3928
0
      else if( !swapped )
3929
0
      {
3930
0
        tempCS->initStructData( encTestMode.qp );
3931
0
        tempCS->copyStructure( *bestCS, partitioner.chType, partitioner.treeType );
3932
0
        tempCS->getPredBuf().copyFrom( bestCS->getPredBuf() );
3933
0
        bestCost = bestCS->cost;
3934
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3935
0
        swapped = true;
3936
0
      }
3937
0
      else
3938
0
      {
3939
0
        tempCS->clearTUs();
3940
0
        bestCost = bestCS->cost;
3941
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3942
0
      }
3943
3944
      //we need to restart the distortion for the new tempCS, the bit count and the cost
3945
0
      tempCS->dist     = 0;
3946
0
      tempCS->fracBits = 0;
3947
0
      tempCS->cost     = MAX_DOUBLE;
3948
0
      cu->skip         = false;
3949
3950
3951
      //set SBT info
3952
0
      cu->sbtInfo = (sbtPos << 4) + sbtIdx;
3953
3954
      //try residual coding
3955
0
      m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3956
0
      numRDOTried++;
3957
3958
0
      xEncodeDontSplit( *tempCS, partitioner );
3959
0
      xCheckDQP( *tempCS, partitioner );
3960
3961
0
      if( NULL != bestHasNonResi && ( bestCostInternal > tempCS->cost ) )
3962
0
      {
3963
0
        bestCostInternal = tempCS->cost;
3964
0
        if( !( cu->ciip ) )
3965
0
          *bestHasNonResi = !cu->rootCbf;
3966
0
      }
3967
3968
0
      if( tempCS->cost < currBestCost )
3969
0
      {
3970
0
        currBestSbt = cu->sbtInfo;
3971
0
        currBestCost = tempCS->cost;
3972
0
      }
3973
0
      else if( m_pcEncCfg->m_SBT > 2 )
3974
0
      {
3975
0
        sbtModeIdx = numSbtRdo;
3976
0
      }
3977
3978
0
      DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3979
0
      xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3980
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3981
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3982
0
    }
3983
3984
0
    if( bestCostBegin != bestCS->cost )
3985
0
    {
3986
0
      m_sbtCostSave[0] = sbtOffCost;
3987
0
      m_sbtCostSave[1] = currBestCost;
3988
0
    }
3989
3990
0
    if( histBestSbt == MAX_UCHAR && doPreAnalyzeResi && numRDOTried > 1 )
3991
0
    {
3992
0
      auto slsSbt = static_cast<CacheBlkInfoCtrl&>( m_modeCtrl );
3993
0
      int slShift = 4 + std::min( Log2( cu->lwidth() ) + Log2( cu->lheight() ), 9 );
3994
0
      slsSbt.saveBestSbt( cu->cs->area, (uint32_t)( curPuSse >> slShift ), currBestSbt );
3995
0
    }
3996
    
3997
0
    if( ETM_INTER_ME == encTestMode.type )
3998
0
    {
3999
0
      if( equBcwCost != NULL )
4000
0
      {
4001
0
        if( tempCS->cost < ( *equBcwCost ) && cu->BcwIdx == BCW_DEFAULT )
4002
0
        {
4003
0
          ( *equBcwCost ) = tempCS->cost;
4004
0
        }
4005
0
      }
4006
0
      else
4007
0
      {
4008
0
        CHECK( equBcwCost == NULL, "equBcwCost == NULL" );
4009
0
      }
4010
0
      if( tempCS->slice->checkLDC && !cu->imv && cu->BcwIdx != BCW_DEFAULT && tempCS->cost < m_bestBcwCost[1] )
4011
0
      {
4012
0
        if( tempCS->cost < m_bestBcwCost[0] )
4013
0
        {
4014
0
          m_bestBcwCost[1] = m_bestBcwCost[0];
4015
0
          m_bestBcwCost[0] = tempCS->cost;
4016
0
          m_bestBcwIdx[1] = m_bestBcwIdx[0];
4017
0
          m_bestBcwIdx[0] = cu->BcwIdx;
4018
0
        }
4019
0
        else
4020
0
        {
4021
0
          m_bestBcwCost[1] = tempCS->cost;
4022
0
          m_bestBcwIdx[1] = cu->BcwIdx;
4023
0
        }
4024
0
      }
4025
0
    }
4026
0
  }
4027
4028
0
  tempCS->cost = currBestCost;
4029
0
}
4030
4031
void EncCu::xEncodeDontSplit( CodingStructure &cs, Partitioner &partitioner )
4032
99.3k
{
4033
99.3k
  m_CABACEstimator->resetBits();
4034
4035
99.3k
  m_CABACEstimator->split_cu_mode( CU_DONT_SPLIT, cs, partitioner );
4036
99.3k
  if( partitioner.treeType == TREE_C )
4037
99.3k
    CHECK( m_CABACEstimator->getEstFracBits() != 0, "must be 0 bit" );
4038
4039
99.3k
  cs.fracBits += m_CABACEstimator->getEstFracBits(); // split bits
4040
99.3k
  cs.cost      = m_cRdCost.calcRdCost( cs.fracBits, cs.dist );
4041
99.3k
}
4042
4043
void EncCu::xReuseCachedResult( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner )
4044
0
{
4045
0
  EncTestMode cachedMode;
4046
4047
0
  if( ! m_modeCtrl.setCsFrom( *tempCS, cachedMode, partitioner ) )
4048
0
  {
4049
0
    THROW( "Should never happen!" );
4050
0
  }
4051
4052
0
  CodingUnit& cu = *tempCS->cus.front();
4053
0
  partitioner.setCUData( cu );
4054
4055
0
  if( CU::isIntra( cu ) )
4056
0
  {
4057
0
    if( isLuma( cu.chType ) )
4058
0
    {
4059
0
      cu.getMotionBuf().memset( -1 ); // clear motion buf
4060
0
    }
4061
0
    xReconIntraQT( cu );
4062
0
  }
4063
0
  else
4064
0
  {
4065
0
    xDeriveCUMV( cu );
4066
0
    xReconInter( cu );
4067
0
  }
4068
4069
0
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
4070
0
  m_CABACEstimator->resetBits();
4071
4072
0
  CUCtx cuCtx;
4073
0
  cuCtx.isDQPCoded = true;
4074
0
  cuCtx.isChromaQpAdjCoded = true;
4075
0
  m_CABACEstimator->coding_unit( cu, partitioner, cuCtx );
4076
4077
0
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
4078
0
  tempCS->cost     = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
4079
4080
0
  xEncodeDontSplit( *tempCS,         partitioner );
4081
0
  xCheckDQP       ( *tempCS,         partitioner );
4082
0
  xCheckBestMode  (  tempCS, bestCS, partitioner, cachedMode, m_EDO );
4083
0
}
4084
4085
uint64_t EncCu::xCalcPuMeBits( const CodingUnit &cu )
4086
0
{
4087
0
  CHECK( !cu.mergeFlag, "Should only be used for merge!" );
4088
0
  CHECK( CU::isIBC( cu ), "Shound not be used for IBC" );
4089
4090
0
  m_CABACEstimator->resetBits();
4091
0
  m_CABACEstimator->merge_flag(cu);
4092
0
  if( cu.mergeFlag )
4093
0
  {
4094
0
    m_CABACEstimator->merge_data( cu );
4095
0
  }
4096
0
  return m_CABACEstimator->getEstFracBits();
4097
0
}
4098
4099
double EncCu::xCalcDistortion(CodingStructure *&cur_CS, ChannelType chType, int BitDepth, int imv)
4100
0
{
4101
0
  const auto currDist1 = m_cRdCost.getDistPart(cur_CS->getOrgBuf( COMP_Y ), cur_CS->getPredBuf( COMP_Y ), BitDepth, COMP_Y, m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
4102
0
  unsigned int uiMvBits = 0;
4103
0
  unsigned imvShift = imv == IMV_HPEL ? 1 : (imv << 1);
4104
0
  const CodingUnit& cu = *cur_CS->getCU( chType, TREE_D);
4105
0
  if (cu.interDir != 2)
4106
0
  {
4107
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[0][0].hor, cu.mvd[0][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4108
0
  }
4109
0
  if (cu.interDir != 1)
4110
0
  {
4111
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[1][0].hor, cu.mvd[1][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4112
0
  }
4113
0
  return (double(currDist1) + (double)m_cRdCost.getCost(uiMvBits));
4114
0
}
4115
4116
int EncCu::xCheckMMVDCand(MmvdIdx& mmvdMergeCand, int& bestDir, int tempNum, double& bestCostOffset, double& bestCostMerge, double bestCostList )
4117
0
{
4118
0
  int baseIdx = mmvdMergeCand.val / MMVD_MAX_REFINE_NUM;
4119
0
  int CandCur = mmvdMergeCand.val - MMVD_MAX_REFINE_NUM * baseIdx;
4120
4121
0
  if( m_pcEncCfg->m_MMVD > 2 )
4122
0
  {
4123
0
    if( CandCur % 4 == 0 )
4124
0
    {
4125
0
      if( ( bestCostOffset >= bestCostMerge ) && ( CandCur >= 4 ) )
4126
0
      {
4127
0
        if( mmvdMergeCand.val > MMVD_MAX_REFINE_NUM )
4128
0
        {
4129
0
          return 2;
4130
0
        }
4131
0
        else
4132
0
        {
4133
0
          mmvdMergeCand.val = MMVD_MAX_REFINE_NUM;
4134
0
          if( tempNum == mmvdMergeCand.val )
4135
0
          {
4136
0
            return 2;
4137
0
          }
4138
0
        }
4139
0
      }
4140
      //reset
4141
0
      bestCostOffset = MAX_DOUBLE;
4142
0
      bestCostMerge  = bestCostList;
4143
0
    }
4144
0
  }
4145
4146
0
  if( mmvdMergeCand.val == MMVD_MAX_REFINE_NUM )
4147
0
  {
4148
0
    bestDir = 0;
4149
0
  }
4150
0
  if( CandCur >= 4 )
4151
0
  {
4152
0
    if( CandCur % 4 != bestDir )
4153
0
    {
4154
0
      return 1;
4155
0
    }
4156
0
  }
4157
0
  return 0;
4158
0
}
4159
4160
void EncCu::xInitLumaDeltaQpLUT()
4161
0
{
4162
  //TODO: to be discussed: add config parameters for these?
4163
0
  std::vector<int> defaultLumaLevelTodQp_QpChangePoints   =  { -3,  -2,  -1,   0,   1,   2,   3,   4,   5,   6 };
4164
0
  std::vector<int> defaultLumaLevelTodQp_LumaChangePoints =  {  0, 301, 367, 434, 501, 567, 634, 701, 767, 834 };
4165
  
4166
0
  int         lastDeltaQPValue = 0;
4167
0
  std::size_t nextSparseIndex = 0;
4168
0
  for( int index = 0; index < LUMA_LEVEL_TO_DQP_LUT_MAXSIZE; index++ )
4169
0
  {
4170
0
    while( nextSparseIndex < defaultLumaLevelTodQp_QpChangePoints.size() && index >= defaultLumaLevelTodQp_LumaChangePoints[nextSparseIndex] )
4171
0
    {
4172
0
      lastDeltaQPValue = defaultLumaLevelTodQp_QpChangePoints[nextSparseIndex];
4173
0
      nextSparseIndex++;
4174
0
    }
4175
0
    m_lumaLevelToDeltaQPLUT[index] = lastDeltaQPValue;
4176
0
  }
4177
0
}
4178
4179
MergeItem::MergeItem()
4180
146k
{
4181
4182
146k
}
4183
MergeItem::~MergeItem()
4184
146k
{
4185
4186
146k
}
4187
4188
void MergeItem::create( ChromaFormat chromaFormat, const Area &area )
4189
146k
{
4190
146k
  if( m_pelStorage.bufs.empty() )
4191
146k
  {
4192
146k
    m_pelStorage.create( chromaFormat, area );
4193
146k
    m_mvStorage .resize( area.area() >> ( MIN_CU_LOG2 << 1 ) );
4194
146k
  }
4195
4196
146k
  init();
4197
146k
}
4198
4199
void MergeItem::init()
4200
146k
{
4201
  // reset data
4202
146k
  cost        = MAX_DOUBLE;
4203
146k
  mergeIdx    = 0;
4204
146k
  bcwIdx      = 0;
4205
146k
  interDir    = 0;
4206
146k
  useAltHpelIf  = false;
4207
146k
  affineType    = AFFINEMODEL_4PARAM;
4208
146k
  mergeItemType = MergeItemType::NUM;
4209
4210
146k
  noBdofRefine  = false;
4211
146k
  noResidual    = false;
4212
4213
146k
  lumaPredReady   = false;
4214
146k
  chromaPredReady = false;
4215
146k
}
4216
4217
void MergeItem::importMergeInfo(const MergeCtx& mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu)
4218
0
{
4219
0
  mergeIdx      = _mergeIdx;
4220
0
  mergeItemType = _mergeItemType;
4221
4222
0
  if( mergeItemType != MergeItemType::GPM && mergeItemType != MergeItemType::MMVD )
4223
0
  {
4224
0
    mvField[REF_PIC_LIST_0][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_0];
4225
0
    mvField[REF_PIC_LIST_1][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_1];
4226
0
    interDir                   = mergeCtx.interDirNeighbours[mergeIdx];
4227
0
    bcwIdx                     = mergeCtx.BcwIdx            [mergeIdx];
4228
0
    useAltHpelIf               = mergeCtx.useAltHpelIf      [mergeIdx];
4229
0
  }
4230
4231
0
  switch( _mergeItemType )
4232
0
  {
4233
0
  case MergeItemType::REGULAR:
4234
0
  case MergeItemType::CIIP:
4235
0
    break;
4236
4237
0
  case MergeItemType::MMVD:
4238
0
  {
4239
0
    MmvdIdx candIdx;
4240
4241
0
    candIdx.val                = mergeIdx;
4242
0
    mvField[L0][0]             . setMvField( pu.mv[L0][0], pu.refIdx[0] );
4243
0
    mvField[L1][0]             . setMvField( pu.mv[L1][0], pu.refIdx[1] );
4244
0
    interDir                   = pu.interDir;
4245
0
    bcwIdx                     = pu.BcwIdx;
4246
0
    useAltHpelIf               = mergeCtx.useAltHpelIf[candIdx.pos.baseIdx];
4247
4248
0
    break;
4249
0
  }
4250
4251
0
  case MergeItemType::GPM:
4252
0
    mvField[L0][0]             . setMvField( Mv( 0, 0 ), -1 );
4253
0
    mvField[L1][0]             . setMvField( Mv( 0, 0 ), -1 );
4254
0
    bcwIdx                     = BCW_DEFAULT;
4255
0
    useAltHpelIf               = false;
4256
4257
0
    break;
4258
4259
0
  case MergeItemType::IBC:
4260
0
  default:
4261
0
    THROW( "Wrong merge item type" );
4262
0
  }
4263
4264
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4265
0
}
4266
4267
void MergeItem::importMergeInfo( const AffineMergeCtx &mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu )
4268
0
{
4269
0
  mergeIdx      = _mergeIdx;
4270
0
  mergeItemType = _mergeItemType;
4271
4272
0
  affineType    = mergeCtx.affineType         [mergeIdx];
4273
0
  interDir      = mergeCtx.interDirNeighbours [mergeIdx];
4274
0
  bcwIdx        = mergeCtx.BcwIdx             [mergeIdx];
4275
0
  useAltHpelIf  = false;
4276
4277
0
  switch( _mergeItemType )
4278
0
  {
4279
0
  case MergeItemType::SBTMVP:
4280
    // the pu motion was already generated preparing for IFP check (unconditional)
4281
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4282
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4283
4284
0
    break;
4285
4286
0
  case MergeItemType::AFFINE:
4287
    // the pu motion was already generated preparing for IFP check (unconditional)
4288
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4289
0
    mvField[L0][1] . setMvField( pu.mv[L0][1], pu.refIdx[L0] );
4290
0
    mvField[L0][2] . setMvField( pu.mv[L0][2], pu.refIdx[L0] );
4291
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4292
0
    mvField[L1][1] . setMvField( pu.mv[L1][1], pu.refIdx[L1] );
4293
0
    mvField[L1][2] . setMvField( pu.mv[L1][2], pu.refIdx[L1] );
4294
4295
0
    break;
4296
4297
0
  default:
4298
0
    THROW( "Wrong merge item type" );
4299
0
  }
4300
4301
  // the MI buf was already generated preparing for IFP check (unconditional)
4302
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4303
0
}
4304
4305
bool MergeItem::exportMergeInfo( CodingUnit &pu, bool forceNoResidual ) const
4306
0
{
4307
0
  pu.mergeFlag        = true;
4308
0
  pu.mmvdMergeFlag    = false;
4309
0
  pu.interDir         = interDir;
4310
0
  pu.mergeIdx         = mergeIdx;
4311
0
  pu.mergeType        = MRG_TYPE_DEFAULT_N;
4312
0
  pu.mv[REF_PIC_LIST_0][0]  = mvField[REF_PIC_LIST_0][0].mv;
4313
0
  pu.mv[REF_PIC_LIST_1][0]  = mvField[REF_PIC_LIST_1][0].mv;
4314
0
  pu.refIdx[REF_PIC_LIST_0] = mvField[REF_PIC_LIST_0][0].refIdx;
4315
0
  pu.refIdx[REF_PIC_LIST_1] = mvField[REF_PIC_LIST_1][0].refIdx;
4316
0
  pu.mvd[REF_PIC_LIST_0][0] = Mv();
4317
0
  pu.mvd[REF_PIC_LIST_1][0] = Mv();
4318
0
  pu.mvpIdx[REF_PIC_LIST_0] = NOT_VALID;
4319
0
  pu.mvpIdx[REF_PIC_LIST_1] = NOT_VALID;
4320
0
  pu.mvpNum[REF_PIC_LIST_0] = NOT_VALID;
4321
0
  pu.mvpNum[REF_PIC_LIST_1] = NOT_VALID;
4322
0
  pu.BcwIdx         = ( interDir == 3 ) ? bcwIdx : BCW_DEFAULT;
4323
0
  pu.mcControl      = 0;
4324
0
  pu.mmvdSkip       = false;
4325
0
  pu.affine         = false;
4326
0
  pu.affineType     = AFFINEMODEL_4PARAM;
4327
0
  pu.geo            = false;
4328
0
  pu.mtsFlag        = false;
4329
0
  pu.ciip           = false;
4330
0
  pu.imv            = ( !pu.geo && useAltHpelIf ) ? IMV_HPEL : IMV_OFF;
4331
0
  pu.mvRefine       = false;
4332
4333
0
  const bool resetCiip2Regular = mergeItemType == MergeItemType::CIIP && forceNoResidual;
4334
0
  MergeItemType updatedType    = resetCiip2Regular ? MergeItemType::REGULAR : mergeItemType;
4335
4336
0
  switch( updatedType )
4337
0
  {
4338
0
  case MergeItemType::REGULAR:
4339
0
    CU::restrictBiPredMergeCandsOne( pu );
4340
0
    break;
4341
4342
0
  case MergeItemType::CIIP:
4343
0
    CHECK( forceNoResidual, "Cannot force no residuals for CIIP" );
4344
0
    pu.ciip           = true;
4345
0
    pu.intraDir[CH_L] = PLANAR_IDX;
4346
0
    pu.intraDir[CH_C] = DM_CHROMA_IDX;
4347
0
    break;
4348
4349
0
  case MergeItemType::MMVD:
4350
0
    pu.mmvdMergeFlag    = true;
4351
0
    pu.mmvdMergeIdx.val = mergeIdx;
4352
0
    if( forceNoResidual )
4353
0
    {
4354
0
      pu.mmvdSkip       = true;
4355
0
    }
4356
0
    CU::restrictBiPredMergeCandsOne( pu );
4357
0
    break;
4358
4359
0
  case MergeItemType::SBTMVP:
4360
0
    pu.affine    = true;
4361
0
    pu.mergeType = MRG_TYPE_SUBPU_ATMVP;
4362
0
    break;
4363
4364
0
  case MergeItemType::AFFINE:
4365
0
    pu.affine     = true;
4366
0
    pu.affineType = affineType;
4367
0
    pu.mv[L0][0]  = mvField[L0][0].mv;
4368
0
    pu.mv[L1][0]  = mvField[L1][0].mv;
4369
0
    pu.mv[L0][1]  = mvField[L0][1].mv;
4370
0
    pu.mv[L1][1]  = mvField[L1][1].mv;
4371
0
    pu.mv[L0][2]  = mvField[L0][2].mv;
4372
0
    pu.mv[L1][2]  = mvField[L1][2].mv;
4373
0
    pu.refIdx[L0] = mvField[L0][0].refIdx;
4374
0
    pu.refIdx[L1] = mvField[L1][0].refIdx;
4375
0
    break;
4376
4377
0
  case MergeItemType::GPM:
4378
0
    pu.mergeIdx = -1;
4379
0
    pu.geo      = true;
4380
0
    pu.BcwIdx   = BCW_DEFAULT;
4381
0
    updateGpmIdx( mergeIdx, pu.geoSplitDir, pu.geoMergeIdx );
4382
0
    pu.imv      = IMV_OFF;
4383
0
    break;
4384
4385
0
  case MergeItemType::IBC:
4386
0
  default:
4387
0
    THROW( "Wrong merge item type" );
4388
0
  }
4389
4390
0
  pu.getMotionBuf().copyFrom( getMvBuf( pu ) );
4391
4392
0
  return resetCiip2Regular;
4393
0
}
4394
4395
MergeItemList::MergeItemList()
4396
18.3k
{
4397
4398
18.3k
}
4399
4400
MergeItemList::~MergeItemList()
4401
18.3k
{
4402
18.3k
  for( MergeItem* p : m_list )
4403
0
  {
4404
0
    delete p;
4405
0
  }
4406
18.3k
  m_list.clear();
4407
4408
18.3k
  for( MergeItem *p : m_mergeItems )
4409
146k
  {
4410
146k
    delete p;
4411
146k
  }
4412
18.3k
  m_mergeItems.clear();
4413
18.3k
}
4414
4415
void MergeItemList::init( size_t maxSize, size_t maxExtSize, ChromaFormat chromaFormat, SizeType ctuWidth, SizeType ctuHeight )
4416
18.3k
{
4417
18.3k
  CHECK( !m_mergeItems.empty() || !m_list.empty(), "MergeItemList already initialized" );
4418
4419
18.3k
  m_list      . reserve( maxSize + 1 ); // to avoid reallocation when inserting a new item
4420
18.3k
  m_mergeItems. reserve( maxSize + 1 );
4421
18.3k
  m_maxSize   = maxSize;
4422
18.3k
  m_maxExtSize= maxExtSize;
4423
18.3k
  m_numExt    = 0;
4424
4425
165k
  for( int i = 0; i < maxSize + m_maxExtSize; i++ )
4426
146k
  {
4427
146k
    MergeItem *p = new MergeItem;
4428
146k
    p->create( chromaFormat, Area{ 0, 0, ctuWidth, ctuHeight } );
4429
146k
    m_mergeItems.push_back( p );
4430
146k
  }
4431
18.3k
}
4432
4433
MergeItem *MergeItemList::allocateNewMergeItem()
4434
0
{
4435
0
  m_numExt++;
4436
0
  CHECK( m_mergeItems.empty(), "Missing merge items!" );
4437
0
  CHECK( m_numExt > m_maxExtSize, "Taking out more external items than specified during list allocation!" );
4438
0
  MergeItem *p = m_mergeItems.back();
4439
0
  m_mergeItems.pop_back();
4440
0
  p->init();
4441
0
  return p;
4442
0
}
4443
4444
bool MergeItemList::insertMergeItemToList( MergeItem *p )
4445
0
{
4446
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4447
4448
0
  m_numExt--;
4449
4450
0
  if( m_list.empty() )
4451
0
  {
4452
0
    m_list.push_back( p );
4453
0
  }
4454
0
  else if( m_list.size() == m_maxTrackingNum && p->cost >= m_list.back()->cost )
4455
0
  {
4456
0
    m_mergeItems.push_back( p );
4457
0
    return false;
4458
0
  }
4459
0
  else
4460
0
  {
4461
0
    if( m_list.size() == m_maxTrackingNum )
4462
0
    {
4463
0
      m_mergeItems.push_back( m_list.back() );
4464
0
      m_list      .pop_back();
4465
0
    }
4466
0
    auto it = std::find_if( m_list.begin(), m_list.end(), [&p]( const MergeItem *mi ) { return p->cost < mi->cost; } );
4467
0
    m_list.insert( it, p );
4468
0
  }
4469
4470
0
  return true;
4471
0
}
4472
4473
void MergeItemList::giveBackMergeItem( MergeItem *p )
4474
0
{
4475
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4476
4477
0
  m_numExt--;
4478
4479
0
  m_mergeItems.push_back( p );
4480
0
}
4481
4482
MergeItem *MergeItemList::getMergeItemInList( size_t index )
4483
0
{
4484
0
  return index < m_maxTrackingNum ? m_list[index] : nullptr;
4485
0
}
4486
4487
void MergeItemList::resetList( size_t maxTrackingNum )
4488
0
{
4489
0
  CHECK( maxTrackingNum > m_maxSize, "Not enough items allocated to track " << maxTrackingNum << " items" );
4490
4491
0
  for( auto p : m_list )
4492
0
  {
4493
0
    m_mergeItems.push_back( p );
4494
0
  }
4495
0
  m_list.clear  ();
4496
4497
0
  m_maxTrackingNum = maxTrackingNum;
4498
0
}
4499
4500
void MergeItemList::shrinkList( size_t reduceTo )
4501
0
{
4502
0
  CHECK( reduceTo > m_maxSize, "Not enough items allocated to track " << reduceTo << " items" );
4503
4504
0
  while( m_list.size() > reduceTo )
4505
0
  {
4506
0
    m_mergeItems.push_back( m_list.back() );
4507
0
    m_list      .pop_back();
4508
0
  }
4509
0
}
4510
4511
} // namespace vvenc
4512
4513
//! \}