/work/libde265/libde265/acceleration.h
Line | Count | Source |
1 | | /* |
2 | | * H.265 video codec. |
3 | | * Copyright (c) 2013-2014 struktur AG, Dirk Farin <farin@struktur.de> |
4 | | * |
5 | | * This file is part of libde265. |
6 | | * |
7 | | * libde265 is free software: you can redistribute it and/or modify |
8 | | * it under the terms of the GNU Lesser General Public License as |
9 | | * published by the Free Software Foundation, either version 3 of |
10 | | * the License, or (at your option) any later version. |
11 | | * |
12 | | * libde265 is distributed in the hope that it will be useful, |
13 | | * but WITHOUT ANY WARRANTY; without even the implied warranty of |
14 | | * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the |
15 | | * GNU Lesser General Public License for more details. |
16 | | * |
17 | | * You should have received a copy of the GNU Lesser General Public License |
18 | | * along with libde265. If not, see <http://www.gnu.org/licenses/>. |
19 | | */ |
20 | | |
21 | | #ifndef DE265_ACCELERATION_H |
22 | | #define DE265_ACCELERATION_H |
23 | | |
24 | | #include <stddef.h> |
25 | | #include <stdint.h> |
26 | | #include <assert.h> |
27 | | |
28 | | |
29 | | // Highest bit depth at which the intermediate motion-compensation samples |
30 | | // (predSamplesLX, spec 8.5.3.3.3) still fit into int16_t. Above it, the |
31 | | // "_16_32" kernels with int32_t intermediates are used. |
32 | | constexpr int MC_MAX_BIT_DEPTH_INT16 = 12; |
33 | | |
34 | | |
35 | | struct acceleration_functions |
36 | | { |
37 | | void (*put_weighted_pred_avg_8)(uint8_t *_dst, ptrdiff_t dststride, |
38 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
39 | | int width, int height); |
40 | | |
41 | | void (*put_unweighted_pred_8)(uint8_t *_dst, ptrdiff_t dststride, |
42 | | const int16_t *src, ptrdiff_t srcstride, |
43 | | int width, int height); |
44 | | |
45 | | void (*put_weighted_pred_8)(uint8_t *_dst, ptrdiff_t dststride, |
46 | | const int16_t *src, ptrdiff_t srcstride, |
47 | | int width, int height, |
48 | | int w,int o,int log2WD); |
49 | | void (*put_weighted_bipred_8)(uint8_t *_dst, ptrdiff_t dststride, |
50 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
51 | | int width, int height, |
52 | | int w1,int o1, int w2,int o2, int log2WD); |
53 | | |
54 | | |
55 | | void (*put_weighted_pred_avg_16)(uint16_t *_dst, ptrdiff_t dststride, |
56 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
57 | | int width, int height, int bit_depth); |
58 | | |
59 | | void (*put_unweighted_pred_16)(uint16_t *_dst, ptrdiff_t dststride, |
60 | | const int16_t *src, ptrdiff_t srcstride, |
61 | | int width, int height, int bit_depth); |
62 | | |
63 | | void (*put_weighted_pred_16)(uint16_t *_dst, ptrdiff_t dststride, |
64 | | const int16_t *src, ptrdiff_t srcstride, |
65 | | int width, int height, |
66 | | int w,int o,int log2WD, int bit_depth); |
67 | | void (*put_weighted_bipred_16)(uint16_t *_dst, ptrdiff_t dststride, |
68 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
69 | | int width, int height, |
70 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth); |
71 | | |
72 | | |
73 | | // --- BitDepth > MC_MAX_BIT_DEPTH_INT16 --- |
74 | | // The intermediate prediction samples (predSamplesLX, spec 8.5.3.3.3) have |
75 | | // max(14, BitDepth+2) bits plus the overshoot of the interpolation filters, |
76 | | // so above 12 bits they do not fit into int16_t anymore. These variants of the |
77 | | // 16-bit pixel kernels take int32_t intermediates instead. |
78 | | |
79 | | void (*put_weighted_pred_avg_16_32)(uint16_t *_dst, ptrdiff_t dststride, |
80 | | const int32_t *src1, const int32_t *src2, ptrdiff_t srcstride, |
81 | | int width, int height, int bit_depth); |
82 | | |
83 | | void (*put_unweighted_pred_16_32)(uint16_t *_dst, ptrdiff_t dststride, |
84 | | const int32_t *src, ptrdiff_t srcstride, |
85 | | int width, int height, int bit_depth); |
86 | | |
87 | | void (*put_weighted_pred_16_32)(uint16_t *_dst, ptrdiff_t dststride, |
88 | | const int32_t *src, ptrdiff_t srcstride, |
89 | | int width, int height, |
90 | | int w,int o,int log2WD, int bit_depth); |
91 | | void (*put_weighted_bipred_16_32)(uint16_t *_dst, ptrdiff_t dststride, |
92 | | const int32_t *src1, const int32_t *src2, ptrdiff_t srcstride, |
93 | | int width, int height, |
94 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth); |
95 | | |
96 | | |
97 | | // Dispatch on bit depth; the int32_t overloads are for BitDepth > MC_MAX_BIT_DEPTH_INT16 only. |
98 | | |
99 | | void put_weighted_pred_avg(void *_dst, ptrdiff_t dststride, |
100 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
101 | | int width, int height, int bit_depth) const; |
102 | | void put_weighted_pred_avg(void *_dst, ptrdiff_t dststride, |
103 | | const int32_t *src1, const int32_t *src2, ptrdiff_t srcstride, |
104 | | int width, int height, int bit_depth) const; |
105 | | |
106 | | void put_unweighted_pred(void *_dst, ptrdiff_t dststride, |
107 | | const int16_t *src, ptrdiff_t srcstride, |
108 | | int width, int height, int bit_depth) const; |
109 | | void put_unweighted_pred(void *_dst, ptrdiff_t dststride, |
110 | | const int32_t *src, ptrdiff_t srcstride, |
111 | | int width, int height, int bit_depth) const; |
112 | | |
113 | | void put_weighted_pred(void *_dst, ptrdiff_t dststride, |
114 | | const int16_t *src, ptrdiff_t srcstride, |
115 | | int width, int height, |
116 | | int w,int o,int log2WD, int bit_depth) const; |
117 | | void put_weighted_pred(void *_dst, ptrdiff_t dststride, |
118 | | const int32_t *src, ptrdiff_t srcstride, |
119 | | int width, int height, |
120 | | int w,int o,int log2WD, int bit_depth) const; |
121 | | |
122 | | void put_weighted_bipred(void *_dst, ptrdiff_t dststride, |
123 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
124 | | int width, int height, |
125 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth) const; |
126 | | void put_weighted_bipred(void *_dst, ptrdiff_t dststride, |
127 | | const int32_t *src1, const int32_t *src2, ptrdiff_t srcstride, |
128 | | int width, int height, |
129 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth) const; |
130 | | |
131 | | |
132 | | |
133 | | |
134 | | void (*put_hevc_epel_8)(int16_t *dst, ptrdiff_t dststride, |
135 | | const uint8_t *src, ptrdiff_t srcstride, int width, int height, |
136 | | int mx, int my, int16_t* mcbuffer); |
137 | | void (*put_hevc_epel_h_8)(int16_t *dst, ptrdiff_t dststride, |
138 | | const uint8_t *src, ptrdiff_t srcstride, int width, int height, |
139 | | int mx, int my, int16_t* mcbuffer, int bit_depth); |
140 | | void (*put_hevc_epel_v_8)(int16_t *dst, ptrdiff_t dststride, |
141 | | const uint8_t *src, ptrdiff_t srcstride, int width, int height, |
142 | | int mx, int my, int16_t* mcbuffer, int bit_depth); |
143 | | void (*put_hevc_epel_hv_8)(int16_t *dst, ptrdiff_t dststride, |
144 | | const uint8_t *src, ptrdiff_t srcstride, int width, int height, |
145 | | int mx, int my, int16_t* mcbuffer, int bit_depth); |
146 | | |
147 | | void (*put_hevc_qpel_8[4][4])(int16_t *dst, ptrdiff_t dststride, |
148 | | const uint8_t *src, ptrdiff_t srcstride, int width, int height, |
149 | | int16_t* mcbuffer); |
150 | | |
151 | | |
152 | | void (*put_hevc_epel_16)(int16_t *dst, ptrdiff_t dststride, |
153 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
154 | | int mx, int my, int16_t* mcbuffer, int bit_depth); |
155 | | void (*put_hevc_epel_h_16)(int16_t *dst, ptrdiff_t dststride, |
156 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
157 | | int mx, int my, int16_t* mcbuffer, int bit_depth); |
158 | | void (*put_hevc_epel_v_16)(int16_t *dst, ptrdiff_t dststride, |
159 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
160 | | int mx, int my, int16_t* mcbuffer, int bit_depth); |
161 | | void (*put_hevc_epel_hv_16)(int16_t *dst, ptrdiff_t dststride, |
162 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
163 | | int mx, int my, int16_t* mcbuffer, int bit_depth); |
164 | | |
165 | | void (*put_hevc_qpel_16[4][4])(int16_t *dst, ptrdiff_t dststride, |
166 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
167 | | int16_t* mcbuffer, int bit_depth); |
168 | | |
169 | | |
170 | | // --- BitDepth > MC_MAX_BIT_DEPTH_INT16: 16-bit pixels, int32_t intermediates (see above) --- |
171 | | |
172 | | void (*put_hevc_epel_16_32)(int32_t *dst, ptrdiff_t dststride, |
173 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
174 | | int mx, int my, int32_t* mcbuffer, int bit_depth); |
175 | | void (*put_hevc_epel_h_16_32)(int32_t *dst, ptrdiff_t dststride, |
176 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
177 | | int mx, int my, int32_t* mcbuffer, int bit_depth); |
178 | | void (*put_hevc_epel_v_16_32)(int32_t *dst, ptrdiff_t dststride, |
179 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
180 | | int mx, int my, int32_t* mcbuffer, int bit_depth); |
181 | | void (*put_hevc_epel_hv_16_32)(int32_t *dst, ptrdiff_t dststride, |
182 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
183 | | int mx, int my, int32_t* mcbuffer, int bit_depth); |
184 | | |
185 | | // One generic kernel for all fractional positions (including full-sample), |
186 | | // since this path is not performance critical and per-position copies of |
187 | | // the filter would cost ~50 KB of code. |
188 | | void (*put_hevc_qpel_16_32)(int32_t *dst, ptrdiff_t dststride, |
189 | | const uint16_t *src, ptrdiff_t srcstride, int width, int height, |
190 | | int32_t* mcbuffer, int xFrac, int yFrac, int bit_depth); |
191 | | |
192 | | |
193 | | // Dispatch on bit depth; the int32_t overloads are for BitDepth > MC_MAX_BIT_DEPTH_INT16 only. |
194 | | |
195 | | void put_hevc_epel(int16_t *dst, ptrdiff_t dststride, |
196 | | const void *src, ptrdiff_t srcstride, int width, int height, |
197 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const; |
198 | | void put_hevc_epel_h(int16_t *dst, ptrdiff_t dststride, |
199 | | const void *src, ptrdiff_t srcstride, int width, int height, |
200 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const; |
201 | | void put_hevc_epel_v(int16_t *dst, ptrdiff_t dststride, |
202 | | const void *src, ptrdiff_t srcstride, int width, int height, |
203 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const; |
204 | | void put_hevc_epel_hv(int16_t *dst, ptrdiff_t dststride, |
205 | | const void *src, ptrdiff_t srcstride, int width, int height, |
206 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const; |
207 | | |
208 | | void put_hevc_qpel(int16_t *dst, ptrdiff_t dststride, |
209 | | const void *src, ptrdiff_t srcstride, int width, int height, |
210 | | int16_t* mcbuffer, int dX,int dY, int bit_depth) const; |
211 | | |
212 | | void put_hevc_epel(int32_t *dst, ptrdiff_t dststride, |
213 | | const void *src, ptrdiff_t srcstride, int width, int height, |
214 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const; |
215 | | void put_hevc_epel_h(int32_t *dst, ptrdiff_t dststride, |
216 | | const void *src, ptrdiff_t srcstride, int width, int height, |
217 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const; |
218 | | void put_hevc_epel_v(int32_t *dst, ptrdiff_t dststride, |
219 | | const void *src, ptrdiff_t srcstride, int width, int height, |
220 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const; |
221 | | void put_hevc_epel_hv(int32_t *dst, ptrdiff_t dststride, |
222 | | const void *src, ptrdiff_t srcstride, int width, int height, |
223 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const; |
224 | | |
225 | | void put_hevc_qpel(int32_t *dst, ptrdiff_t dststride, |
226 | | const void *src, ptrdiff_t srcstride, int width, int height, |
227 | | int32_t* mcbuffer, int dX,int dY, int bit_depth) const; |
228 | | |
229 | | |
230 | | // --- inverse transforms --- |
231 | | |
232 | | void (*transform_bypass)(int32_t *residual, const int16_t *coeffs, int nT); |
233 | | void (*transform_bypass_rdpcm_v)(int32_t *r, const int16_t *coeffs, int nT); |
234 | | void (*transform_bypass_rdpcm_h)(int32_t *r, const int16_t *coeffs, int nT); |
235 | | |
236 | | // 8 bit |
237 | | |
238 | | void (*transform_skip_8)(uint8_t *_dst, const int16_t *coeffs, ptrdiff_t _stride); // no transform |
239 | | void (*transform_skip_rdpcm_v_8)(uint8_t *_dst, const int16_t *coeffs, int nT, ptrdiff_t _stride); |
240 | | void (*transform_skip_rdpcm_h_8)(uint8_t *_dst, const int16_t *coeffs, int nT, ptrdiff_t _stride); |
241 | | void (*transform_4x4_dst_add_8)(uint8_t *dst, const int16_t *coeffs, ptrdiff_t stride); // iDST |
242 | | void (*transform_add_8[4])(uint8_t *dst, const int16_t *coeffs, ptrdiff_t stride); // iDCT |
243 | | |
244 | | // 9-16 bit |
245 | | |
246 | | void (*transform_skip_16)(uint16_t *_dst, const int16_t *coeffs, ptrdiff_t _stride, int bit_depth); // no transform |
247 | | void (*transform_4x4_dst_add_16)(uint16_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth); // iDST |
248 | | void (*transform_add_16[4])(uint16_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth); // iDCT |
249 | | |
250 | | |
251 | | void (*rotate_coefficients)(int16_t *coeff, int nT); |
252 | | |
253 | | void (*transform_idst_4x4)(int32_t *dst, const int16_t *coeffs, int bdShift, int max_coeff_bits); |
254 | | void (*transform_idct_4x4)(int32_t *dst, const int16_t *coeffs, int bdShift, int max_coeff_bits); |
255 | | void (*transform_idct_8x8)(int32_t *dst, const int16_t *coeffs, int bdShift, int max_coeff_bits); |
256 | | void (*transform_idct_16x16)(int32_t *dst,const int16_t *coeffs,int bdShift, int max_coeff_bits); |
257 | | void (*transform_idct_32x32)(int32_t *dst,const int16_t *coeffs,int bdShift, int max_coeff_bits); |
258 | | void (*add_residual_8)(uint8_t *dst, ptrdiff_t stride, const int32_t* r, int nT, int bit_depth); |
259 | | void (*add_residual_16)(uint16_t *dst,ptrdiff_t stride,const int32_t* r, int nT, int bit_depth); |
260 | | |
261 | | template <class pixel_t> |
262 | | void add_residual(pixel_t *dst, ptrdiff_t stride, const int32_t* r, int nT, int bit_depth) const; |
263 | | |
264 | | // Inverse quantization (no scaling list): for each of the nCoeff entries, |
265 | | // coeffBuf[coeffPos[i]] = Clip16( (coeffList[i]*fact + offset) >> bdShift ). |
266 | | // Contract: fact small enough that coeffList[i]*fact+offset fits in int32 |
267 | | // (caller checks this; the rare int64 case stays scalar). bdShift >= 1. |
268 | | void (*dequant_coeff_block)(int16_t* coeffBuf, const int16_t* coeffList, |
269 | | const int16_t* coeffPos, int nCoeff, |
270 | | int32_t fact, int32_t offset, int32_t bdShift); |
271 | | |
272 | | // --- deblocking (8 bit; one 4-line edge segment) --- |
273 | | void (*deblock_luma_8)(uint8_t* ptr, ptrdiff_t stride, int vertical, |
274 | | int dE, int dEp, int dEq, int tc, int filterP, int filterQ); |
275 | | void (*deblock_chroma_8)(uint8_t* ptr, ptrdiff_t stride, int vertical, |
276 | | int tc, int filterP, int filterQ); |
277 | | |
278 | | void (*rdpcm_v)(int32_t* residual, const int16_t* coeffs, int nT,int tsShift,int bdShift); |
279 | | void (*rdpcm_h)(int32_t* residual, const int16_t* coeffs, int nT,int tsShift,int bdShift); |
280 | | |
281 | | void (*transform_skip_residual)(int32_t *residual, const int16_t *coeffs, int nT, |
282 | | int tsShift,int bdShift); |
283 | | |
284 | | |
285 | | template <class pixel_t> void transform_skip(pixel_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth) const; |
286 | | template <class pixel_t> void transform_skip_rdpcm_v(pixel_t *dst, const int16_t *coeffs, int nT, ptrdiff_t stride, int bit_depth) const; |
287 | | template <class pixel_t> void transform_skip_rdpcm_h(pixel_t *dst, const int16_t *coeffs, int nT, ptrdiff_t stride, int bit_depth) const; |
288 | | template <class pixel_t> void transform_4x4_dst_add(pixel_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth) const; |
289 | | template <class pixel_t> void transform_add(int sizeIdx, pixel_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth) const; |
290 | | |
291 | | |
292 | | // --- intra prediction --- |
293 | | |
294 | | void (*intra_pred_dc_8 )(uint8_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint8_t* border); |
295 | | void (*intra_pred_dc_16)(uint16_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint16_t* border); |
296 | | void (*intra_pred_planar_8 )(uint8_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint8_t* border); |
297 | | void (*intra_pred_planar_16)(uint16_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint16_t* border); |
298 | | void (*intra_pred_angular_8 )(uint8_t* dst, ptrdiff_t stride, int bit_depth, int disableBoundaryFilter, |
299 | | int xB0, int yB0, int mode, int nT, int cIdx, const uint8_t* border); |
300 | | void (*intra_pred_angular_16)(uint16_t* dst, ptrdiff_t stride, int bit_depth, int disableBoundaryFilter, |
301 | | int xB0, int yB0, int mode, int nT, int cIdx, const uint16_t* border); |
302 | | |
303 | | template <class pixel_t> void intra_pred_dc(pixel_t* dst, ptrdiff_t stride, int nT, int cIdx, const pixel_t* border) const; |
304 | | template <class pixel_t> void intra_pred_planar(pixel_t* dst, ptrdiff_t stride, int nT, int cIdx, const pixel_t* border) const; |
305 | | template <class pixel_t> void intra_pred_angular(pixel_t* dst, ptrdiff_t stride, int bit_depth, int disableBoundaryFilter, |
306 | | int xB0, int yB0, int mode, int nT, int cIdx, const pixel_t* border) const; |
307 | | |
308 | | |
309 | | // --- forward transforms --- |
310 | | |
311 | | void (*fwd_transform_4x4_dst_8)(int16_t *coeffs, const int16_t* src, ptrdiff_t stride); // fDST |
312 | | |
313 | | // indexed with (log2TbSize-2) |
314 | | void (*fwd_transform_8[4]) (int16_t *coeffs, const int16_t *src, ptrdiff_t stride); // fDCT |
315 | | |
316 | | |
317 | | // forward Hadamard transform (without scaling factor) |
318 | | // (4x4,8x8,16x16,32x32) indexed with (log2TbSize-2) |
319 | | void (*hadamard_transform_8[4]) (int16_t *coeffs, const int16_t *src, ptrdiff_t stride); |
320 | | }; |
321 | | |
322 | | |
323 | | /* |
324 | | template <> inline void acceleration_functions::put_weighted_pred_avg<uint8_t>(uint8_t *_dst, ptrdiff_t dststride, |
325 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
326 | | int width, int height, int bit_depth) { put_weighted_pred_avg_8(_dst,dststride,src1,src2,srcstride,width,height); } |
327 | | template <> inline void acceleration_functions::put_weighted_pred_avg<uint16_t>(uint16_t *_dst, ptrdiff_t dststride, |
328 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
329 | | int width, int height, int bit_depth) { put_weighted_pred_avg_16(_dst,dststride,src1,src2, |
330 | | srcstride,width,height,bit_depth); } |
331 | | |
332 | | template <> inline void acceleration_functions::put_unweighted_pred<uint8_t>(uint8_t *_dst, ptrdiff_t dststride, |
333 | | const int16_t *src, ptrdiff_t srcstride, |
334 | | int width, int height, int bit_depth) { put_unweighted_pred_8(_dst,dststride,src,srcstride,width,height); } |
335 | | template <> inline void acceleration_functions::put_unweighted_pred<uint16_t>(uint16_t *_dst, ptrdiff_t dststride, |
336 | | const int16_t *src, ptrdiff_t srcstride, |
337 | | int width, int height, int bit_depth) { put_unweighted_pred_16(_dst,dststride,src,srcstride,width,height,bit_depth); } |
338 | | |
339 | | template <> inline void acceleration_functions::put_weighted_pred<uint8_t>(uint8_t *_dst, ptrdiff_t dststride, |
340 | | const int16_t *src, ptrdiff_t srcstride, |
341 | | int width, int height, |
342 | | int w,int o,int log2WD, int bit_depth) { put_weighted_pred_8(_dst,dststride,src,srcstride,width,height,w,o,log2WD); } |
343 | | template <> inline void acceleration_functions::put_weighted_pred<uint16_t>(uint16_t *_dst, ptrdiff_t dststride, |
344 | | const int16_t *src, ptrdiff_t srcstride, |
345 | | int width, int height, |
346 | | int w,int o,int log2WD, int bit_depth) { put_weighted_pred_16(_dst,dststride,src,srcstride,width,height,w,o,log2WD,bit_depth); } |
347 | | |
348 | | template <> inline void acceleration_functions::put_weighted_bipred<uint8_t>(uint8_t *_dst, ptrdiff_t dststride, |
349 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
350 | | int width, int height, |
351 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth) { put_weighted_bipred_8(_dst,dststride,src1,src2,srcstride, |
352 | | width,height, |
353 | | w1,o1,w2,o2,log2WD); } |
354 | | template <> inline void acceleration_functions::put_weighted_bipred<uint16_t>(uint16_t *_dst, ptrdiff_t dststride, |
355 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
356 | | int width, int height, |
357 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth) { put_weighted_bipred_16(_dst,dststride,src1,src2,srcstride, |
358 | | width,height, |
359 | | w1,o1,w2,o2,log2WD,bit_depth); } |
360 | | */ |
361 | | |
362 | | |
363 | | inline void acceleration_functions::put_weighted_pred_avg(void* _dst, ptrdiff_t dststride, |
364 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
365 | | int width, int height, int bit_depth) const |
366 | 0 | { |
367 | 0 | if (bit_depth <= 8) |
368 | 0 | put_weighted_pred_avg_8((uint8_t*)_dst,dststride,src1,src2,srcstride,width,height); |
369 | 0 | else |
370 | 0 | put_weighted_pred_avg_16((uint16_t*)_dst,dststride,src1,src2,srcstride,width,height,bit_depth); |
371 | 0 | } |
372 | | |
373 | | |
374 | | inline void acceleration_functions::put_unweighted_pred(void* _dst, ptrdiff_t dststride, |
375 | | const int16_t *src, ptrdiff_t srcstride, |
376 | | int width, int height, int bit_depth) const |
377 | 0 | { |
378 | 0 | if (bit_depth <= 8) |
379 | 0 | put_unweighted_pred_8((uint8_t*)_dst,dststride,src,srcstride,width,height); |
380 | 0 | else |
381 | 0 | put_unweighted_pred_16((uint16_t*)_dst,dststride,src,srcstride,width,height,bit_depth); |
382 | 0 | } |
383 | | |
384 | | |
385 | | inline void acceleration_functions::put_weighted_pred(void* _dst, ptrdiff_t dststride, |
386 | | const int16_t *src, ptrdiff_t srcstride, |
387 | | int width, int height, |
388 | | int w,int o,int log2WD, int bit_depth) const |
389 | 0 | { |
390 | 0 | if (bit_depth <= 8) |
391 | 0 | put_weighted_pred_8((uint8_t*)_dst,dststride,src,srcstride,width,height,w,o,log2WD); |
392 | 0 | else |
393 | 0 | put_weighted_pred_16((uint16_t*)_dst,dststride,src,srcstride,width,height,w,o,log2WD,bit_depth); |
394 | 0 | } |
395 | | |
396 | | |
397 | | inline void acceleration_functions::put_weighted_bipred(void* _dst, ptrdiff_t dststride, |
398 | | const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride, |
399 | | int width, int height, |
400 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth) const |
401 | 0 | { |
402 | 0 | if (bit_depth <= 8) |
403 | 0 | put_weighted_bipred_8((uint8_t*)_dst,dststride,src1,src2,srcstride, width,height, w1,o1,w2,o2,log2WD); |
404 | 0 | else |
405 | 0 | put_weighted_bipred_16((uint16_t*)_dst,dststride,src1,src2,srcstride, width,height, w1,o1,w2,o2,log2WD,bit_depth); |
406 | 0 | } |
407 | | |
408 | | |
409 | | |
410 | | inline void acceleration_functions::put_hevc_epel(int16_t *dst, ptrdiff_t dststride, |
411 | | const void *src, ptrdiff_t srcstride, int width, int height, |
412 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const |
413 | 0 | { |
414 | 0 | if (bit_depth <= 8) |
415 | 0 | put_hevc_epel_8(dst,dststride,(const uint8_t*)src,srcstride,width,height,mx,my,mcbuffer); |
416 | 0 | else |
417 | 0 | put_hevc_epel_16(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer, bit_depth); |
418 | 0 | } |
419 | | |
420 | | inline void acceleration_functions::put_hevc_epel_h(int16_t *dst, ptrdiff_t dststride, |
421 | | const void *src, ptrdiff_t srcstride, int width, int height, |
422 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const |
423 | 0 | { |
424 | 0 | if (bit_depth <= 8) |
425 | 0 | put_hevc_epel_h_8(dst,dststride,(const uint8_t*)src,srcstride,width,height,mx,my,mcbuffer,bit_depth); |
426 | 0 | else |
427 | 0 | put_hevc_epel_h_16(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer,bit_depth); |
428 | 0 | } |
429 | | |
430 | | inline void acceleration_functions::put_hevc_epel_v(int16_t *dst, ptrdiff_t dststride, |
431 | | const void *src, ptrdiff_t srcstride, int width, int height, |
432 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const |
433 | 0 | { |
434 | 0 | if (bit_depth <= 8) |
435 | 0 | put_hevc_epel_v_8(dst,dststride,(const uint8_t*)src,srcstride,width,height,mx,my,mcbuffer,bit_depth); |
436 | 0 | else |
437 | 0 | put_hevc_epel_v_16(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer, bit_depth); |
438 | 0 | } |
439 | | |
440 | | inline void acceleration_functions::put_hevc_epel_hv(int16_t *dst, ptrdiff_t dststride, |
441 | | const void *src, ptrdiff_t srcstride, int width, int height, |
442 | | int mx, int my, int16_t* mcbuffer, int bit_depth) const |
443 | 0 | { |
444 | 0 | if (bit_depth <= 8) |
445 | 0 | put_hevc_epel_hv_8(dst,dststride,(const uint8_t*)src,srcstride,width,height,mx,my,mcbuffer,bit_depth); |
446 | 0 | else |
447 | 0 | put_hevc_epel_hv_16(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer, bit_depth); |
448 | 0 | } |
449 | | |
450 | | inline void acceleration_functions::put_hevc_qpel(int16_t *dst, ptrdiff_t dststride, |
451 | | const void *src, ptrdiff_t srcstride, int width, int height, |
452 | | int16_t* mcbuffer, int dX,int dY, int bit_depth) const |
453 | 0 | { |
454 | 0 | if (bit_depth <= 8) |
455 | 0 | put_hevc_qpel_8[dX][dY](dst,dststride,(const uint8_t*)src,srcstride,width,height,mcbuffer); |
456 | 0 | else |
457 | 0 | put_hevc_qpel_16[dX][dY](dst,dststride,(const uint16_t*)src,srcstride,width,height,mcbuffer, bit_depth); |
458 | 0 | } |
459 | | |
460 | | |
461 | | // --- BitDepth > MC_MAX_BIT_DEPTH_INT16: int32_t intermediates, always 16-bit pixels --- |
462 | | |
463 | | inline void acceleration_functions::put_weighted_pred_avg(void* _dst, ptrdiff_t dststride, |
464 | | const int32_t *src1, const int32_t *src2, ptrdiff_t srcstride, |
465 | | int width, int height, int bit_depth) const |
466 | 0 | { |
467 | 0 | assert(bit_depth > 8); |
468 | 0 | put_weighted_pred_avg_16_32((uint16_t*)_dst,dststride,src1,src2,srcstride,width,height,bit_depth); |
469 | 0 | } |
470 | | |
471 | | |
472 | | inline void acceleration_functions::put_unweighted_pred(void* _dst, ptrdiff_t dststride, |
473 | | const int32_t *src, ptrdiff_t srcstride, |
474 | | int width, int height, int bit_depth) const |
475 | 0 | { |
476 | 0 | assert(bit_depth > 8); |
477 | 0 | put_unweighted_pred_16_32((uint16_t*)_dst,dststride,src,srcstride,width,height,bit_depth); |
478 | 0 | } |
479 | | |
480 | | |
481 | | inline void acceleration_functions::put_weighted_pred(void* _dst, ptrdiff_t dststride, |
482 | | const int32_t *src, ptrdiff_t srcstride, |
483 | | int width, int height, |
484 | | int w,int o,int log2WD, int bit_depth) const |
485 | 0 | { |
486 | 0 | assert(bit_depth > 8); |
487 | 0 | put_weighted_pred_16_32((uint16_t*)_dst,dststride,src,srcstride,width,height,w,o,log2WD,bit_depth); |
488 | 0 | } |
489 | | |
490 | | |
491 | | inline void acceleration_functions::put_weighted_bipred(void* _dst, ptrdiff_t dststride, |
492 | | const int32_t *src1, const int32_t *src2, ptrdiff_t srcstride, |
493 | | int width, int height, |
494 | | int w1,int o1, int w2,int o2, int log2WD, int bit_depth) const |
495 | 0 | { |
496 | 0 | assert(bit_depth > 8); |
497 | 0 | put_weighted_bipred_16_32((uint16_t*)_dst,dststride,src1,src2,srcstride, width,height, w1,o1,w2,o2,log2WD,bit_depth); |
498 | 0 | } |
499 | | |
500 | | |
501 | | inline void acceleration_functions::put_hevc_epel(int32_t *dst, ptrdiff_t dststride, |
502 | | const void *src, ptrdiff_t srcstride, int width, int height, |
503 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const |
504 | 0 | { |
505 | 0 | assert(bit_depth > 8); |
506 | 0 | put_hevc_epel_16_32(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer, bit_depth); |
507 | 0 | } |
508 | | |
509 | | inline void acceleration_functions::put_hevc_epel_h(int32_t *dst, ptrdiff_t dststride, |
510 | | const void *src, ptrdiff_t srcstride, int width, int height, |
511 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const |
512 | 0 | { |
513 | 0 | assert(bit_depth > 8); |
514 | 0 | put_hevc_epel_h_16_32(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer,bit_depth); |
515 | 0 | } |
516 | | |
517 | | inline void acceleration_functions::put_hevc_epel_v(int32_t *dst, ptrdiff_t dststride, |
518 | | const void *src, ptrdiff_t srcstride, int width, int height, |
519 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const |
520 | 0 | { |
521 | 0 | assert(bit_depth > 8); |
522 | 0 | put_hevc_epel_v_16_32(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer, bit_depth); |
523 | 0 | } |
524 | | |
525 | | inline void acceleration_functions::put_hevc_epel_hv(int32_t *dst, ptrdiff_t dststride, |
526 | | const void *src, ptrdiff_t srcstride, int width, int height, |
527 | | int mx, int my, int32_t* mcbuffer, int bit_depth) const |
528 | 0 | { |
529 | 0 | assert(bit_depth > 8); |
530 | 0 | put_hevc_epel_hv_16_32(dst,dststride,(const uint16_t*)src,srcstride,width,height,mx,my,mcbuffer, bit_depth); |
531 | 0 | } |
532 | | |
533 | | inline void acceleration_functions::put_hevc_qpel(int32_t *dst, ptrdiff_t dststride, |
534 | | const void *src, ptrdiff_t srcstride, int width, int height, |
535 | | int32_t* mcbuffer, int dX,int dY, int bit_depth) const |
536 | 0 | { |
537 | 0 | assert(bit_depth > 8); |
538 | 0 | put_hevc_qpel_16_32(dst,dststride,(const uint16_t*)src,srcstride,width,height,mcbuffer, dX,dY, bit_depth); |
539 | 0 | } |
540 | | |
541 | | |
542 | 0 | template <> inline void acceleration_functions::transform_skip<uint8_t>(uint8_t *dst, const int16_t *coeffs,ptrdiff_t stride, int bit_depth) const { transform_skip_8(dst,coeffs,stride); } |
543 | 0 | template <> inline void acceleration_functions::transform_skip<uint16_t>(uint16_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth) const { transform_skip_16(dst,coeffs,stride, bit_depth); } |
544 | | |
545 | 0 | template <> inline void acceleration_functions::transform_skip_rdpcm_v<uint8_t>(uint8_t *dst, const int16_t *coeffs, int nT, ptrdiff_t stride, int bit_depth) const { assert(bit_depth==8); transform_skip_rdpcm_v_8(dst,coeffs,nT,stride); } |
546 | 0 | template <> inline void acceleration_functions::transform_skip_rdpcm_h<uint8_t>(uint8_t *dst, const int16_t *coeffs, int nT, ptrdiff_t stride, int bit_depth) const { assert(bit_depth==8); transform_skip_rdpcm_h_8(dst,coeffs,nT,stride); } |
547 | 0 | template <> inline void acceleration_functions::transform_skip_rdpcm_v<uint16_t>(uint16_t *dst, const int16_t *coeffs, int nT, ptrdiff_t stride, int bit_depth) const { assert(false); /*transform_skip_rdpcm_v_8(dst,coeffs,nT,stride);*/ } |
548 | 0 | template <> inline void acceleration_functions::transform_skip_rdpcm_h<uint16_t>(uint16_t *dst, const int16_t *coeffs, int nT, ptrdiff_t stride, int bit_depth) const { assert(false); /*transform_skip_rdpcm_h_8(dst,coeffs,nT,stride);*/ } |
549 | | |
550 | | |
551 | 0 | template <> inline void acceleration_functions::transform_4x4_dst_add<uint8_t>(uint8_t *dst, const int16_t *coeffs, ptrdiff_t stride,int bit_depth) const { transform_4x4_dst_add_8(dst,coeffs,stride); } |
552 | 0 | template <> inline void acceleration_functions::transform_4x4_dst_add<uint16_t>(uint16_t *dst, const int16_t *coeffs, ptrdiff_t stride,int bit_depth) const { transform_4x4_dst_add_16(dst,coeffs,stride,bit_depth); } |
553 | | |
554 | 0 | template <> inline void acceleration_functions::transform_add<uint8_t>(int sizeIdx, uint8_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth) const { transform_add_8[sizeIdx](dst,coeffs,stride); } |
555 | 0 | template <> inline void acceleration_functions::transform_add<uint16_t>(int sizeIdx, uint16_t *dst, const int16_t *coeffs, ptrdiff_t stride, int bit_depth) const { transform_add_16[sizeIdx](dst,coeffs,stride,bit_depth); } |
556 | | |
557 | 0 | template <> inline void acceleration_functions::add_residual(uint8_t *dst, ptrdiff_t stride, const int32_t* r, int nT, int bit_depth) const { add_residual_8(dst,stride,r,nT,bit_depth); } |
558 | 0 | template <> inline void acceleration_functions::add_residual(uint16_t *dst, ptrdiff_t stride, const int32_t* r, int nT, int bit_depth) const { add_residual_16(dst,stride,r,nT,bit_depth); } |
559 | | |
560 | 0 | template <> inline void acceleration_functions::intra_pred_dc<uint8_t> (uint8_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint8_t* border) const { intra_pred_dc_8 (dst,stride,nT,cIdx,border); } |
561 | 0 | template <> inline void acceleration_functions::intra_pred_dc<uint16_t>(uint16_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint16_t* border) const { intra_pred_dc_16(dst,stride,nT,cIdx,border); } |
562 | | |
563 | 0 | template <> inline void acceleration_functions::intra_pred_planar<uint8_t> (uint8_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint8_t* border) const { intra_pred_planar_8 (dst,stride,nT,cIdx,border); } |
564 | 0 | template <> inline void acceleration_functions::intra_pred_planar<uint16_t>(uint16_t* dst, ptrdiff_t stride, int nT, int cIdx, const uint16_t* border) const { intra_pred_planar_16(dst,stride,nT,cIdx,border); } |
565 | | |
566 | 0 | template <> inline void acceleration_functions::intra_pred_angular<uint8_t> (uint8_t* dst, ptrdiff_t stride, int bit_depth, int disableBoundaryFilter, int xB0, int yB0, int mode, int nT, int cIdx, const uint8_t* border) const { intra_pred_angular_8 (dst,stride,bit_depth,disableBoundaryFilter,xB0,yB0,mode,nT,cIdx,border); } |
567 | 0 | template <> inline void acceleration_functions::intra_pred_angular<uint16_t>(uint16_t* dst, ptrdiff_t stride, int bit_depth, int disableBoundaryFilter, int xB0, int yB0, int mode, int nT, int cIdx, const uint16_t* border) const { intra_pred_angular_16(dst,stride,bit_depth,disableBoundaryFilter,xB0,yB0,mode,nT,cIdx,border); } |
568 | | |
569 | | #endif |