Chromium Code Reviews
chromiumcodereview-hr@appspot.gserviceaccount.com (chromiumcodereview-hr) | Please choose your nickname with Settings | Help | Chromium Project | Gerrit Changes | Sign out
(441)

Side by Side Diff: third_party/libvpx/source/libvpx/vp9/encoder/vp9_rd.c

Issue 1158913006: Move libvpx from DEPS to src (Closed) Base URL: https://chromium.googlesource.com/chromium/src.git@master
Patch Set: add DEPS file with #include paths Created 5 years, 6 months ago
Use n/p to move between diff chunks; N/P to move between comments. Draft comments are only viewable by you.
Jump to:
View unified diff | Download patch
OLDNEW
(Empty)
1 /*
2 * Copyright (c) 2010 The WebM project authors. All Rights Reserved.
3 *
4 * Use of this source code is governed by a BSD-style license
5 * that can be found in the LICENSE file in the root of the source
6 * tree. An additional intellectual property rights grant can be found
7 * in the file PATENTS. All contributing project authors may
8 * be found in the AUTHORS file in the root of the source tree.
9 */
10
11 #include <assert.h>
12 #include <math.h>
13 #include <stdio.h>
14
15 #include "./vp9_rtcd.h"
16
17 #include "vpx_mem/vpx_mem.h"
18
19 #include "vp9/common/vp9_common.h"
20 #include "vp9/common/vp9_entropy.h"
21 #include "vp9/common/vp9_entropymode.h"
22 #include "vp9/common/vp9_mvref_common.h"
23 #include "vp9/common/vp9_pred_common.h"
24 #include "vp9/common/vp9_quant_common.h"
25 #include "vp9/common/vp9_reconinter.h"
26 #include "vp9/common/vp9_reconintra.h"
27 #include "vp9/common/vp9_seg_common.h"
28 #include "vp9/common/vp9_systemdependent.h"
29
30 #include "vp9/encoder/vp9_cost.h"
31 #include "vp9/encoder/vp9_encodemb.h"
32 #include "vp9/encoder/vp9_encodemv.h"
33 #include "vp9/encoder/vp9_encoder.h"
34 #include "vp9/encoder/vp9_mcomp.h"
35 #include "vp9/encoder/vp9_quantize.h"
36 #include "vp9/encoder/vp9_ratectrl.h"
37 #include "vp9/encoder/vp9_rd.h"
38 #include "vp9/encoder/vp9_tokenize.h"
39 #include "vp9/encoder/vp9_variance.h"
40
41 #define RD_THRESH_POW 1.25
42 #define RD_MULT_EPB_RATIO 64
43
44 // Factor to weigh the rate for switchable interp filters.
45 #define SWITCHABLE_INTERP_RATE_FACTOR 1
46
47 void vp9_rd_cost_reset(RD_COST *rd_cost) {
48 rd_cost->rate = INT_MAX;
49 rd_cost->dist = INT64_MAX;
50 rd_cost->rdcost = INT64_MAX;
51 }
52
53 void vp9_rd_cost_init(RD_COST *rd_cost) {
54 rd_cost->rate = 0;
55 rd_cost->dist = 0;
56 rd_cost->rdcost = 0;
57 }
58
59 // The baseline rd thresholds for breaking out of the rd loop for
60 // certain modes are assumed to be based on 8x8 blocks.
61 // This table is used to correct for block size.
62 // The factors here are << 2 (2 = x0.5, 32 = x8 etc).
63 static const uint8_t rd_thresh_block_size_factor[BLOCK_SIZES] = {
64 2, 3, 3, 4, 6, 6, 8, 12, 12, 16, 24, 24, 32
65 };
66
67 static void fill_mode_costs(VP9_COMP *cpi) {
68 const FRAME_CONTEXT *const fc = cpi->common.fc;
69 int i, j;
70
71 for (i = 0; i < INTRA_MODES; ++i)
72 for (j = 0; j < INTRA_MODES; ++j)
73 vp9_cost_tokens(cpi->y_mode_costs[i][j], vp9_kf_y_mode_prob[i][j],
74 vp9_intra_mode_tree);
75
76 vp9_cost_tokens(cpi->mbmode_cost, fc->y_mode_prob[1], vp9_intra_mode_tree);
77 vp9_cost_tokens(cpi->intra_uv_mode_cost[KEY_FRAME],
78 vp9_kf_uv_mode_prob[TM_PRED], vp9_intra_mode_tree);
79 vp9_cost_tokens(cpi->intra_uv_mode_cost[INTER_FRAME],
80 fc->uv_mode_prob[TM_PRED], vp9_intra_mode_tree);
81
82 for (i = 0; i < SWITCHABLE_FILTER_CONTEXTS; ++i)
83 vp9_cost_tokens(cpi->switchable_interp_costs[i],
84 fc->switchable_interp_prob[i], vp9_switchable_interp_tree);
85 }
86
87 static void fill_token_costs(vp9_coeff_cost *c,
88 vp9_coeff_probs_model (*p)[PLANE_TYPES]) {
89 int i, j, k, l;
90 TX_SIZE t;
91 for (t = TX_4X4; t <= TX_32X32; ++t)
92 for (i = 0; i < PLANE_TYPES; ++i)
93 for (j = 0; j < REF_TYPES; ++j)
94 for (k = 0; k < COEF_BANDS; ++k)
95 for (l = 0; l < BAND_COEFF_CONTEXTS(k); ++l) {
96 vp9_prob probs[ENTROPY_NODES];
97 vp9_model_to_full_probs(p[t][i][j][k][l], probs);
98 vp9_cost_tokens((int *)c[t][i][j][k][0][l], probs,
99 vp9_coef_tree);
100 vp9_cost_tokens_skip((int *)c[t][i][j][k][1][l], probs,
101 vp9_coef_tree);
102 assert(c[t][i][j][k][0][l][EOB_TOKEN] ==
103 c[t][i][j][k][1][l][EOB_TOKEN]);
104 }
105 }
106
107 // Values are now correlated to quantizer.
108 static int sad_per_bit16lut_8[QINDEX_RANGE];
109 static int sad_per_bit4lut_8[QINDEX_RANGE];
110
111 #if CONFIG_VP9_HIGHBITDEPTH
112 static int sad_per_bit16lut_10[QINDEX_RANGE];
113 static int sad_per_bit4lut_10[QINDEX_RANGE];
114 static int sad_per_bit16lut_12[QINDEX_RANGE];
115 static int sad_per_bit4lut_12[QINDEX_RANGE];
116 #endif
117
118 static void init_me_luts_bd(int *bit16lut, int *bit4lut, int range,
119 vpx_bit_depth_t bit_depth) {
120 int i;
121 // Initialize the sad lut tables using a formulaic calculation for now.
122 // This is to make it easier to resolve the impact of experimental changes
123 // to the quantizer tables.
124 for (i = 0; i < range; i++) {
125 const double q = vp9_convert_qindex_to_q(i, bit_depth);
126 bit16lut[i] = (int)(0.0418 * q + 2.4107);
127 bit4lut[i] = (int)(0.063 * q + 2.742);
128 }
129 }
130
131 void vp9_init_me_luts() {
132 init_me_luts_bd(sad_per_bit16lut_8, sad_per_bit4lut_8, QINDEX_RANGE,
133 VPX_BITS_8);
134 #if CONFIG_VP9_HIGHBITDEPTH
135 init_me_luts_bd(sad_per_bit16lut_10, sad_per_bit4lut_10, QINDEX_RANGE,
136 VPX_BITS_10);
137 init_me_luts_bd(sad_per_bit16lut_12, sad_per_bit4lut_12, QINDEX_RANGE,
138 VPX_BITS_12);
139 #endif
140 }
141
142 static const int rd_boost_factor[16] = {
143 64, 32, 32, 32, 24, 16, 12, 12,
144 8, 8, 4, 4, 2, 2, 1, 0
145 };
146 static const int rd_frame_type_factor[FRAME_UPDATE_TYPES] = {
147 128, 144, 128, 128, 144
148 };
149
150 int vp9_compute_rd_mult(const VP9_COMP *cpi, int qindex) {
151 const int64_t q = vp9_dc_quant(qindex, 0, cpi->common.bit_depth);
152 #if CONFIG_VP9_HIGHBITDEPTH
153 int64_t rdmult = 0;
154 switch (cpi->common.bit_depth) {
155 case VPX_BITS_8:
156 rdmult = 88 * q * q / 24;
157 break;
158 case VPX_BITS_10:
159 rdmult = ROUND_POWER_OF_TWO(88 * q * q / 24, 4);
160 break;
161 case VPX_BITS_12:
162 rdmult = ROUND_POWER_OF_TWO(88 * q * q / 24, 8);
163 break;
164 default:
165 assert(0 && "bit_depth should be VPX_BITS_8, VPX_BITS_10 or VPX_BITS_12");
166 return -1;
167 }
168 #else
169 int64_t rdmult = 88 * q * q / 24;
170 #endif // CONFIG_VP9_HIGHBITDEPTH
171 if (cpi->oxcf.pass == 2 && (cpi->common.frame_type != KEY_FRAME)) {
172 const GF_GROUP *const gf_group = &cpi->twopass.gf_group;
173 const FRAME_UPDATE_TYPE frame_type = gf_group->update_type[gf_group->index];
174 const int boost_index = MIN(15, (cpi->rc.gfu_boost / 100));
175
176 rdmult = (rdmult * rd_frame_type_factor[frame_type]) >> 7;
177 rdmult += ((rdmult * rd_boost_factor[boost_index]) >> 7);
178 }
179 return (int)rdmult;
180 }
181
182 static int compute_rd_thresh_factor(int qindex, vpx_bit_depth_t bit_depth) {
183 double q;
184 #if CONFIG_VP9_HIGHBITDEPTH
185 switch (bit_depth) {
186 case VPX_BITS_8:
187 q = vp9_dc_quant(qindex, 0, VPX_BITS_8) / 4.0;
188 break;
189 case VPX_BITS_10:
190 q = vp9_dc_quant(qindex, 0, VPX_BITS_10) / 16.0;
191 break;
192 case VPX_BITS_12:
193 q = vp9_dc_quant(qindex, 0, VPX_BITS_12) / 64.0;
194 break;
195 default:
196 assert(0 && "bit_depth should be VPX_BITS_8, VPX_BITS_10 or VPX_BITS_12");
197 return -1;
198 }
199 #else
200 (void) bit_depth;
201 q = vp9_dc_quant(qindex, 0, VPX_BITS_8) / 4.0;
202 #endif // CONFIG_VP9_HIGHBITDEPTH
203 // TODO(debargha): Adjust the function below.
204 return MAX((int)(pow(q, RD_THRESH_POW) * 5.12), 8);
205 }
206
207 void vp9_initialize_me_consts(VP9_COMP *cpi, MACROBLOCK *x, int qindex) {
208 #if CONFIG_VP9_HIGHBITDEPTH
209 switch (cpi->common.bit_depth) {
210 case VPX_BITS_8:
211 x->sadperbit16 = sad_per_bit16lut_8[qindex];
212 x->sadperbit4 = sad_per_bit4lut_8[qindex];
213 break;
214 case VPX_BITS_10:
215 x->sadperbit16 = sad_per_bit16lut_10[qindex];
216 x->sadperbit4 = sad_per_bit4lut_10[qindex];
217 break;
218 case VPX_BITS_12:
219 x->sadperbit16 = sad_per_bit16lut_12[qindex];
220 x->sadperbit4 = sad_per_bit4lut_12[qindex];
221 break;
222 default:
223 assert(0 && "bit_depth should be VPX_BITS_8, VPX_BITS_10 or VPX_BITS_12");
224 }
225 #else
226 (void)cpi;
227 x->sadperbit16 = sad_per_bit16lut_8[qindex];
228 x->sadperbit4 = sad_per_bit4lut_8[qindex];
229 #endif // CONFIG_VP9_HIGHBITDEPTH
230 }
231
232 static void set_block_thresholds(const VP9_COMMON *cm, RD_OPT *rd) {
233 int i, bsize, segment_id;
234
235 for (segment_id = 0; segment_id < MAX_SEGMENTS; ++segment_id) {
236 const int qindex =
237 clamp(vp9_get_qindex(&cm->seg, segment_id, cm->base_qindex) +
238 cm->y_dc_delta_q, 0, MAXQ);
239 const int q = compute_rd_thresh_factor(qindex, cm->bit_depth);
240
241 for (bsize = 0; bsize < BLOCK_SIZES; ++bsize) {
242 // Threshold here seems unnecessarily harsh but fine given actual
243 // range of values used for cpi->sf.thresh_mult[].
244 const int t = q * rd_thresh_block_size_factor[bsize];
245 const int thresh_max = INT_MAX / t;
246
247 if (bsize >= BLOCK_8X8) {
248 for (i = 0; i < MAX_MODES; ++i)
249 rd->threshes[segment_id][bsize][i] =
250 rd->thresh_mult[i] < thresh_max
251 ? rd->thresh_mult[i] * t / 4
252 : INT_MAX;
253 } else {
254 for (i = 0; i < MAX_REFS; ++i)
255 rd->threshes[segment_id][bsize][i] =
256 rd->thresh_mult_sub8x8[i] < thresh_max
257 ? rd->thresh_mult_sub8x8[i] * t / 4
258 : INT_MAX;
259 }
260 }
261 }
262 }
263
264 void vp9_initialize_rd_consts(VP9_COMP *cpi) {
265 VP9_COMMON *const cm = &cpi->common;
266 MACROBLOCK *const x = &cpi->td.mb;
267 RD_OPT *const rd = &cpi->rd;
268 int i;
269
270 vp9_clear_system_state();
271
272 rd->RDDIV = RDDIV_BITS; // In bits (to multiply D by 128).
273 rd->RDMULT = vp9_compute_rd_mult(cpi, cm->base_qindex + cm->y_dc_delta_q);
274
275 x->errorperbit = rd->RDMULT / RD_MULT_EPB_RATIO;
276 x->errorperbit += (x->errorperbit == 0);
277
278 x->select_tx_size = (cpi->sf.tx_size_search_method == USE_LARGESTALL &&
279 cm->frame_type != KEY_FRAME) ? 0 : 1;
280
281 set_block_thresholds(cm, rd);
282
283 if (!cpi->sf.use_nonrd_pick_mode || cm->frame_type == KEY_FRAME)
284 fill_token_costs(x->token_costs, cm->fc->coef_probs);
285
286 if (cpi->sf.partition_search_type != VAR_BASED_PARTITION ||
287 cm->frame_type == KEY_FRAME) {
288 for (i = 0; i < PARTITION_CONTEXTS; ++i)
289 vp9_cost_tokens(cpi->partition_cost[i], get_partition_probs(cm, i),
290 vp9_partition_tree);
291 }
292
293 if (!cpi->sf.use_nonrd_pick_mode || (cm->current_video_frame & 0x07) == 1 ||
294 cm->frame_type == KEY_FRAME) {
295 fill_mode_costs(cpi);
296
297 if (!frame_is_intra_only(cm)) {
298 vp9_build_nmv_cost_table(x->nmvjointcost,
299 cm->allow_high_precision_mv ? x->nmvcost_hp
300 : x->nmvcost,
301 &cm->fc->nmvc, cm->allow_high_precision_mv);
302
303 for (i = 0; i < INTER_MODE_CONTEXTS; ++i)
304 vp9_cost_tokens((int *)cpi->inter_mode_cost[i],
305 cm->fc->inter_mode_probs[i], vp9_inter_mode_tree);
306 }
307 }
308 }
309
310 static void model_rd_norm(int xsq_q10, int *r_q10, int *d_q10) {
311 // NOTE: The tables below must be of the same size.
312
313 // The functions described below are sampled at the four most significant
314 // bits of x^2 + 8 / 256.
315
316 // Normalized rate:
317 // This table models the rate for a Laplacian source with given variance
318 // when quantized with a uniform quantizer with given stepsize. The
319 // closed form expression is:
320 // Rn(x) = H(sqrt(r)) + sqrt(r)*[1 + H(r)/(1 - r)],
321 // where r = exp(-sqrt(2) * x) and x = qpstep / sqrt(variance),
322 // and H(x) is the binary entropy function.
323 static const int rate_tab_q10[] = {
324 65536, 6086, 5574, 5275, 5063, 4899, 4764, 4651,
325 4553, 4389, 4255, 4142, 4044, 3958, 3881, 3811,
326 3748, 3635, 3538, 3453, 3376, 3307, 3244, 3186,
327 3133, 3037, 2952, 2877, 2809, 2747, 2690, 2638,
328 2589, 2501, 2423, 2353, 2290, 2232, 2179, 2130,
329 2084, 2001, 1928, 1862, 1802, 1748, 1698, 1651,
330 1608, 1530, 1460, 1398, 1342, 1290, 1243, 1199,
331 1159, 1086, 1021, 963, 911, 864, 821, 781,
332 745, 680, 623, 574, 530, 490, 455, 424,
333 395, 345, 304, 269, 239, 213, 190, 171,
334 154, 126, 104, 87, 73, 61, 52, 44,
335 38, 28, 21, 16, 12, 10, 8, 6,
336 5, 3, 2, 1, 1, 1, 0, 0,
337 };
338 // Normalized distortion:
339 // This table models the normalized distortion for a Laplacian source
340 // with given variance when quantized with a uniform quantizer
341 // with given stepsize. The closed form expression is:
342 // Dn(x) = 1 - 1/sqrt(2) * x / sinh(x/sqrt(2))
343 // where x = qpstep / sqrt(variance).
344 // Note the actual distortion is Dn * variance.
345 static const int dist_tab_q10[] = {
346 0, 0, 1, 1, 1, 2, 2, 2,
347 3, 3, 4, 5, 5, 6, 7, 7,
348 8, 9, 11, 12, 13, 15, 16, 17,
349 18, 21, 24, 26, 29, 31, 34, 36,
350 39, 44, 49, 54, 59, 64, 69, 73,
351 78, 88, 97, 106, 115, 124, 133, 142,
352 151, 167, 184, 200, 215, 231, 245, 260,
353 274, 301, 327, 351, 375, 397, 418, 439,
354 458, 495, 528, 559, 587, 613, 637, 659,
355 680, 717, 749, 777, 801, 823, 842, 859,
356 874, 899, 919, 936, 949, 960, 969, 977,
357 983, 994, 1001, 1006, 1010, 1013, 1015, 1017,
358 1018, 1020, 1022, 1022, 1023, 1023, 1023, 1024,
359 };
360 static const int xsq_iq_q10[] = {
361 0, 4, 8, 12, 16, 20, 24, 28,
362 32, 40, 48, 56, 64, 72, 80, 88,
363 96, 112, 128, 144, 160, 176, 192, 208,
364 224, 256, 288, 320, 352, 384, 416, 448,
365 480, 544, 608, 672, 736, 800, 864, 928,
366 992, 1120, 1248, 1376, 1504, 1632, 1760, 1888,
367 2016, 2272, 2528, 2784, 3040, 3296, 3552, 3808,
368 4064, 4576, 5088, 5600, 6112, 6624, 7136, 7648,
369 8160, 9184, 10208, 11232, 12256, 13280, 14304, 15328,
370 16352, 18400, 20448, 22496, 24544, 26592, 28640, 30688,
371 32736, 36832, 40928, 45024, 49120, 53216, 57312, 61408,
372 65504, 73696, 81888, 90080, 98272, 106464, 114656, 122848,
373 131040, 147424, 163808, 180192, 196576, 212960, 229344, 245728,
374 };
375 const int tmp = (xsq_q10 >> 2) + 8;
376 const int k = get_msb(tmp) - 3;
377 const int xq = (k << 3) + ((tmp >> k) & 0x7);
378 const int one_q10 = 1 << 10;
379 const int a_q10 = ((xsq_q10 - xsq_iq_q10[xq]) << 10) >> (2 + k);
380 const int b_q10 = one_q10 - a_q10;
381 *r_q10 = (rate_tab_q10[xq] * b_q10 + rate_tab_q10[xq + 1] * a_q10) >> 10;
382 *d_q10 = (dist_tab_q10[xq] * b_q10 + dist_tab_q10[xq + 1] * a_q10) >> 10;
383 }
384
385 void vp9_model_rd_from_var_lapndz(unsigned int var, unsigned int n_log2,
386 unsigned int qstep, int *rate,
387 int64_t *dist) {
388 // This function models the rate and distortion for a Laplacian
389 // source with given variance when quantized with a uniform quantizer
390 // with given stepsize. The closed form expressions are in:
391 // Hang and Chen, "Source Model for transform video coder and its
392 // application - Part I: Fundamental Theory", IEEE Trans. Circ.
393 // Sys. for Video Tech., April 1997.
394 if (var == 0) {
395 *rate = 0;
396 *dist = 0;
397 } else {
398 int d_q10, r_q10;
399 static const uint32_t MAX_XSQ_Q10 = 245727;
400 const uint64_t xsq_q10_64 =
401 (((uint64_t)qstep * qstep << (n_log2 + 10)) + (var >> 1)) / var;
402 const int xsq_q10 = (int)MIN(xsq_q10_64, MAX_XSQ_Q10);
403 model_rd_norm(xsq_q10, &r_q10, &d_q10);
404 *rate = ((r_q10 << n_log2) + 2) >> 2;
405 *dist = (var * (int64_t)d_q10 + 512) >> 10;
406 }
407 }
408
409 void vp9_get_entropy_contexts(BLOCK_SIZE bsize, TX_SIZE tx_size,
410 const struct macroblockd_plane *pd,
411 ENTROPY_CONTEXT t_above[16],
412 ENTROPY_CONTEXT t_left[16]) {
413 const BLOCK_SIZE plane_bsize = get_plane_block_size(bsize, pd);
414 const int num_4x4_w = num_4x4_blocks_wide_lookup[plane_bsize];
415 const int num_4x4_h = num_4x4_blocks_high_lookup[plane_bsize];
416 const ENTROPY_CONTEXT *const above = pd->above_context;
417 const ENTROPY_CONTEXT *const left = pd->left_context;
418
419 int i;
420 switch (tx_size) {
421 case TX_4X4:
422 memcpy(t_above, above, sizeof(ENTROPY_CONTEXT) * num_4x4_w);
423 memcpy(t_left, left, sizeof(ENTROPY_CONTEXT) * num_4x4_h);
424 break;
425 case TX_8X8:
426 for (i = 0; i < num_4x4_w; i += 2)
427 t_above[i] = !!*(const uint16_t *)&above[i];
428 for (i = 0; i < num_4x4_h; i += 2)
429 t_left[i] = !!*(const uint16_t *)&left[i];
430 break;
431 case TX_16X16:
432 for (i = 0; i < num_4x4_w; i += 4)
433 t_above[i] = !!*(const uint32_t *)&above[i];
434 for (i = 0; i < num_4x4_h; i += 4)
435 t_left[i] = !!*(const uint32_t *)&left[i];
436 break;
437 case TX_32X32:
438 for (i = 0; i < num_4x4_w; i += 8)
439 t_above[i] = !!*(const uint64_t *)&above[i];
440 for (i = 0; i < num_4x4_h; i += 8)
441 t_left[i] = !!*(const uint64_t *)&left[i];
442 break;
443 default:
444 assert(0 && "Invalid transform size.");
445 break;
446 }
447 }
448
449 void vp9_mv_pred(VP9_COMP *cpi, MACROBLOCK *x,
450 uint8_t *ref_y_buffer, int ref_y_stride,
451 int ref_frame, BLOCK_SIZE block_size) {
452 MACROBLOCKD *xd = &x->e_mbd;
453 MB_MODE_INFO *mbmi = &xd->mi[0]->mbmi;
454 int i;
455 int zero_seen = 0;
456 int best_index = 0;
457 int best_sad = INT_MAX;
458 int this_sad = INT_MAX;
459 int max_mv = 0;
460 int near_same_nearest;
461 uint8_t *src_y_ptr = x->plane[0].src.buf;
462 uint8_t *ref_y_ptr;
463 const int num_mv_refs = MAX_MV_REF_CANDIDATES +
464 (cpi->sf.adaptive_motion_search &&
465 block_size < x->max_partition_size);
466
467 MV pred_mv[3];
468 pred_mv[0] = mbmi->ref_mvs[ref_frame][0].as_mv;
469 pred_mv[1] = mbmi->ref_mvs[ref_frame][1].as_mv;
470 pred_mv[2] = x->pred_mv[ref_frame];
471 assert(num_mv_refs <= (int)(sizeof(pred_mv) / sizeof(pred_mv[0])));
472
473 near_same_nearest =
474 mbmi->ref_mvs[ref_frame][0].as_int == mbmi->ref_mvs[ref_frame][1].as_int;
475 // Get the sad for each candidate reference mv.
476 for (i = 0; i < num_mv_refs; ++i) {
477 const MV *this_mv = &pred_mv[i];
478 int fp_row, fp_col;
479
480 if (i == 1 && near_same_nearest)
481 continue;
482 fp_row = (this_mv->row + 3 + (this_mv->row >= 0)) >> 3;
483 fp_col = (this_mv->col + 3 + (this_mv->col >= 0)) >> 3;
484 max_mv = MAX(max_mv, MAX(abs(this_mv->row), abs(this_mv->col)) >> 3);
485
486 if (fp_row ==0 && fp_col == 0 && zero_seen)
487 continue;
488 zero_seen |= (fp_row ==0 && fp_col == 0);
489
490 ref_y_ptr =&ref_y_buffer[ref_y_stride * fp_row + fp_col];
491 // Find sad for current vector.
492 this_sad = cpi->fn_ptr[block_size].sdf(src_y_ptr, x->plane[0].src.stride,
493 ref_y_ptr, ref_y_stride);
494 // Note if it is the best so far.
495 if (this_sad < best_sad) {
496 best_sad = this_sad;
497 best_index = i;
498 }
499 }
500
501 // Note the index of the mv that worked best in the reference list.
502 x->mv_best_ref_index[ref_frame] = best_index;
503 x->max_mv_context[ref_frame] = max_mv;
504 x->pred_mv_sad[ref_frame] = best_sad;
505 }
506
507 void vp9_setup_pred_block(const MACROBLOCKD *xd,
508 struct buf_2d dst[MAX_MB_PLANE],
509 const YV12_BUFFER_CONFIG *src,
510 int mi_row, int mi_col,
511 const struct scale_factors *scale,
512 const struct scale_factors *scale_uv) {
513 int i;
514
515 dst[0].buf = src->y_buffer;
516 dst[0].stride = src->y_stride;
517 dst[1].buf = src->u_buffer;
518 dst[2].buf = src->v_buffer;
519 dst[1].stride = dst[2].stride = src->uv_stride;
520
521 for (i = 0; i < MAX_MB_PLANE; ++i) {
522 setup_pred_plane(dst + i, dst[i].buf, dst[i].stride, mi_row, mi_col,
523 i ? scale_uv : scale,
524 xd->plane[i].subsampling_x, xd->plane[i].subsampling_y);
525 }
526 }
527
528 int vp9_raster_block_offset(BLOCK_SIZE plane_bsize,
529 int raster_block, int stride) {
530 const int bw = b_width_log2_lookup[plane_bsize];
531 const int y = 4 * (raster_block >> bw);
532 const int x = 4 * (raster_block & ((1 << bw) - 1));
533 return y * stride + x;
534 }
535
536 int16_t* vp9_raster_block_offset_int16(BLOCK_SIZE plane_bsize,
537 int raster_block, int16_t *base) {
538 const int stride = 4 * num_4x4_blocks_wide_lookup[plane_bsize];
539 return base + vp9_raster_block_offset(plane_bsize, raster_block, stride);
540 }
541
542 YV12_BUFFER_CONFIG *vp9_get_scaled_ref_frame(const VP9_COMP *cpi,
543 int ref_frame) {
544 const VP9_COMMON *const cm = &cpi->common;
545 const int scaled_idx = cpi->scaled_ref_idx[ref_frame - 1];
546 const int ref_idx = get_ref_frame_buf_idx(cpi, ref_frame);
547 return
548 (scaled_idx != ref_idx && scaled_idx != INVALID_IDX) ?
549 &cm->buffer_pool->frame_bufs[scaled_idx].buf : NULL;
550 }
551
552 int vp9_get_switchable_rate(const VP9_COMP *cpi, const MACROBLOCKD *const xd) {
553 const MB_MODE_INFO *const mbmi = &xd->mi[0]->mbmi;
554 const int ctx = vp9_get_pred_context_switchable_interp(xd);
555 return SWITCHABLE_INTERP_RATE_FACTOR *
556 cpi->switchable_interp_costs[ctx][mbmi->interp_filter];
557 }
558
559 void vp9_set_rd_speed_thresholds(VP9_COMP *cpi) {
560 int i;
561 RD_OPT *const rd = &cpi->rd;
562 SPEED_FEATURES *const sf = &cpi->sf;
563
564 // Set baseline threshold values.
565 for (i = 0; i < MAX_MODES; ++i)
566 rd->thresh_mult[i] = cpi->oxcf.mode == BEST ? -500 : 0;
567
568 if (sf->adaptive_rd_thresh) {
569 rd->thresh_mult[THR_NEARESTMV] = 300;
570 rd->thresh_mult[THR_NEARESTG] = 300;
571 rd->thresh_mult[THR_NEARESTA] = 300;
572 } else {
573 rd->thresh_mult[THR_NEARESTMV] = 0;
574 rd->thresh_mult[THR_NEARESTG] = 0;
575 rd->thresh_mult[THR_NEARESTA] = 0;
576 }
577
578 rd->thresh_mult[THR_DC] += 1000;
579
580 rd->thresh_mult[THR_NEWMV] += 1000;
581 rd->thresh_mult[THR_NEWA] += 1000;
582 rd->thresh_mult[THR_NEWG] += 1000;
583
584 rd->thresh_mult[THR_NEARMV] += 1000;
585 rd->thresh_mult[THR_NEARA] += 1000;
586 rd->thresh_mult[THR_COMP_NEARESTLA] += 1000;
587 rd->thresh_mult[THR_COMP_NEARESTGA] += 1000;
588
589 rd->thresh_mult[THR_TM] += 1000;
590
591 rd->thresh_mult[THR_COMP_NEARLA] += 1500;
592 rd->thresh_mult[THR_COMP_NEWLA] += 2000;
593 rd->thresh_mult[THR_NEARG] += 1000;
594 rd->thresh_mult[THR_COMP_NEARGA] += 1500;
595 rd->thresh_mult[THR_COMP_NEWGA] += 2000;
596
597 rd->thresh_mult[THR_ZEROMV] += 2000;
598 rd->thresh_mult[THR_ZEROG] += 2000;
599 rd->thresh_mult[THR_ZEROA] += 2000;
600 rd->thresh_mult[THR_COMP_ZEROLA] += 2500;
601 rd->thresh_mult[THR_COMP_ZEROGA] += 2500;
602
603 rd->thresh_mult[THR_H_PRED] += 2000;
604 rd->thresh_mult[THR_V_PRED] += 2000;
605 rd->thresh_mult[THR_D45_PRED ] += 2500;
606 rd->thresh_mult[THR_D135_PRED] += 2500;
607 rd->thresh_mult[THR_D117_PRED] += 2500;
608 rd->thresh_mult[THR_D153_PRED] += 2500;
609 rd->thresh_mult[THR_D207_PRED] += 2500;
610 rd->thresh_mult[THR_D63_PRED] += 2500;
611 }
612
613 void vp9_set_rd_speed_thresholds_sub8x8(VP9_COMP *cpi) {
614 static const int thresh_mult[2][MAX_REFS] =
615 {{2500, 2500, 2500, 4500, 4500, 2500},
616 {2000, 2000, 2000, 4000, 4000, 2000}};
617 RD_OPT *const rd = &cpi->rd;
618 const int idx = cpi->oxcf.mode == BEST;
619 memcpy(rd->thresh_mult_sub8x8, thresh_mult[idx], sizeof(thresh_mult[idx]));
620 }
621
622 void vp9_update_rd_thresh_fact(int (*factor_buf)[MAX_MODES], int rd_thresh,
623 int bsize, int best_mode_index) {
624 if (rd_thresh > 0) {
625 const int top_mode = bsize < BLOCK_8X8 ? MAX_REFS : MAX_MODES;
626 int mode;
627 for (mode = 0; mode < top_mode; ++mode) {
628 const BLOCK_SIZE min_size = MAX(bsize - 1, BLOCK_4X4);
629 const BLOCK_SIZE max_size = MIN(bsize + 2, BLOCK_64X64);
630 BLOCK_SIZE bs;
631 for (bs = min_size; bs <= max_size; ++bs) {
632 int *const fact = &factor_buf[bs][mode];
633 if (mode == best_mode_index) {
634 *fact -= (*fact >> 4);
635 } else {
636 *fact = MIN(*fact + RD_THRESH_INC,
637 rd_thresh * RD_THRESH_MAX_FACT);
638 }
639 }
640 }
641 }
642 }
643
644 int vp9_get_intra_cost_penalty(int qindex, int qdelta,
645 vpx_bit_depth_t bit_depth) {
646 const int q = vp9_dc_quant(qindex, qdelta, bit_depth);
647 #if CONFIG_VP9_HIGHBITDEPTH
648 switch (bit_depth) {
649 case VPX_BITS_8:
650 return 20 * q;
651 case VPX_BITS_10:
652 return 5 * q;
653 case VPX_BITS_12:
654 return ROUND_POWER_OF_TWO(5 * q, 2);
655 default:
656 assert(0 && "bit_depth should be VPX_BITS_8, VPX_BITS_10 or VPX_BITS_12");
657 return -1;
658 }
659 #else
660 return 20 * q;
661 #endif // CONFIG_VP9_HIGHBITDEPTH
662 }
663
OLDNEW

Powered by Google App Engine
This is Rietveld 408576698