loopfilter.c 52.4 KB
Newer Older
John Koleszar's avatar
John Koleszar committed
1
/*
Yaowu Xu's avatar
Yaowu Xu committed
2
 * Copyright (c) 2016, Alliance for Open Media. All rights reserved
John Koleszar's avatar
John Koleszar committed
3
 *
Yaowu Xu's avatar
Yaowu Xu committed
4 5 6 7 8 9
 * This source code is subject to the terms of the BSD 2 Clause License and
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
 * was not distributed with this source code in the LICENSE file, you can
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
 * Media Patent License 1.0 was not distributed with this source code in the
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
John Koleszar's avatar
John Koleszar committed
10
 */
11

Zoe Liu's avatar
Zoe Liu committed
12 13
#include <stdlib.h>

Yaowu Xu's avatar
Yaowu Xu committed
14 15 16
#include "./aom_config.h"
#include "./aom_dsp_rtcd.h"
#include "aom_dsp/aom_dsp_common.h"
17
#include "aom_ports/mem.h"
John Koleszar's avatar
John Koleszar committed
18

19
static INLINE int8_t signed_char_clamp(int t) {
20
  return (int8_t)clamp(t, -128, 127);
John Koleszar's avatar
John Koleszar committed
21 22
}

23 24 25
#define PARALLEL_DEBLOCKING_11_TAP 0
#define PARALLEL_DEBLOCKING_9_TAP 0

Ola Hugosson's avatar
Ola Hugosson committed
26 27 28 29 30 31 32 33
#if CONFIG_DEBLOCK_13TAP
#define PARALLEL_DEBLOCKING_13_TAP 1
#define PARALLEL_DEBLOCKING_5_TAP_CHROMA 1
#else
#define PARALLEL_DEBLOCKING_13_TAP 0
#define PARALLEL_DEBLOCKING_5_TAP_CHROMA 0
#endif

34
#if CONFIG_HIGHBITDEPTH
35 36
static INLINE int16_t signed_char_clamp_high(int t, int bd) {
  switch (bd) {
clang-format's avatar
clang-format committed
37 38
    case 10: return (int16_t)clamp(t, -128 * 4, 128 * 4 - 1);
    case 12: return (int16_t)clamp(t, -128 * 16, 128 * 16 - 1);
39
    case 8:
clang-format's avatar
clang-format committed
40
    default: return (int16_t)clamp(t, -128, 128 - 1);
41 42 43
  }
}
#endif
44
#if CONFIG_PARALLEL_DEBLOCKING
Dmitry Kovalev's avatar
Dmitry Kovalev committed
45
// should we apply any filter at all: 11111111 yes, 00000000 no
46 47 48 49 50 51 52 53 54
static INLINE int8_t filter_mask2(uint8_t limit, uint8_t blimit, uint8_t p1,
                                  uint8_t p0, uint8_t q0, uint8_t q1) {
  int8_t mask = 0;
  mask |= (abs(p1 - p0) > limit) * -1;
  mask |= (abs(q1 - q0) > limit) * -1;
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit) * -1;
  return ~mask;
}
#endif  // CONFIG_PARALLEL_DEBLOCKING
clang-format's avatar
clang-format committed
55 56 57
static INLINE int8_t filter_mask(uint8_t limit, uint8_t blimit, uint8_t p3,
                                 uint8_t p2, uint8_t p1, uint8_t p0, uint8_t q0,
                                 uint8_t q1, uint8_t q2, uint8_t q3) {
58
  int8_t mask = 0;
John Koleszar's avatar
John Koleszar committed
59 60 61 62 63 64
  mask |= (abs(p3 - p2) > limit) * -1;
  mask |= (abs(p2 - p1) > limit) * -1;
  mask |= (abs(p1 - p0) > limit) * -1;
  mask |= (abs(q1 - q0) > limit) * -1;
  mask |= (abs(q2 - q1) > limit) * -1;
  mask |= (abs(q3 - q2) > limit) * -1;
clang-format's avatar
clang-format committed
65
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit) * -1;
Dmitry Kovalev's avatar
Dmitry Kovalev committed
66
  return ~mask;
John Koleszar's avatar
John Koleszar committed
67 68
}

Ola Hugosson's avatar
Ola Hugosson committed
69 70 71 72 73 74 75 76 77 78 79 80 81
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE int8_t flat_mask3_chroma(uint8_t thresh, uint8_t p2, uint8_t p1,
                                       uint8_t p0, uint8_t q0, uint8_t q1,
                                       uint8_t q2) {
  int8_t mask = 0;
  mask |= (abs(p1 - p0) > thresh) * -1;
  mask |= (abs(q1 - q0) > thresh) * -1;
  mask |= (abs(p2 - p0) > thresh) * -1;
  mask |= (abs(q2 - q0) > thresh) * -1;
  return ~mask;
}
#endif

clang-format's avatar
clang-format committed
82 83
static INLINE int8_t flat_mask4(uint8_t thresh, uint8_t p3, uint8_t p2,
                                uint8_t p1, uint8_t p0, uint8_t q0, uint8_t q1,
84
                                uint8_t q2, uint8_t q3) {
Dmitry Kovalev's avatar
Dmitry Kovalev committed
85 86 87 88 89 90 91 92
  int8_t mask = 0;
  mask |= (abs(p1 - p0) > thresh) * -1;
  mask |= (abs(q1 - q0) > thresh) * -1;
  mask |= (abs(p2 - p0) > thresh) * -1;
  mask |= (abs(q2 - q0) > thresh) * -1;
  mask |= (abs(p3 - p0) > thresh) * -1;
  mask |= (abs(q3 - q0) > thresh) * -1;
  return ~mask;
93 94
}

95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117
#if PARALLEL_DEBLOCKING_9_TAP
static INLINE int8_t flat_mask2(uint8_t thresh, uint8_t p4, uint8_t p0,
                                uint8_t q0, uint8_t q4) {
  int8_t mask = 0;
  mask |= (abs(p4 - p0) > thresh) * -1;
  mask |= (abs(q4 - q0) > thresh) * -1;
  return ~mask;
}
#endif

#if PARALLEL_DEBLOCKING_11_TAP
static INLINE int8_t flat_mask3(uint8_t thresh, uint8_t p5, uint8_t p4,
                                uint8_t p0, uint8_t q0, uint8_t q4,
                                uint8_t q5) {
  int8_t mask = 0;
  mask |= (abs(p4 - p0) > thresh) * -1;
  mask |= (abs(q4 - q0) > thresh) * -1;
  mask |= (abs(p5 - p0) > thresh) * -1;
  mask |= (abs(q5 - q0) > thresh) * -1;
  return ~mask;
}
#endif

118
#if !PARALLEL_DEBLOCKING_13_TAP
clang-format's avatar
clang-format committed
119 120 121 122
static INLINE int8_t flat_mask5(uint8_t thresh, uint8_t p4, uint8_t p3,
                                uint8_t p2, uint8_t p1, uint8_t p0, uint8_t q0,
                                uint8_t q1, uint8_t q2, uint8_t q3,
                                uint8_t q4) {
Dmitry Kovalev's avatar
Dmitry Kovalev committed
123 124 125 126
  int8_t mask = ~flat_mask4(thresh, p3, p2, p1, p0, q0, q1, q2, q3);
  mask |= (abs(p4 - p0) > thresh) * -1;
  mask |= (abs(q4 - q0) > thresh) * -1;
  return ~mask;
127
}
128
#endif
129

130
// is there high edge variance internal edge: 11111111 yes, 00000000 no
131 132
static INLINE int8_t hev_mask(uint8_t thresh, uint8_t p1, uint8_t p0,
                              uint8_t q0, uint8_t q1) {
133
  int8_t hev = 0;
clang-format's avatar
clang-format committed
134 135
  hev |= (abs(p1 - p0) > thresh) * -1;
  hev |= (abs(q1 - q0) > thresh) * -1;
John Koleszar's avatar
John Koleszar committed
136
  return hev;
John Koleszar's avatar
John Koleszar committed
137 138
}

139
static INLINE void filter4(int8_t mask, uint8_t thresh, uint8_t *op1,
140
                           uint8_t *op0, uint8_t *oq0, uint8_t *oq1) {
Dmitry Kovalev's avatar
Dmitry Kovalev committed
141
  int8_t filter1, filter2;
John Koleszar's avatar
John Koleszar committed
142

clang-format's avatar
clang-format committed
143 144 145 146
  const int8_t ps1 = (int8_t)*op1 ^ 0x80;
  const int8_t ps0 = (int8_t)*op0 ^ 0x80;
  const int8_t qs0 = (int8_t)*oq0 ^ 0x80;
  const int8_t qs1 = (int8_t)*oq1 ^ 0x80;
147
  const uint8_t hev = hev_mask(thresh, *op1, *op0, *oq0, *oq1);
Dmitry Kovalev's avatar
Dmitry Kovalev committed
148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164

  // add outer taps if we have high edge variance
  int8_t filter = signed_char_clamp(ps1 - qs1) & hev;

  // inner taps
  filter = signed_char_clamp(filter + 3 * (qs0 - ps0)) & mask;

  // save bottom 3 bits so that we round one side +4 and the other +3
  // if it equals 4 we'll set to adjust by -1 to account for the fact
  // we'd round 3 the other way
  filter1 = signed_char_clamp(filter + 4) >> 3;
  filter2 = signed_char_clamp(filter + 3) >> 3;

  *oq0 = signed_char_clamp(qs0 - filter1) ^ 0x80;
  *op0 = signed_char_clamp(ps0 + filter2) ^ 0x80;

  // outer tap adjustments
Dmitry Kovalev's avatar
Dmitry Kovalev committed
165
  filter = ROUND_POWER_OF_TWO(filter1, 1) & ~hev;
John Koleszar's avatar
John Koleszar committed
166

Dmitry Kovalev's avatar
Dmitry Kovalev committed
167 168
  *oq1 = signed_char_clamp(qs1 - filter) ^ 0x80;
  *op1 = signed_char_clamp(ps1 + filter) ^ 0x80;
John Koleszar's avatar
John Koleszar committed
169
}
170

Yaowu Xu's avatar
Yaowu Xu committed
171
void aom_lpf_horizontal_4_c(uint8_t *s, int p /* pitch */,
Jim Bankoski's avatar
Jim Bankoski committed
172
                            const uint8_t *blimit, const uint8_t *limit,
173
                            const uint8_t *thresh) {
174
  int i;
175
#if CONFIG_PARALLEL_DEBLOCKING
176 177 178 179
  int count = 4;
#else
  int count = 8;
#endif
John Koleszar's avatar
John Koleszar committed
180

Dmitry Kovalev's avatar
Dmitry Kovalev committed
181 182
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
183
  for (i = 0; i < count; ++i) {
184
#if !CONFIG_PARALLEL_DEBLOCKING
185
    const uint8_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
clang-format's avatar
clang-format committed
186 187 188
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
189 190 191 192 193
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint8_t p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p];
    const int8_t mask = filter_mask2(*limit, *blimit, p1, p0, q0, q1);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
194
    filter4(mask, *thresh, s - 2 * p, s - 1 * p, s, s + 1 * p);
John Koleszar's avatar
John Koleszar committed
195
    ++s;
196
  }
John Koleszar's avatar
John Koleszar committed
197 198
}

Yaowu Xu's avatar
Yaowu Xu committed
199
void aom_lpf_horizontal_4_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
200 201 202
                                 const uint8_t *limit0, const uint8_t *thresh0,
                                 const uint8_t *blimit1, const uint8_t *limit1,
                                 const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
203
  aom_lpf_horizontal_4_c(s, p, blimit0, limit0, thresh0);
Yi Luo's avatar
Yi Luo committed
204 205 206
#if CONFIG_PARALLEL_DEBLOCKING
  aom_lpf_horizontal_4_c(s + 4, p, blimit1, limit1, thresh1);
#else
Yaowu Xu's avatar
Yaowu Xu committed
207
  aom_lpf_horizontal_4_c(s + 8, p, blimit1, limit1, thresh1);
Yi Luo's avatar
Yi Luo committed
208
#endif
209 210
}

Yaowu Xu's avatar
Yaowu Xu committed
211
void aom_lpf_vertical_4_c(uint8_t *s, int pitch, const uint8_t *blimit,
212
                          const uint8_t *limit, const uint8_t *thresh) {
213
  int i;
214
#if CONFIG_PARALLEL_DEBLOCKING
215 216 217 218
  int count = 4;
#else
  int count = 8;
#endif
John Koleszar's avatar
John Koleszar committed
219

Dmitry Kovalev's avatar
Dmitry Kovalev committed
220 221
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
222
  for (i = 0; i < count; ++i) {
223
#if !CONFIG_PARALLEL_DEBLOCKING
224
    const uint8_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
clang-format's avatar
clang-format committed
225 226 227
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
228 229 230 231 232
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint8_t p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1];
    const int8_t mask = filter_mask2(*limit, *blimit, p1, p0, q0, q1);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
233
    filter4(mask, *thresh, s - 2, s - 1, s, s + 1);
Dmitry Kovalev's avatar
Dmitry Kovalev committed
234
    s += pitch;
235
  }
John Koleszar's avatar
John Koleszar committed
236
}
237

Yaowu Xu's avatar
Yaowu Xu committed
238
void aom_lpf_vertical_4_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
239 240 241
                               const uint8_t *limit0, const uint8_t *thresh0,
                               const uint8_t *blimit1, const uint8_t *limit1,
                               const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
242
  aom_lpf_vertical_4_c(s, pitch, blimit0, limit0, thresh0);
Yi Luo's avatar
Yi Luo committed
243 244 245
#if CONFIG_PARALLEL_DEBLOCKING
  aom_lpf_vertical_4_c(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
#else
Yaowu Xu's avatar
Yaowu Xu committed
246
  aom_lpf_vertical_4_c(s + 8 * pitch, pitch, blimit1, limit1, thresh1);
Yi Luo's avatar
Yi Luo committed
247
#endif
248 249
}

Ola Hugosson's avatar
Ola Hugosson committed
250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE void filter6(int8_t mask, uint8_t thresh, int8_t flat,
                           uint8_t *op2, uint8_t *op1, uint8_t *op0,
                           uint8_t *oq0, uint8_t *oq1, uint8_t *oq2) {
  if (flat && mask) {
    const uint8_t p2 = *op2, p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2;

    // 5-tap filter [1, 2, 2, 2, 1]
    *op1 = ROUND_POWER_OF_TWO(p2 * 3 + p1 * 2 + p0 * 2 + q0, 3);
    *op0 = ROUND_POWER_OF_TWO(p2 + p1 * 2 + p0 * 2 + q0 * 2 + q1, 3);
    *oq0 = ROUND_POWER_OF_TWO(p1 + p0 * 2 + q0 * 2 + q1 * 2 + q2, 3);
    *oq1 = ROUND_POWER_OF_TWO(p0 + q0 * 2 + q1 * 2 + q2 * 3, 3);
  } else {
    filter4(mask, thresh, op1, op0, oq0, oq1);
  }
}
#endif

269
static INLINE void filter8(int8_t mask, uint8_t thresh, int8_t flat,
clang-format's avatar
clang-format committed
270 271
                           uint8_t *op3, uint8_t *op2, uint8_t *op1,
                           uint8_t *op0, uint8_t *oq0, uint8_t *oq1,
272
                           uint8_t *oq2, uint8_t *oq3) {
John Koleszar's avatar
John Koleszar committed
273
  if (flat && mask) {
274 275
    const uint8_t p3 = *op3, p2 = *op2, p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3;
John Koleszar's avatar
John Koleszar committed
276

Dmitry Kovalev's avatar
Dmitry Kovalev committed
277 278 279 280 281 282 283
    // 7-tap filter [1, 1, 1, 2, 1, 1, 1]
    *op2 = ROUND_POWER_OF_TWO(p3 + p3 + p3 + 2 * p2 + p1 + p0 + q0, 3);
    *op1 = ROUND_POWER_OF_TWO(p3 + p3 + p2 + 2 * p1 + p0 + q0 + q1, 3);
    *op0 = ROUND_POWER_OF_TWO(p3 + p2 + p1 + 2 * p0 + q0 + q1 + q2, 3);
    *oq0 = ROUND_POWER_OF_TWO(p2 + p1 + p0 + 2 * q0 + q1 + q2 + q3, 3);
    *oq1 = ROUND_POWER_OF_TWO(p1 + p0 + q0 + 2 * q1 + q2 + q3 + q3, 3);
    *oq2 = ROUND_POWER_OF_TWO(p0 + q0 + q1 + 2 * q2 + q3 + q3 + q3, 3);
John Koleszar's avatar
John Koleszar committed
284
  } else {
clang-format's avatar
clang-format committed
285
    filter4(mask, thresh, op1, op0, oq0, oq1);
John Koleszar's avatar
John Koleszar committed
286
  }
287
}
288

Ola Hugosson's avatar
Ola Hugosson committed
289 290 291 292
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
void aom_lpf_horizontal_6_c(uint8_t *s, int p, const uint8_t *blimit,
                            const uint8_t *limit, const uint8_t *thresh) {
  int i;
293
#if CONFIG_PARALLEL_DEBLOCKING
Ola Hugosson's avatar
Ola Hugosson committed
294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314
  int count = 4;
#else
  int count = 8;
#endif

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
  for (i = 0; i < count; ++i) {
    const uint8_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];

    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
    const int8_t flat = flat_mask3_chroma(1, p2, p1, p0, q0, q1, q2);
    filter6(mask, *thresh, flat, s - 3 * p, s - 2 * p, s - 1 * p, s, s + 1 * p,
            s + 2 * p);
    ++s;
  }
}
#endif

Yaowu Xu's avatar
Yaowu Xu committed
315
void aom_lpf_horizontal_8_c(uint8_t *s, int p, const uint8_t *blimit,
316
                            const uint8_t *limit, const uint8_t *thresh) {
317
  int i;
318
#if CONFIG_PARALLEL_DEBLOCKING
319 320 321 322
  int count = 4;
#else
  int count = 8;
#endif
John Koleszar's avatar
John Koleszar committed
323

Dmitry Kovalev's avatar
Dmitry Kovalev committed
324 325
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
326
  for (i = 0; i < count; ++i) {
327 328 329
    const uint8_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];

clang-format's avatar
clang-format committed
330 331
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
332
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
clang-format's avatar
clang-format committed
333 334
    filter8(mask, *thresh, flat, s - 4 * p, s - 3 * p, s - 2 * p, s - 1 * p, s,
            s + 1 * p, s + 2 * p, s + 3 * p);
John Koleszar's avatar
John Koleszar committed
335
    ++s;
336
  }
John Koleszar's avatar
John Koleszar committed
337
}
338

Yaowu Xu's avatar
Yaowu Xu committed
339
void aom_lpf_horizontal_8_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
340 341 342
                                 const uint8_t *limit0, const uint8_t *thresh0,
                                 const uint8_t *blimit1, const uint8_t *limit1,
                                 const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
343
  aom_lpf_horizontal_8_c(s, p, blimit0, limit0, thresh0);
Yi Luo's avatar
Yi Luo committed
344 345 346
#if CONFIG_PARALLEL_DEBLOCKING
  aom_lpf_horizontal_8_c(s + 4, p, blimit1, limit1, thresh1);
#else
Yaowu Xu's avatar
Yaowu Xu committed
347
  aom_lpf_horizontal_8_c(s + 8, p, blimit1, limit1, thresh1);
Yi Luo's avatar
Yi Luo committed
348
#endif
349 350
}

Ola Hugosson's avatar
Ola Hugosson committed
351 352 353 354
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
void aom_lpf_vertical_6_c(uint8_t *s, int pitch, const uint8_t *blimit,
                          const uint8_t *limit, const uint8_t *thresh) {
  int i;
355
#if CONFIG_PARALLEL_DEBLOCKING
Ola Hugosson's avatar
Ola Hugosson committed
356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372
  int count = 4;
#else
  int count = 8;
#endif

  for (i = 0; i < count; ++i) {
    const uint8_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
    const int8_t flat = flat_mask3_chroma(1, p2, p1, p0, q0, q1, q2);
    filter6(mask, *thresh, flat, s - 3, s - 2, s - 1, s, s + 1, s + 2);
    s += pitch;
  }
}
#endif

Yaowu Xu's avatar
Yaowu Xu committed
373
void aom_lpf_vertical_8_c(uint8_t *s, int pitch, const uint8_t *blimit,
374
                          const uint8_t *limit, const uint8_t *thresh) {
375
  int i;
376
#if CONFIG_PARALLEL_DEBLOCKING
377 378 379 380
  int count = 4;
#else
  int count = 8;
#endif
381

382
  for (i = 0; i < count; ++i) {
383 384
    const uint8_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
clang-format's avatar
clang-format committed
385 386
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
387
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
clang-format's avatar
clang-format committed
388 389
    filter8(mask, *thresh, flat, s - 4, s - 3, s - 2, s - 1, s, s + 1, s + 2,
            s + 3);
Dmitry Kovalev's avatar
Dmitry Kovalev committed
390
    s += pitch;
391
  }
John Koleszar's avatar
John Koleszar committed
392 393
}

Yaowu Xu's avatar
Yaowu Xu committed
394
void aom_lpf_vertical_8_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
395 396 397
                               const uint8_t *limit0, const uint8_t *thresh0,
                               const uint8_t *blimit1, const uint8_t *limit1,
                               const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
398
  aom_lpf_vertical_8_c(s, pitch, blimit0, limit0, thresh0);
Yi Luo's avatar
Yi Luo committed
399 400 401
#if CONFIG_PARALLEL_DEBLOCKING
  aom_lpf_vertical_8_c(s + 4 * pitch, pitch, blimit1, limit1, thresh1);
#else
Yaowu Xu's avatar
Yaowu Xu committed
402
  aom_lpf_vertical_8_c(s + 8 * pitch, pitch, blimit1, limit1, thresh1);
Yi Luo's avatar
Yi Luo committed
403
#endif
404 405
}

Ola Hugosson's avatar
Ola Hugosson committed
406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455
#if PARALLEL_DEBLOCKING_13_TAP
static INLINE void filter14(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op6, uint8_t *op5,
                            uint8_t *op4, uint8_t *op3, uint8_t *op2,
                            uint8_t *op1, uint8_t *op0, uint8_t *oq0,
                            uint8_t *oq1, uint8_t *oq2, uint8_t *oq3,
                            uint8_t *oq4, uint8_t *oq5, uint8_t *oq6) {
  if (flat2 && flat && mask) {
    const uint8_t p6 = *op6, p5 = *op5, p4 = *op4, p3 = *op3, p2 = *op2,
                  p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4,
                  q5 = *oq5, q6 = *oq6;

    // 13-tap filter [1, 1, 1, 1, 1, 2, 2, 2, 1, 1, 1, 1, 1]
    *op5 = ROUND_POWER_OF_TWO(p6 * 7 + p5 * 2 + p4 * 2 + p3 + p2 + p1 + p0 + q0,
                              4);
    *op4 = ROUND_POWER_OF_TWO(
        p6 * 5 + p5 * 2 + p4 * 2 + p3 * 2 + p2 + p1 + p0 + q0 + q1, 4);
    *op3 = ROUND_POWER_OF_TWO(
        p6 * 4 + p5 + p4 * 2 + p3 * 2 + p2 * 2 + p1 + p0 + q0 + q1 + q2, 4);
    *op2 = ROUND_POWER_OF_TWO(
        p6 * 3 + p5 + p4 + p3 * 2 + p2 * 2 + p1 * 2 + p0 + q0 + q1 + q2 + q3,
        4);
    *op1 = ROUND_POWER_OF_TWO(p6 * 2 + p5 + p4 + p3 + p2 * 2 + p1 * 2 + p0 * 2 +
                                  q0 + q1 + q2 + q3 + q4,
                              4);
    *op0 = ROUND_POWER_OF_TWO(p6 + p5 + p4 + p3 + p2 + p1 * 2 + p0 * 2 +
                                  q0 * 2 + q1 + q2 + q3 + q4 + q5,
                              4);
    *oq0 = ROUND_POWER_OF_TWO(p5 + p4 + p3 + p2 + p1 + p0 * 2 + q0 * 2 +
                                  q1 * 2 + q2 + q3 + q4 + q5 + q6,
                              4);
    *oq1 = ROUND_POWER_OF_TWO(p4 + p3 + p2 + p1 + p0 + q0 * 2 + q1 * 2 +
                                  q2 * 2 + q3 + q4 + q5 + q6 * 2,
                              4);
    *oq2 = ROUND_POWER_OF_TWO(
        p3 + p2 + p1 + p0 + q0 + q1 * 2 + q2 * 2 + q3 * 2 + q4 + q5 + q6 * 3,
        4);
    *oq3 = ROUND_POWER_OF_TWO(
        p2 + p1 + p0 + q0 + q1 + q2 * 2 + q3 * 2 + q4 * 2 + q5 + q6 * 4, 4);
    *oq4 = ROUND_POWER_OF_TWO(
        p1 + p0 + q0 + q1 + q2 + q3 * 2 + q4 * 2 + q5 * 2 + q6 * 5, 4);
    *oq5 = ROUND_POWER_OF_TWO(p0 + q0 + q1 + q2 + q3 + q4 * 2 + q5 * 2 + q6 * 7,
                              4);
  } else {
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
  }
}
#endif

456
#if PARALLEL_DEBLOCKING_11_TAP
457 458
static INLINE void filter12(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op5, uint8_t *op4,
459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486
                            uint8_t *op3, uint8_t *op2, uint8_t *op1,
                            uint8_t *op0, uint8_t *oq0, uint8_t *oq1,
                            uint8_t *oq2, uint8_t *oq3, uint8_t *oq4,
                            uint8_t *oq5) {
  if (flat2 && flat && mask) {
    const uint8_t p5 = *op5, p4 = *op4, p3 = *op3, p2 = *op2, p1 = *op1,
                  p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4,
                  q5 = *oq5;

    // 11-tap filter [1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1]
    *op4 = (p5 * 5 + p4 * 2 + p3 + p2 + p1 + p0 + q0 + 6) / 12;
    *op3 = (p5 * 4 + p4 + p3 * 2 + p2 + p1 + p0 + q0 + q1 + 6) / 12;
    *op2 = (p5 * 3 + p4 + p3 + p2 * 2 + p1 + p0 + q0 + q1 + q2 + 6) / 12;
    *op1 = (p5 * 2 + p4 + p3 + p2 + p1 * 2 + p0 + q0 + q1 + q2 + q3 + 6) / 12;
    *op0 = (p5 + p4 + p3 + p2 + p1 + p0 * 2 + q0 + q1 + q2 + q3 + q4 + 6) / 12;
    *oq0 = (p4 + p3 + p2 + p1 + p0 + q0 * 2 + q1 + q2 + q3 + q4 + q5 + 6) / 12;
    *oq1 = (p3 + p2 + p1 + p0 + q0 + q1 * 2 + q2 + q3 + q4 + q5 * 2 + 6) / 12;
    *oq2 = (p2 + p1 + p0 + q0 + q1 + q2 * 2 + q3 + q4 + q5 * 3 + 6) / 12;
    *oq3 = (p1 + p0 + q0 + q1 + q2 + q3 * 2 + q4 + q5 * 4 + 6) / 12;
    *oq4 = (p0 + q0 + q1 + q2 + q3 + q4 * 2 + q5 * 5 + 6) / 12;
  } else {
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
  }
}
#endif

#if PARALLEL_DEBLOCKING_9_TAP
487 488
static INLINE void filter10(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op4, uint8_t *op3,
489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510
                            uint8_t *op2, uint8_t *op1, uint8_t *op0,
                            uint8_t *oq0, uint8_t *oq1, uint8_t *oq2,
                            uint8_t *oq3, uint8_t *oq4) {
  if (flat2 && flat && mask) {
    const uint8_t p4 = *op4, p3 = *op3, p2 = *op2, p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4;

    // 9-tap filter [1, 1, 1, 1, 2, 1, 1, 1, 1]
    *op3 = (p4 * 4 + p3 * 2 + p2 + p1 + p0 + q0 + 5) / 10;
    *op2 = (p4 * 3 + p3 + p2 * 2 + p1 + p0 + q0 + q1 + 5) / 10;
    *op1 = (p4 * 2 + p3 + p2 + p1 * 2 + p0 + q0 + q1 + q2 + 5) / 10;
    *op0 = (p4 + p3 + p2 + p1 + p0 * 2 + q0 + q1 + q2 + q3 + 5) / 10;
    *oq0 = (p3 + p2 + p1 + p0 + q0 * 2 + q1 + q2 + q3 + q4 + 5) / 10;
    *oq1 = (p2 + p1 + p0 + q0 + q1 * 2 + q2 + q3 + q4 * 2 + 5) / 10;
    *oq2 = (p1 + p0 + q0 + q1 + q2 * 2 + q3 + q4 * 3 + 5) / 10;
    *oq3 = (p0 + q0 + q1 + q2 + q3 * 2 + q4 * 4 + 5) / 10;
  } else {
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
  }
}
#endif

511
#if !PARALLEL_DEBLOCKING_13_TAP
512 513
static INLINE void filter16(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op7, uint8_t *op6,
clang-format's avatar
clang-format committed
514 515 516 517
                            uint8_t *op5, uint8_t *op4, uint8_t *op3,
                            uint8_t *op2, uint8_t *op1, uint8_t *op0,
                            uint8_t *oq0, uint8_t *oq1, uint8_t *oq2,
                            uint8_t *oq3, uint8_t *oq4, uint8_t *oq5,
518
                            uint8_t *oq6, uint8_t *oq7) {
519
  if (flat2 && flat && mask) {
clang-format's avatar
clang-format committed
520 521
    const uint8_t p7 = *op7, p6 = *op6, p5 = *op5, p4 = *op4, p3 = *op3,
                  p2 = *op2, p1 = *op1, p0 = *op0;
522

clang-format's avatar
clang-format committed
523 524
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4,
                  q5 = *oq5, q6 = *oq6, q7 = *oq7;
525

Dmitry Kovalev's avatar
Dmitry Kovalev committed
526
    // 15-tap filter [1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1]
clang-format's avatar
clang-format committed
527 528 529 530 531 532 533 534 535 536 537
    *op6 = ROUND_POWER_OF_TWO(
        p7 * 7 + p6 * 2 + p5 + p4 + p3 + p2 + p1 + p0 + q0, 4);
    *op5 = ROUND_POWER_OF_TWO(
        p7 * 6 + p6 + p5 * 2 + p4 + p3 + p2 + p1 + p0 + q0 + q1, 4);
    *op4 = ROUND_POWER_OF_TWO(
        p7 * 5 + p6 + p5 + p4 * 2 + p3 + p2 + p1 + p0 + q0 + q1 + q2, 4);
    *op3 = ROUND_POWER_OF_TWO(
        p7 * 4 + p6 + p5 + p4 + p3 * 2 + p2 + p1 + p0 + q0 + q1 + q2 + q3, 4);
    *op2 = ROUND_POWER_OF_TWO(
        p7 * 3 + p6 + p5 + p4 + p3 + p2 * 2 + p1 + p0 + q0 + q1 + q2 + q3 + q4,
        4);
538
    *op1 = ROUND_POWER_OF_TWO(p7 * 2 + p6 + p5 + p4 + p3 + p2 + p1 * 2 + p0 +
clang-format's avatar
clang-format committed
539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560
                                  q0 + q1 + q2 + q3 + q4 + q5,
                              4);
    *op0 = ROUND_POWER_OF_TWO(p7 + p6 + p5 + p4 + p3 + p2 + p1 + p0 * 2 + q0 +
                                  q1 + q2 + q3 + q4 + q5 + q6,
                              4);
    *oq0 = ROUND_POWER_OF_TWO(p6 + p5 + p4 + p3 + p2 + p1 + p0 + q0 * 2 + q1 +
                                  q2 + q3 + q4 + q5 + q6 + q7,
                              4);
    *oq1 = ROUND_POWER_OF_TWO(p5 + p4 + p3 + p2 + p1 + p0 + q0 + q1 * 2 + q2 +
                                  q3 + q4 + q5 + q6 + q7 * 2,
                              4);
    *oq2 = ROUND_POWER_OF_TWO(
        p4 + p3 + p2 + p1 + p0 + q0 + q1 + q2 * 2 + q3 + q4 + q5 + q6 + q7 * 3,
        4);
    *oq3 = ROUND_POWER_OF_TWO(
        p3 + p2 + p1 + p0 + q0 + q1 + q2 + q3 * 2 + q4 + q5 + q6 + q7 * 4, 4);
    *oq4 = ROUND_POWER_OF_TWO(
        p2 + p1 + p0 + q0 + q1 + q2 + q3 + q4 * 2 + q5 + q6 + q7 * 5, 4);
    *oq5 = ROUND_POWER_OF_TWO(
        p1 + p0 + q0 + q1 + q2 + q3 + q4 + q5 * 2 + q6 + q7 * 6, 4);
    *oq6 = ROUND_POWER_OF_TWO(
        p0 + q0 + q1 + q2 + q3 + q4 + q5 + q6 * 2 + q7 * 7, 4);
561
  } else {
562
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
563 564
  }
}
565
#endif
566

567 568 569
static void mb_lpf_horizontal_edge_w(uint8_t *s, int p, const uint8_t *blimit,
                                     const uint8_t *limit,
                                     const uint8_t *thresh, int count) {
570
  int i;
571
#if CONFIG_PARALLEL_DEBLOCKING
572 573 574 575
  int step = 4;
#else
  int step = 8;
#endif
576

Dmitry Kovalev's avatar
Dmitry Kovalev committed
577 578
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
579
  for (i = 0; i < step * count; ++i) {
580 581 582 583 584
    const uint8_t p7 = s[-8 * p], p6 = s[-7 * p], p5 = s[-6 * p],
                  p4 = s[-5 * p], p3 = s[-4 * p], p2 = s[-3 * p],
                  p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p],
                  q4 = s[4 * p], q5 = s[5 * p], q6 = s[6 * p], q7 = s[7 * p];
clang-format's avatar
clang-format committed
585 586
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
587
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
588

Ola Hugosson's avatar
Ola Hugosson committed
589 590 591 592 593 594 595 596 597 598
#if PARALLEL_DEBLOCKING_13_TAP
    (void)p7;
    (void)q7;
    const int8_t flat2 = flat_mask4(1, p6, p5, p4, p0, q0, q4, q5, q6);

    filter14(mask, *thresh, flat, flat2, s - 7 * p, s - 6 * p, s - 5 * p,
             s - 4 * p, s - 3 * p, s - 2 * p, s - 1 * p, s, s + 1 * p,
             s + 2 * p, s + 3 * p, s + 4 * p, s + 5 * p, s + 6 * p);

#elif PARALLEL_DEBLOCKING_11_TAP
599 600 601 602 603 604 605 606 607 608 609 610 611 612
    const int8_t flat2 = flat_mask3(1, p5, p4, p0, q0, q4, q5);

    filter12(mask, *thresh, flat, flat2, s - 6 * p, s - 5 * p, s - 4 * p,
             s - 3 * p, s - 2 * p, s - 1 * p, s, s + 1 * p, s + 2 * p,
             s + 3 * p, s + 4 * p, s + 5 * p);

#elif PARALLEL_DEBLOCKING_9_TAP
    const int8_t flat2 = flat_mask2(1, p4, p0, q0, q4);

    filter10(mask, *thresh, flat, flat2, s - 5 * p, s - 4 * p, s - 3 * p,
             s - 2 * p, s - 1 * p, s, s + 1 * p, s + 2 * p, s + 3 * p,
             s + 4 * p);
#else
    const int8_t flat2 = flat_mask5(1, p7, p6, p5, p4, p0, q0, q4, q5, q6, q7);
clang-format's avatar
clang-format committed
613 614 615 616 617

    filter16(mask, *thresh, flat, flat2, s - 8 * p, s - 7 * p, s - 6 * p,
             s - 5 * p, s - 4 * p, s - 3 * p, s - 2 * p, s - 1 * p, s,
             s + 1 * p, s + 2 * p, s + 3 * p, s + 4 * p, s + 5 * p, s + 6 * p,
             s + 7 * p);
618 619
#endif

620
    ++s;
621
  }
622
}
623

James Zern's avatar
James Zern committed
624 625
void aom_lpf_horizontal_16_c(uint8_t *s, int p, const uint8_t *blimit,
                             const uint8_t *limit, const uint8_t *thresh) {
626 627 628
  mb_lpf_horizontal_edge_w(s, p, blimit, limit, thresh, 1);
}

James Zern's avatar
James Zern committed
629
void aom_lpf_horizontal_16_dual_c(uint8_t *s, int p, const uint8_t *blimit,
630
                                  const uint8_t *limit, const uint8_t *thresh) {
631
#if CONFIG_PARALLEL_DEBLOCKING
632 633
  mb_lpf_horizontal_edge_w(s, p, blimit, limit, thresh, 1);
#else
634
  mb_lpf_horizontal_edge_w(s, p, blimit, limit, thresh, 2);
635
#endif
636 637
}

clang-format's avatar
clang-format committed
638 639
static void mb_lpf_vertical_edge_w(uint8_t *s, int p, const uint8_t *blimit,
                                   const uint8_t *limit, const uint8_t *thresh,
640
                                   int count) {
641 642
  int i;

643
  for (i = 0; i < count; ++i) {
644 645 646 647
    const uint8_t p7 = s[-8], p6 = s[-7], p5 = s[-6], p4 = s[-5], p3 = s[-4],
                  p2 = s[-3], p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3], q4 = s[4],
                  q5 = s[5], q6 = s[6], q7 = s[7];
clang-format's avatar
clang-format committed
648 649
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
650
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
651

Ola Hugosson's avatar
Ola Hugosson committed
652 653 654 655 656 657 658 659
#if PARALLEL_DEBLOCKING_13_TAP
    (void)p7;
    (void)q7;
    const int8_t flat2 = flat_mask4(1, p6, p5, p4, p0, q0, q4, q5, q6);

    filter14(mask, *thresh, flat, flat2, s - 7, s - 6, s - 5, s - 4, s - 3,
             s - 2, s - 1, s, s + 1, s + 2, s + 3, s + 4, s + 5, s + 6);
#elif PARALLEL_DEBLOCKING_11_TAP
660 661 662 663 664 665 666 667 668 669 670 671
    const int8_t flat2 = flat_mask3(1, p5, p4, p0, q0, q4, q5);

    filter12(mask, *thresh, flat, flat2, s - 6, s - 5, s - 4, s - 3, s - 2,
             s - 1, s, s + 1, s + 2, s + 3, s + 4, s + 5);
#elif PARALLEL_DEBLOCKING_9_TAP
    const int8_t flat2 = flat_mask2(1, p4, p0, q0, q4);

    filter10(mask, *thresh, flat, flat2, s - 5, s - 4, s - 3, s - 2, s - 1, s,
             s + 1, s + 2, s + 3, s + 4);

#else
    const int8_t flat2 = flat_mask5(1, p7, p6, p5, p4, p0, q0, q4, q5, q6, q7);
672

clang-format's avatar
clang-format committed
673 674 675
    filter16(mask, *thresh, flat, flat2, s - 8, s - 7, s - 6, s - 5, s - 4,
             s - 3, s - 2, s - 1, s, s + 1, s + 2, s + 3, s + 4, s + 5, s + 6,
             s + 7);
676 677
#endif

678
    s += p;
679
  }
680
}
681

Yaowu Xu's avatar
Yaowu Xu committed
682
void aom_lpf_vertical_16_c(uint8_t *s, int p, const uint8_t *blimit,
Jim Bankoski's avatar
Jim Bankoski committed
683
                           const uint8_t *limit, const uint8_t *thresh) {
684
#if CONFIG_PARALLEL_DEBLOCKING
685 686
  mb_lpf_vertical_edge_w(s, p, blimit, limit, thresh, 4);
#else
687
  mb_lpf_vertical_edge_w(s, p, blimit, limit, thresh, 8);
688
#endif
689 690
}

Yaowu Xu's avatar
Yaowu Xu committed
691
void aom_lpf_vertical_16_dual_c(uint8_t *s, int p, const uint8_t *blimit,
Jim Bankoski's avatar
Jim Bankoski committed
692
                                const uint8_t *limit, const uint8_t *thresh) {
Yi Luo's avatar
Yi Luo committed
693 694 695
#if CONFIG_PARALLEL_DEBLOCKING
  mb_lpf_vertical_edge_w(s, p, blimit, limit, thresh, 8);
#else
696
  mb_lpf_vertical_edge_w(s, p, blimit, limit, thresh, 16);
Yi Luo's avatar
Yi Luo committed
697
#endif
698
}
699

700
#if CONFIG_HIGHBITDEPTH
701 702 703 704 705 706 707 708 709 710 711 712 713 714 715
#if CONFIG_PARALLEL_DEBLOCKING
// Should we apply any filter at all: 11111111 yes, 00000000 no ?
static INLINE int8_t highbd_filter_mask2(uint8_t limit, uint8_t blimit,
                                         uint16_t p1, uint16_t p0, uint16_t q0,
                                         uint16_t q1, int bd) {
  int8_t mask = 0;
  int16_t limit16 = (uint16_t)limit << (bd - 8);
  int16_t blimit16 = (uint16_t)blimit << (bd - 8);
  mask |= (abs(p1 - p0) > limit16) * -1;
  mask |= (abs(q1 - q0) > limit16) * -1;
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit16) * -1;
  return ~mask;
}
#endif  // CONFIG_PARALLEL_DEBLOCKING

716
// Should we apply any filter at all: 11111111 yes, 00000000 no ?
717
static INLINE int8_t highbd_filter_mask(uint8_t limit, uint8_t blimit,
clang-format's avatar
clang-format committed
718 719
                                        uint16_t p3, uint16_t p2, uint16_t p1,
                                        uint16_t p0, uint16_t q0, uint16_t q1,
720
                                        uint16_t q2, uint16_t q3, int bd) {
721 722 723 724 725 726 727 728 729
  int8_t mask = 0;
  int16_t limit16 = (uint16_t)limit << (bd - 8);
  int16_t blimit16 = (uint16_t)blimit << (bd - 8);
  mask |= (abs(p3 - p2) > limit16) * -1;
  mask |= (abs(p2 - p1) > limit16) * -1;
  mask |= (abs(p1 - p0) > limit16) * -1;
  mask |= (abs(q1 - q0) > limit16) * -1;
  mask |= (abs(q2 - q1) > limit16) * -1;
  mask |= (abs(q3 - q2) > limit16) * -1;
clang-format's avatar
clang-format committed
730
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit16) * -1;
731 732 733
  return ~mask;
}

Ola Hugosson's avatar
Ola Hugosson committed
734 735 736 737 738 739 740 741 742 743 744 745 746 747 748
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE int8_t highbd_flat_mask3_chroma(uint8_t thresh, uint16_t p2,
                                              uint16_t p1, uint16_t p0,
                                              uint16_t q0, uint16_t q1,
                                              uint16_t q2, int bd) {
  int8_t mask = 0;
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  mask |= (abs(p1 - p0) > thresh16) * -1;
  mask |= (abs(q1 - q0) > thresh16) * -1;
  mask |= (abs(p2 - p0) > thresh16) * -1;
  mask |= (abs(q2 - q0) > thresh16) * -1;
  return ~mask;
}
#endif

clang-format's avatar
clang-format committed
749 750 751 752
static INLINE int8_t highbd_flat_mask4(uint8_t thresh, uint16_t p3, uint16_t p2,
                                       uint16_t p1, uint16_t p0, uint16_t q0,
                                       uint16_t q1, uint16_t q2, uint16_t q3,
                                       int bd) {
753 754 755 756 757 758 759 760 761 762 763
  int8_t mask = 0;
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  mask |= (abs(p1 - p0) > thresh16) * -1;
  mask |= (abs(q1 - q0) > thresh16) * -1;
  mask |= (abs(p2 - p0) > thresh16) * -1;
  mask |= (abs(q2 - q0) > thresh16) * -1;
  mask |= (abs(p3 - p0) > thresh16) * -1;
  mask |= (abs(q3 - q0) > thresh16) * -1;
  return ~mask;
}

764
#if !PARALLEL_DEBLOCKING_13_TAP
clang-format's avatar
clang-format committed
765 766 767
static INLINE int8_t highbd_flat_mask5(uint8_t thresh, uint16_t p4, uint16_t p3,
                                       uint16_t p2, uint16_t p1, uint16_t p0,
                                       uint16_t q0, uint16_t q1, uint16_t q2,
768 769
                                       uint16_t q3, uint16_t q4, int bd) {
  int8_t mask = ~highbd_flat_mask4(thresh, p3, p2, p1, p0, q0, q1, q2, q3, bd);
770 771 772 773 774
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  mask |= (abs(p4 - p0) > thresh16) * -1;
  mask |= (abs(q4 - q0) > thresh16) * -1;
  return ~mask;
}
775
#endif
776 777 778

// Is there high edge variance internal edge:
// 11111111_11111111 yes, 00000000_00000000 no ?
779 780
static INLINE int16_t highbd_hev_mask(uint8_t thresh, uint16_t p1, uint16_t p0,
                                      uint16_t q0, uint16_t q1, int bd) {
781 782 783 784 785 786 787
  int16_t hev = 0;
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  hev |= (abs(p1 - p0) > thresh16) * -1;
  hev |= (abs(q1 - q0) > thresh16) * -1;
  return hev;
}

788 789 790
static INLINE void highbd_filter4(int8_t mask, uint8_t thresh, uint16_t *op1,
                                  uint16_t *op0, uint16_t *oq0, uint16_t *oq1,
                                  int bd) {
791 792 793 794 795 796 797 798
  int16_t filter1, filter2;
  // ^0x80 equivalent to subtracting 0x80 from the values to turn them
  // into -128 to +127 instead of 0 to 255.
  int shift = bd - 8;
  const int16_t ps1 = (int16_t)*op1 - (0x80 << shift);
  const int16_t ps0 = (int16_t)*op0 - (0x80 << shift);
  const int16_t qs0 = (int16_t)*oq0 - (0x80 << shift);
  const int16_t qs1 = (int16_t)*oq1 - (0x80 << shift);
799
  const uint16_t hev = highbd_hev_mask(thresh, *op1, *op0, *oq0, *oq1, bd);
800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822

  // Add outer taps if we have high edge variance.
  int16_t filter = signed_char_clamp_high(ps1 - qs1, bd) & hev;

  // Inner taps.
  filter = signed_char_clamp_high(filter + 3 * (qs0 - ps0), bd) & mask;

  // Save bottom 3 bits so that we round one side +4 and the other +3
  // if it equals 4 we'll set to adjust by -1 to account for the fact
  // we'd round 3 the other way.
  filter1 = signed_char_clamp_high(filter + 4, bd) >> 3;
  filter2 = signed_char_clamp_high(filter + 3, bd) >> 3;

  *oq0 = signed_char_clamp_high(qs0 - filter1, bd) + (0x80 << shift);
  *op0 = signed_char_clamp_high(ps0 + filter2, bd) + (0x80 << shift);

  // Outer tap adjustments.
  filter = ROUND_POWER_OF_TWO(filter1, 1) & ~hev;

  *oq1 = signed_char_clamp_high(qs1 - filter, bd) + (0x80 << shift);
  *op1 = signed_char_clamp_high(ps1 + filter, bd) + (0x80 << shift);
}

Yaowu Xu's avatar
Yaowu Xu committed
823
void aom_highbd_lpf_horizontal_4_c(uint16_t *s, int p /* pitch */,
824
                                   const uint8_t *blimit, const uint8_t *limit,
825
                                   const uint8_t *thresh, int bd) {
826
  int i;
827
#if CONFIG_PARALLEL_DEBLOCKING
828 829 830 831
  int count = 4;
#else
  int count = 8;
#endif
832 833 834

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
835
  for (i = 0; i < count; ++i) {
836
#if !CONFIG_PARALLEL_DEBLOCKING
837 838 839 840 841 842 843 844
    const uint16_t p3 = s[-4 * p];
    const uint16_t p2 = s[-3 * p];
    const uint16_t p1 = s[-2 * p];
    const uint16_t p0 = s[-p];
    const uint16_t q0 = s[0 * p];
    const uint16_t q1 = s[1 * p];
    const uint16_t q2 = s[2 * p];
    const uint16_t q3 = s[3 * p];
clang-format's avatar
clang-format committed
845 846
    const int8_t mask =
        highbd_filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3, bd);
847 848 849 850 851 852 853 854
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint16_t p1 = s[-2 * p];
    const uint16_t p0 = s[-p];
    const uint16_t q0 = s[0 * p];
    const uint16_t q1 = s[1 * p];
    const int8_t mask =
        highbd_filter_mask2(*limit, *blimit, p1, p0, q0, q1, bd);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
855
    highbd_filter4(mask, *thresh, s - 2 * p, s - 1 * p, s, s + 1 * p, bd);
856 857 858 859
    ++s;
  }
}

Yaowu Xu's avatar
Yaowu Xu committed
860
void aom_highbd_lpf_horizontal_4_dual_c(
clang-format's avatar
clang-format committed
861 862 863
    uint16_t *s, int p, const uint8_t *blimit0, const uint8_t *limit0,
    const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
    const uint8_t *thresh1, int bd) {
Yaowu Xu's avatar
Yaowu Xu committed
864
  aom_highbd_lpf_horizontal_4_c(s, p, blimit0, limit0, thresh0, bd);
Yi Luo's avatar
Yi Luo committed
865 866 867
#if CONFIG_PARALLEL_DEBLOCKING
  aom_highbd_lpf_horizontal_4_c(s + 4, p, blimit1, limit1, thresh1, bd);
#else
Yaowu Xu's avatar
Yaowu Xu committed
868
  aom_highbd_lpf_horizontal_4_c(s + 8, p, blimit1, limit1, thresh1, bd);
Yi Luo's avatar
Yi Luo committed
869
#endif
870 871
}

Yaowu Xu's avatar
Yaowu Xu committed
872
void aom_highbd_lpf_vertical_4_c(uint16_t *s, int pitch, const uint8_t *blimit,
873
                                 const uint8_t *limit, const uint8_t *thresh,
874
                                 int bd) {
875
  int i;
876
#if CONFIG_PARALLEL_DEBLOCKING
877 878 879 880
  int count = 4;
#else
  int count = 8;
#endif
881 882 883

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
884
  for (i = 0; i < count; ++i) {
885
#if !CONFIG_PARALLEL_DEBLOCKING
886
    const uint16_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
clang-format's avatar
clang-format committed
887 888 889
    const uint16_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
    const int8_t mask =
        highbd_filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3, bd);
890 891 892 893 894 895
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint16_t p1 = s[-2], p0 = s[-1];
    const uint16_t q0 = s[0], q1 = s[1];
    const int8_t mask =
        highbd_filter_mask2(*limit, *blimit, p1, p0, q0, q1, bd);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
896
    highbd_filter4(mask, *thresh, s - 2, s - 1, s, s + 1, bd);
897 898 899 900
    s += pitch;
  }
}

Yaowu Xu's avatar
Yaowu Xu committed
901
void aom_highbd_lpf_vertical_4_dual_c(
clang-format's avatar
clang-format committed
902 903 904
    uint16_t *s, int pitch, const uint8_t *blimit0, const uint8_t *limit0,
    const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
    const uint8_t *thresh1, int bd) {
Yaowu Xu's avatar
Yaowu Xu committed
905
  aom_highbd_lpf_vertical_4_c(s, pitch, blimit0, limit0, thresh0, bd);
Yi Luo's avatar
Yi Luo committed
906 907 908 909
#if CONFIG_PARALLEL_DEBLOCKING
  aom_highbd_lpf_vertical_4_c(s + 4 * pitch, pitch, blimit1, limit1, thresh1,
                              bd);
#else
Yaowu Xu's avatar
Yaowu Xu committed
910
  aom_highbd_lpf_vertical_4_c(s + 8 * pitch, pitch, blimit1, limit1, thresh1,
clang-format's avatar
clang-format committed
911
                              bd);
Yi Luo's avatar
Yi Luo committed
912
#endif
913 914
}

Ola Hugosson's avatar
Ola Hugosson committed
915 916 917 918 919 920 921 922 923 924 925 926 927 928 929 930 931 932 933 934
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE void highbd_filter6(int8_t mask, uint8_t thresh, int8_t flat,
                                  uint16_t *op2, uint16_t *op1, uint16_t *op0,
                                  uint16_t *oq0, uint16_t *oq1, uint16_t *oq2,
                                  int bd) {
  if (flat && mask) {
    const uint16_t p2 = *op2, p1 = *op1, p0 = *op0;
    const uint16_t q0 = *oq0, q1 = *oq1, q2 = *oq2;

    // 5-tap filter [1, 2, 2, 2, 1]
    *op1 = ROUND_POWER_OF_TWO(p2 * 3 + p1 * 2 + p0 * 2 + q0, 3);
    *op0 = ROUND_POWER_OF_TWO(p2 + p1 * 2 + p0 * 2 + q0 * 2 + q1, 3);
    *oq0 = ROUND_POWER_OF_TWO(p1 + p0 * 2 + q0 * 2 + q1 * 2 + q2, 3);
    *oq1 = ROUND_POWER_OF_TWO(p0 + q0 * 2 + q1 * 2 + q2 * 3, 3);
  } else {
    highbd_filter4(mask, thresh, op1, op0, oq0, oq1, bd);
  }
}
#endif

935
static INLINE void highbd_filter8(int8_t mask, uint8_t thresh, int8_t flat,
clang-format's avatar
clang-format committed
936 937
                                  uint16_t *op3, uint16_t *op2, uint16_t *op1,
                                  uint16_t *op0, uint16_t *oq0, uint16_t *oq1,
938
                                  uint16_t *oq2, uint16_t *oq3, int bd) {
939 940 941 942 943 944 945 946 947 948 949 950
  if (flat && mask) {
    const uint16_t p3 = *op3, p2 = *op2, p1 = *op1, p0 = *op0;
    const uint16_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3;

    // 7-tap filter [1, 1, 1, 2, 1, 1, 1]
    *op2 = ROUND_POWER_OF_TWO(p3 + p3 + p3 + 2 * p2 + p1 + p0 + q0, 3);
    *op1 = ROUND_POWER_OF_TWO(p3 + p3 + p2 + 2 * p1 + p0 + q0 + q1, 3);
    *op0 = ROUND_POWER_OF_TWO(p3 + p2 + p1 + 2 * p0 + q0 + q1 + q2, 3);
    *oq0 = ROUND_POWER_OF_TWO(p2 + p1 + p0 + 2 * q0 + q1 + q2 + q3, 3);
    *oq1 = ROUND_POWER_OF_TWO(p1 + p0 + q0 + 2 * q1 + q2 + q3 + q3, 3);
    *oq2 = ROUND_POWER_OF_TWO(p0 + q0 + q1 + 2 * q2 + q3 + q3 + q3, 3);
  } else {
clang-format's avatar
clang-format committed
951
    highbd_filter4(mask, thresh, op1, op0, oq0, oq1, bd);
952 953 954
  }
}

Yaowu Xu's avatar
Yaowu Xu committed
955
void aom_highbd_lpf_horizontal_8_c(uint16_t *s, int p, const uint8_t *blimit,
956
                                   const uint8_t *limit, const uint8_t *thresh,
957
                                   int bd) {