loopfilter.c 51.2 KB
Newer Older
John Koleszar's avatar
John Koleszar committed
1
/*
Yaowu Xu's avatar
Yaowu Xu committed
2
 * Copyright (c) 2016, Alliance for Open Media. All rights reserved
John Koleszar's avatar
John Koleszar committed
3
 *
Yaowu Xu's avatar
Yaowu Xu committed
4 5 6 7 8 9
 * This source code is subject to the terms of the BSD 2 Clause License and
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
 * was not distributed with this source code in the LICENSE file, you can
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
 * Media Patent License 1.0 was not distributed with this source code in the
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
John Koleszar's avatar
John Koleszar committed
10
 */
11

Zoe Liu's avatar
Zoe Liu committed
12 13
#include <stdlib.h>

Yaowu Xu's avatar
Yaowu Xu committed
14 15 16
#include "./aom_config.h"
#include "./aom_dsp_rtcd.h"
#include "aom_dsp/aom_dsp_common.h"
17
#include "aom_ports/mem.h"
John Koleszar's avatar
John Koleszar committed
18

19
static INLINE int8_t signed_char_clamp(int t) {
20
  return (int8_t)clamp(t, -128, 127);
John Koleszar's avatar
John Koleszar committed
21 22
}

23 24 25
#define PARALLEL_DEBLOCKING_11_TAP 0
#define PARALLEL_DEBLOCKING_9_TAP 0

Ola Hugosson's avatar
Ola Hugosson committed
26 27 28 29 30 31 32 33
#if CONFIG_DEBLOCK_13TAP
#define PARALLEL_DEBLOCKING_13_TAP 1
#define PARALLEL_DEBLOCKING_5_TAP_CHROMA 1
#else
#define PARALLEL_DEBLOCKING_13_TAP 0
#define PARALLEL_DEBLOCKING_5_TAP_CHROMA 0
#endif

34
#if CONFIG_HIGHBITDEPTH
35 36
static INLINE int16_t signed_char_clamp_high(int t, int bd) {
  switch (bd) {
clang-format's avatar
clang-format committed
37 38
    case 10: return (int16_t)clamp(t, -128 * 4, 128 * 4 - 1);
    case 12: return (int16_t)clamp(t, -128 * 16, 128 * 16 - 1);
39
    case 8:
clang-format's avatar
clang-format committed
40
    default: return (int16_t)clamp(t, -128, 128 - 1);
41 42 43
  }
}
#endif
44
#if CONFIG_PARALLEL_DEBLOCKING
Dmitry Kovalev's avatar
Dmitry Kovalev committed
45
// should we apply any filter at all: 11111111 yes, 00000000 no
46 47 48 49 50 51 52 53 54
static INLINE int8_t filter_mask2(uint8_t limit, uint8_t blimit, uint8_t p1,
                                  uint8_t p0, uint8_t q0, uint8_t q1) {
  int8_t mask = 0;
  mask |= (abs(p1 - p0) > limit) * -1;
  mask |= (abs(q1 - q0) > limit) * -1;
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit) * -1;
  return ~mask;
}
#endif  // CONFIG_PARALLEL_DEBLOCKING
clang-format's avatar
clang-format committed
55 56 57
static INLINE int8_t filter_mask(uint8_t limit, uint8_t blimit, uint8_t p3,
                                 uint8_t p2, uint8_t p1, uint8_t p0, uint8_t q0,
                                 uint8_t q1, uint8_t q2, uint8_t q3) {
58
  int8_t mask = 0;
John Koleszar's avatar
John Koleszar committed
59 60 61 62 63 64
  mask |= (abs(p3 - p2) > limit) * -1;
  mask |= (abs(p2 - p1) > limit) * -1;
  mask |= (abs(p1 - p0) > limit) * -1;
  mask |= (abs(q1 - q0) > limit) * -1;
  mask |= (abs(q2 - q1) > limit) * -1;
  mask |= (abs(q3 - q2) > limit) * -1;
clang-format's avatar
clang-format committed
65
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit) * -1;
Dmitry Kovalev's avatar
Dmitry Kovalev committed
66
  return ~mask;
John Koleszar's avatar
John Koleszar committed
67 68
}

Ola Hugosson's avatar
Ola Hugosson committed
69 70 71 72 73 74 75 76 77 78 79 80 81
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE int8_t flat_mask3_chroma(uint8_t thresh, uint8_t p2, uint8_t p1,
                                       uint8_t p0, uint8_t q0, uint8_t q1,
                                       uint8_t q2) {
  int8_t mask = 0;
  mask |= (abs(p1 - p0) > thresh) * -1;
  mask |= (abs(q1 - q0) > thresh) * -1;
  mask |= (abs(p2 - p0) > thresh) * -1;
  mask |= (abs(q2 - q0) > thresh) * -1;
  return ~mask;
}
#endif

clang-format's avatar
clang-format committed
82 83
static INLINE int8_t flat_mask4(uint8_t thresh, uint8_t p3, uint8_t p2,
                                uint8_t p1, uint8_t p0, uint8_t q0, uint8_t q1,
84
                                uint8_t q2, uint8_t q3) {
Dmitry Kovalev's avatar
Dmitry Kovalev committed
85 86 87 88 89 90 91 92
  int8_t mask = 0;
  mask |= (abs(p1 - p0) > thresh) * -1;
  mask |= (abs(q1 - q0) > thresh) * -1;
  mask |= (abs(p2 - p0) > thresh) * -1;
  mask |= (abs(q2 - q0) > thresh) * -1;
  mask |= (abs(p3 - p0) > thresh) * -1;
  mask |= (abs(q3 - q0) > thresh) * -1;
  return ~mask;
93 94
}

95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117
#if PARALLEL_DEBLOCKING_9_TAP
static INLINE int8_t flat_mask2(uint8_t thresh, uint8_t p4, uint8_t p0,
                                uint8_t q0, uint8_t q4) {
  int8_t mask = 0;
  mask |= (abs(p4 - p0) > thresh) * -1;
  mask |= (abs(q4 - q0) > thresh) * -1;
  return ~mask;
}
#endif

#if PARALLEL_DEBLOCKING_11_TAP
static INLINE int8_t flat_mask3(uint8_t thresh, uint8_t p5, uint8_t p4,
                                uint8_t p0, uint8_t q0, uint8_t q4,
                                uint8_t q5) {
  int8_t mask = 0;
  mask |= (abs(p4 - p0) > thresh) * -1;
  mask |= (abs(q4 - q0) > thresh) * -1;
  mask |= (abs(p5 - p0) > thresh) * -1;
  mask |= (abs(q5 - q0) > thresh) * -1;
  return ~mask;
}
#endif

clang-format's avatar
clang-format committed
118 119 120 121
static INLINE int8_t flat_mask5(uint8_t thresh, uint8_t p4, uint8_t p3,
                                uint8_t p2, uint8_t p1, uint8_t p0, uint8_t q0,
                                uint8_t q1, uint8_t q2, uint8_t q3,
                                uint8_t q4) {
Dmitry Kovalev's avatar
Dmitry Kovalev committed
122 123 124 125
  int8_t mask = ~flat_mask4(thresh, p3, p2, p1, p0, q0, q1, q2, q3);
  mask |= (abs(p4 - p0) > thresh) * -1;
  mask |= (abs(q4 - q0) > thresh) * -1;
  return ~mask;
126 127
}

128
// is there high edge variance internal edge: 11111111 yes, 00000000 no
129 130
static INLINE int8_t hev_mask(uint8_t thresh, uint8_t p1, uint8_t p0,
                              uint8_t q0, uint8_t q1) {
131
  int8_t hev = 0;
clang-format's avatar
clang-format committed
132 133
  hev |= (abs(p1 - p0) > thresh) * -1;
  hev |= (abs(q1 - q0) > thresh) * -1;
John Koleszar's avatar
John Koleszar committed
134
  return hev;
John Koleszar's avatar
John Koleszar committed
135 136
}

137
static INLINE void filter4(int8_t mask, uint8_t thresh, uint8_t *op1,
138
                           uint8_t *op0, uint8_t *oq0, uint8_t *oq1) {
Dmitry Kovalev's avatar
Dmitry Kovalev committed
139
  int8_t filter1, filter2;
John Koleszar's avatar
John Koleszar committed
140

clang-format's avatar
clang-format committed
141 142 143 144
  const int8_t ps1 = (int8_t)*op1 ^ 0x80;
  const int8_t ps0 = (int8_t)*op0 ^ 0x80;
  const int8_t qs0 = (int8_t)*oq0 ^ 0x80;
  const int8_t qs1 = (int8_t)*oq1 ^ 0x80;
145
  const uint8_t hev = hev_mask(thresh, *op1, *op0, *oq0, *oq1);
Dmitry Kovalev's avatar
Dmitry Kovalev committed
146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162

  // add outer taps if we have high edge variance
  int8_t filter = signed_char_clamp(ps1 - qs1) & hev;

  // inner taps
  filter = signed_char_clamp(filter + 3 * (qs0 - ps0)) & mask;

  // save bottom 3 bits so that we round one side +4 and the other +3
  // if it equals 4 we'll set to adjust by -1 to account for the fact
  // we'd round 3 the other way
  filter1 = signed_char_clamp(filter + 4) >> 3;
  filter2 = signed_char_clamp(filter + 3) >> 3;

  *oq0 = signed_char_clamp(qs0 - filter1) ^ 0x80;
  *op0 = signed_char_clamp(ps0 + filter2) ^ 0x80;

  // outer tap adjustments
Dmitry Kovalev's avatar
Dmitry Kovalev committed
163
  filter = ROUND_POWER_OF_TWO(filter1, 1) & ~hev;
John Koleszar's avatar
John Koleszar committed
164

Dmitry Kovalev's avatar
Dmitry Kovalev committed
165 166
  *oq1 = signed_char_clamp(qs1 - filter) ^ 0x80;
  *op1 = signed_char_clamp(ps1 + filter) ^ 0x80;
John Koleszar's avatar
John Koleszar committed
167
}
168

Yaowu Xu's avatar
Yaowu Xu committed
169
void aom_lpf_horizontal_4_c(uint8_t *s, int p /* pitch */,
Jim Bankoski's avatar
Jim Bankoski committed
170
                            const uint8_t *blimit, const uint8_t *limit,
171
                            const uint8_t *thresh) {
172
  int i;
173
#if CONFIG_PARALLEL_DEBLOCKING
174 175 176 177
  int count = 4;
#else
  int count = 8;
#endif
John Koleszar's avatar
John Koleszar committed
178

Dmitry Kovalev's avatar
Dmitry Kovalev committed
179 180
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
181
  for (i = 0; i < count; ++i) {
182
#if !CONFIG_PARALLEL_DEBLOCKING
183
    const uint8_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
clang-format's avatar
clang-format committed
184 185 186
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
187 188 189 190 191
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint8_t p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p];
    const int8_t mask = filter_mask2(*limit, *blimit, p1, p0, q0, q1);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
192
    filter4(mask, *thresh, s - 2 * p, s - 1 * p, s, s + 1 * p);
John Koleszar's avatar
John Koleszar committed
193
    ++s;
194
  }
John Koleszar's avatar
John Koleszar committed
195 196
}

Yaowu Xu's avatar
Yaowu Xu committed
197
void aom_lpf_horizontal_4_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
198 199 200
                                 const uint8_t *limit0, const uint8_t *thresh0,
                                 const uint8_t *blimit1, const uint8_t *limit1,
                                 const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
201 202
  aom_lpf_horizontal_4_c(s, p, blimit0, limit0, thresh0);
  aom_lpf_horizontal_4_c(s + 8, p, blimit1, limit1, thresh1);
203 204
}

Yaowu Xu's avatar
Yaowu Xu committed
205
void aom_lpf_vertical_4_c(uint8_t *s, int pitch, const uint8_t *blimit,
206
                          const uint8_t *limit, const uint8_t *thresh) {
207
  int i;
208
#if CONFIG_PARALLEL_DEBLOCKING
209 210 211 212
  int count = 4;
#else
  int count = 8;
#endif
John Koleszar's avatar
John Koleszar committed
213

Dmitry Kovalev's avatar
Dmitry Kovalev committed
214 215
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
216
  for (i = 0; i < count; ++i) {
217
#if !CONFIG_PARALLEL_DEBLOCKING
218
    const uint8_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
clang-format's avatar
clang-format committed
219 220 221
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
222 223 224 225 226
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint8_t p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1];
    const int8_t mask = filter_mask2(*limit, *blimit, p1, p0, q0, q1);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
227
    filter4(mask, *thresh, s - 2, s - 1, s, s + 1);
Dmitry Kovalev's avatar
Dmitry Kovalev committed
228
    s += pitch;
229
  }
John Koleszar's avatar
John Koleszar committed
230
}
231

Yaowu Xu's avatar
Yaowu Xu committed
232
void aom_lpf_vertical_4_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
233 234 235
                               const uint8_t *limit0, const uint8_t *thresh0,
                               const uint8_t *blimit1, const uint8_t *limit1,
                               const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
236 237
  aom_lpf_vertical_4_c(s, pitch, blimit0, limit0, thresh0);
  aom_lpf_vertical_4_c(s + 8 * pitch, pitch, blimit1, limit1, thresh1);
238 239
}

Ola Hugosson's avatar
Ola Hugosson committed
240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE void filter6(int8_t mask, uint8_t thresh, int8_t flat,
                           uint8_t *op2, uint8_t *op1, uint8_t *op0,
                           uint8_t *oq0, uint8_t *oq1, uint8_t *oq2) {
  if (flat && mask) {
    const uint8_t p2 = *op2, p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2;

    // 5-tap filter [1, 2, 2, 2, 1]
    *op1 = ROUND_POWER_OF_TWO(p2 * 3 + p1 * 2 + p0 * 2 + q0, 3);
    *op0 = ROUND_POWER_OF_TWO(p2 + p1 * 2 + p0 * 2 + q0 * 2 + q1, 3);
    *oq0 = ROUND_POWER_OF_TWO(p1 + p0 * 2 + q0 * 2 + q1 * 2 + q2, 3);
    *oq1 = ROUND_POWER_OF_TWO(p0 + q0 * 2 + q1 * 2 + q2 * 3, 3);
  } else {
    filter4(mask, thresh, op1, op0, oq0, oq1);
  }
}
#endif

259
static INLINE void filter8(int8_t mask, uint8_t thresh, int8_t flat,
clang-format's avatar
clang-format committed
260 261
                           uint8_t *op3, uint8_t *op2, uint8_t *op1,
                           uint8_t *op0, uint8_t *oq0, uint8_t *oq1,
262
                           uint8_t *oq2, uint8_t *oq3) {
John Koleszar's avatar
John Koleszar committed
263
  if (flat && mask) {
264 265
    const uint8_t p3 = *op3, p2 = *op2, p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3;
John Koleszar's avatar
John Koleszar committed
266

Dmitry Kovalev's avatar
Dmitry Kovalev committed
267 268 269 270 271 272 273
    // 7-tap filter [1, 1, 1, 2, 1, 1, 1]
    *op2 = ROUND_POWER_OF_TWO(p3 + p3 + p3 + 2 * p2 + p1 + p0 + q0, 3);
    *op1 = ROUND_POWER_OF_TWO(p3 + p3 + p2 + 2 * p1 + p0 + q0 + q1, 3);
    *op0 = ROUND_POWER_OF_TWO(p3 + p2 + p1 + 2 * p0 + q0 + q1 + q2, 3);
    *oq0 = ROUND_POWER_OF_TWO(p2 + p1 + p0 + 2 * q0 + q1 + q2 + q3, 3);
    *oq1 = ROUND_POWER_OF_TWO(p1 + p0 + q0 + 2 * q1 + q2 + q3 + q3, 3);
    *oq2 = ROUND_POWER_OF_TWO(p0 + q0 + q1 + 2 * q2 + q3 + q3 + q3, 3);
John Koleszar's avatar
John Koleszar committed
274
  } else {
clang-format's avatar
clang-format committed
275
    filter4(mask, thresh, op1, op0, oq0, oq1);
John Koleszar's avatar
John Koleszar committed
276
  }
277
}
278

Ola Hugosson's avatar
Ola Hugosson committed
279 280 281 282
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
void aom_lpf_horizontal_6_c(uint8_t *s, int p, const uint8_t *blimit,
                            const uint8_t *limit, const uint8_t *thresh) {
  int i;
283
#if CONFIG_PARALLEL_DEBLOCKING
Ola Hugosson's avatar
Ola Hugosson committed
284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304
  int count = 4;
#else
  int count = 8;
#endif

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
  for (i = 0; i < count; ++i) {
    const uint8_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];

    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
    const int8_t flat = flat_mask3_chroma(1, p2, p1, p0, q0, q1, q2);
    filter6(mask, *thresh, flat, s - 3 * p, s - 2 * p, s - 1 * p, s, s + 1 * p,
            s + 2 * p);
    ++s;
  }
}
#endif

Yaowu Xu's avatar
Yaowu Xu committed
305
void aom_lpf_horizontal_8_c(uint8_t *s, int p, const uint8_t *blimit,
306
                            const uint8_t *limit, const uint8_t *thresh) {
307
  int i;
308
#if CONFIG_PARALLEL_DEBLOCKING
309 310 311 312
  int count = 4;
#else
  int count = 8;
#endif
John Koleszar's avatar
John Koleszar committed
313

Dmitry Kovalev's avatar
Dmitry Kovalev committed
314 315
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
316
  for (i = 0; i < count; ++i) {
317 318 319
    const uint8_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];

clang-format's avatar
clang-format committed
320 321
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
322
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
clang-format's avatar
clang-format committed
323 324
    filter8(mask, *thresh, flat, s - 4 * p, s - 3 * p, s - 2 * p, s - 1 * p, s,
            s + 1 * p, s + 2 * p, s + 3 * p);
John Koleszar's avatar
John Koleszar committed
325
    ++s;
326
  }
John Koleszar's avatar
John Koleszar committed
327
}
328

Yaowu Xu's avatar
Yaowu Xu committed
329
void aom_lpf_horizontal_8_dual_c(uint8_t *s, int p, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
330 331 332
                                 const uint8_t *limit0, const uint8_t *thresh0,
                                 const uint8_t *blimit1, const uint8_t *limit1,
                                 const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
333 334
  aom_lpf_horizontal_8_c(s, p, blimit0, limit0, thresh0);
  aom_lpf_horizontal_8_c(s + 8, p, blimit1, limit1, thresh1);
335 336
}

Ola Hugosson's avatar
Ola Hugosson committed
337 338 339 340
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
void aom_lpf_vertical_6_c(uint8_t *s, int pitch, const uint8_t *blimit,
                          const uint8_t *limit, const uint8_t *thresh) {
  int i;
341
#if CONFIG_PARALLEL_DEBLOCKING
Ola Hugosson's avatar
Ola Hugosson committed
342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358
  int count = 4;
#else
  int count = 8;
#endif

  for (i = 0; i < count; ++i) {
    const uint8_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
    const int8_t flat = flat_mask3_chroma(1, p2, p1, p0, q0, q1, q2);
    filter6(mask, *thresh, flat, s - 3, s - 2, s - 1, s, s + 1, s + 2);
    s += pitch;
  }
}
#endif

Yaowu Xu's avatar
Yaowu Xu committed
359
void aom_lpf_vertical_8_c(uint8_t *s, int pitch, const uint8_t *blimit,
360
                          const uint8_t *limit, const uint8_t *thresh) {
361
  int i;
362
#if CONFIG_PARALLEL_DEBLOCKING
363 364 365 366
  int count = 4;
#else
  int count = 8;
#endif
367

368
  for (i = 0; i < count; ++i) {
369 370
    const uint8_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
clang-format's avatar
clang-format committed
371 372
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
373
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
clang-format's avatar
clang-format committed
374 375
    filter8(mask, *thresh, flat, s - 4, s - 3, s - 2, s - 1, s, s + 1, s + 2,
            s + 3);
Dmitry Kovalev's avatar
Dmitry Kovalev committed
376
    s += pitch;
377
  }
John Koleszar's avatar
John Koleszar committed
378 379
}

Yaowu Xu's avatar
Yaowu Xu committed
380
void aom_lpf_vertical_8_dual_c(uint8_t *s, int pitch, const uint8_t *blimit0,
Jim Bankoski's avatar
Jim Bankoski committed
381 382 383
                               const uint8_t *limit0, const uint8_t *thresh0,
                               const uint8_t *blimit1, const uint8_t *limit1,
                               const uint8_t *thresh1) {
Yaowu Xu's avatar
Yaowu Xu committed
384 385
  aom_lpf_vertical_8_c(s, pitch, blimit0, limit0, thresh0);
  aom_lpf_vertical_8_c(s + 8 * pitch, pitch, blimit1, limit1, thresh1);
386 387
}

Ola Hugosson's avatar
Ola Hugosson committed
388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437
#if PARALLEL_DEBLOCKING_13_TAP
static INLINE void filter14(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op6, uint8_t *op5,
                            uint8_t *op4, uint8_t *op3, uint8_t *op2,
                            uint8_t *op1, uint8_t *op0, uint8_t *oq0,
                            uint8_t *oq1, uint8_t *oq2, uint8_t *oq3,
                            uint8_t *oq4, uint8_t *oq5, uint8_t *oq6) {
  if (flat2 && flat && mask) {
    const uint8_t p6 = *op6, p5 = *op5, p4 = *op4, p3 = *op3, p2 = *op2,
                  p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4,
                  q5 = *oq5, q6 = *oq6;

    // 13-tap filter [1, 1, 1, 1, 1, 2, 2, 2, 1, 1, 1, 1, 1]
    *op5 = ROUND_POWER_OF_TWO(p6 * 7 + p5 * 2 + p4 * 2 + p3 + p2 + p1 + p0 + q0,
                              4);
    *op4 = ROUND_POWER_OF_TWO(
        p6 * 5 + p5 * 2 + p4 * 2 + p3 * 2 + p2 + p1 + p0 + q0 + q1, 4);
    *op3 = ROUND_POWER_OF_TWO(
        p6 * 4 + p5 + p4 * 2 + p3 * 2 + p2 * 2 + p1 + p0 + q0 + q1 + q2, 4);
    *op2 = ROUND_POWER_OF_TWO(
        p6 * 3 + p5 + p4 + p3 * 2 + p2 * 2 + p1 * 2 + p0 + q0 + q1 + q2 + q3,
        4);
    *op1 = ROUND_POWER_OF_TWO(p6 * 2 + p5 + p4 + p3 + p2 * 2 + p1 * 2 + p0 * 2 +
                                  q0 + q1 + q2 + q3 + q4,
                              4);
    *op0 = ROUND_POWER_OF_TWO(p6 + p5 + p4 + p3 + p2 + p1 * 2 + p0 * 2 +
                                  q0 * 2 + q1 + q2 + q3 + q4 + q5,
                              4);
    *oq0 = ROUND_POWER_OF_TWO(p5 + p4 + p3 + p2 + p1 + p0 * 2 + q0 * 2 +
                                  q1 * 2 + q2 + q3 + q4 + q5 + q6,
                              4);
    *oq1 = ROUND_POWER_OF_TWO(p4 + p3 + p2 + p1 + p0 + q0 * 2 + q1 * 2 +
                                  q2 * 2 + q3 + q4 + q5 + q6 * 2,
                              4);
    *oq2 = ROUND_POWER_OF_TWO(
        p3 + p2 + p1 + p0 + q0 + q1 * 2 + q2 * 2 + q3 * 2 + q4 + q5 + q6 * 3,
        4);
    *oq3 = ROUND_POWER_OF_TWO(
        p2 + p1 + p0 + q0 + q1 + q2 * 2 + q3 * 2 + q4 * 2 + q5 + q6 * 4, 4);
    *oq4 = ROUND_POWER_OF_TWO(
        p1 + p0 + q0 + q1 + q2 + q3 * 2 + q4 * 2 + q5 * 2 + q6 * 5, 4);
    *oq5 = ROUND_POWER_OF_TWO(p0 + q0 + q1 + q2 + q3 + q4 * 2 + q5 * 2 + q6 * 7,
                              4);
  } else {
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
  }
}
#endif

438
#if PARALLEL_DEBLOCKING_11_TAP
439 440
static INLINE void filter12(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op5, uint8_t *op4,
441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468
                            uint8_t *op3, uint8_t *op2, uint8_t *op1,
                            uint8_t *op0, uint8_t *oq0, uint8_t *oq1,
                            uint8_t *oq2, uint8_t *oq3, uint8_t *oq4,
                            uint8_t *oq5) {
  if (flat2 && flat && mask) {
    const uint8_t p5 = *op5, p4 = *op4, p3 = *op3, p2 = *op2, p1 = *op1,
                  p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4,
                  q5 = *oq5;

    // 11-tap filter [1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1]
    *op4 = (p5 * 5 + p4 * 2 + p3 + p2 + p1 + p0 + q0 + 6) / 12;
    *op3 = (p5 * 4 + p4 + p3 * 2 + p2 + p1 + p0 + q0 + q1 + 6) / 12;
    *op2 = (p5 * 3 + p4 + p3 + p2 * 2 + p1 + p0 + q0 + q1 + q2 + 6) / 12;
    *op1 = (p5 * 2 + p4 + p3 + p2 + p1 * 2 + p0 + q0 + q1 + q2 + q3 + 6) / 12;
    *op0 = (p5 + p4 + p3 + p2 + p1 + p0 * 2 + q0 + q1 + q2 + q3 + q4 + 6) / 12;
    *oq0 = (p4 + p3 + p2 + p1 + p0 + q0 * 2 + q1 + q2 + q3 + q4 + q5 + 6) / 12;
    *oq1 = (p3 + p2 + p1 + p0 + q0 + q1 * 2 + q2 + q3 + q4 + q5 * 2 + 6) / 12;
    *oq2 = (p2 + p1 + p0 + q0 + q1 + q2 * 2 + q3 + q4 + q5 * 3 + 6) / 12;
    *oq3 = (p1 + p0 + q0 + q1 + q2 + q3 * 2 + q4 + q5 * 4 + 6) / 12;
    *oq4 = (p0 + q0 + q1 + q2 + q3 + q4 * 2 + q5 * 5 + 6) / 12;
  } else {
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
  }
}
#endif

#if PARALLEL_DEBLOCKING_9_TAP
469 470
static INLINE void filter10(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op4, uint8_t *op3,
471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492
                            uint8_t *op2, uint8_t *op1, uint8_t *op0,
                            uint8_t *oq0, uint8_t *oq1, uint8_t *oq2,
                            uint8_t *oq3, uint8_t *oq4) {
  if (flat2 && flat && mask) {
    const uint8_t p4 = *op4, p3 = *op3, p2 = *op2, p1 = *op1, p0 = *op0;
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4;

    // 9-tap filter [1, 1, 1, 1, 2, 1, 1, 1, 1]
    *op3 = (p4 * 4 + p3 * 2 + p2 + p1 + p0 + q0 + 5) / 10;
    *op2 = (p4 * 3 + p3 + p2 * 2 + p1 + p0 + q0 + q1 + 5) / 10;
    *op1 = (p4 * 2 + p3 + p2 + p1 * 2 + p0 + q0 + q1 + q2 + 5) / 10;
    *op0 = (p4 + p3 + p2 + p1 + p0 * 2 + q0 + q1 + q2 + q3 + 5) / 10;
    *oq0 = (p3 + p2 + p1 + p0 + q0 * 2 + q1 + q2 + q3 + q4 + 5) / 10;
    *oq1 = (p2 + p1 + p0 + q0 + q1 * 2 + q2 + q3 + q4 * 2 + 5) / 10;
    *oq2 = (p1 + p0 + q0 + q1 + q2 * 2 + q3 + q4 * 3 + 5) / 10;
    *oq3 = (p0 + q0 + q1 + q2 + q3 * 2 + q4 * 4 + 5) / 10;
  } else {
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
  }
}
#endif

493 494
static INLINE void filter16(int8_t mask, uint8_t thresh, int8_t flat,
                            int8_t flat2, uint8_t *op7, uint8_t *op6,
clang-format's avatar
clang-format committed
495 496 497 498
                            uint8_t *op5, uint8_t *op4, uint8_t *op3,
                            uint8_t *op2, uint8_t *op1, uint8_t *op0,
                            uint8_t *oq0, uint8_t *oq1, uint8_t *oq2,
                            uint8_t *oq3, uint8_t *oq4, uint8_t *oq5,
499
                            uint8_t *oq6, uint8_t *oq7) {
500
  if (flat2 && flat && mask) {
clang-format's avatar
clang-format committed
501 502
    const uint8_t p7 = *op7, p6 = *op6, p5 = *op5, p4 = *op4, p3 = *op3,
                  p2 = *op2, p1 = *op1, p0 = *op0;
503

clang-format's avatar
clang-format committed
504 505
    const uint8_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3, q4 = *oq4,
                  q5 = *oq5, q6 = *oq6, q7 = *oq7;
506

Dmitry Kovalev's avatar
Dmitry Kovalev committed
507
    // 15-tap filter [1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1]
clang-format's avatar
clang-format committed
508 509 510 511 512 513 514 515 516 517 518
    *op6 = ROUND_POWER_OF_TWO(
        p7 * 7 + p6 * 2 + p5 + p4 + p3 + p2 + p1 + p0 + q0, 4);
    *op5 = ROUND_POWER_OF_TWO(
        p7 * 6 + p6 + p5 * 2 + p4 + p3 + p2 + p1 + p0 + q0 + q1, 4);
    *op4 = ROUND_POWER_OF_TWO(
        p7 * 5 + p6 + p5 + p4 * 2 + p3 + p2 + p1 + p0 + q0 + q1 + q2, 4);
    *op3 = ROUND_POWER_OF_TWO(
        p7 * 4 + p6 + p5 + p4 + p3 * 2 + p2 + p1 + p0 + q0 + q1 + q2 + q3, 4);
    *op2 = ROUND_POWER_OF_TWO(
        p7 * 3 + p6 + p5 + p4 + p3 + p2 * 2 + p1 + p0 + q0 + q1 + q2 + q3 + q4,
        4);
519
    *op1 = ROUND_POWER_OF_TWO(p7 * 2 + p6 + p5 + p4 + p3 + p2 + p1 * 2 + p0 +
clang-format's avatar
clang-format committed
520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541
                                  q0 + q1 + q2 + q3 + q4 + q5,
                              4);
    *op0 = ROUND_POWER_OF_TWO(p7 + p6 + p5 + p4 + p3 + p2 + p1 + p0 * 2 + q0 +
                                  q1 + q2 + q3 + q4 + q5 + q6,
                              4);
    *oq0 = ROUND_POWER_OF_TWO(p6 + p5 + p4 + p3 + p2 + p1 + p0 + q0 * 2 + q1 +
                                  q2 + q3 + q4 + q5 + q6 + q7,
                              4);
    *oq1 = ROUND_POWER_OF_TWO(p5 + p4 + p3 + p2 + p1 + p0 + q0 + q1 * 2 + q2 +
                                  q3 + q4 + q5 + q6 + q7 * 2,
                              4);
    *oq2 = ROUND_POWER_OF_TWO(
        p4 + p3 + p2 + p1 + p0 + q0 + q1 + q2 * 2 + q3 + q4 + q5 + q6 + q7 * 3,
        4);
    *oq3 = ROUND_POWER_OF_TWO(
        p3 + p2 + p1 + p0 + q0 + q1 + q2 + q3 * 2 + q4 + q5 + q6 + q7 * 4, 4);
    *oq4 = ROUND_POWER_OF_TWO(
        p2 + p1 + p0 + q0 + q1 + q2 + q3 + q4 * 2 + q5 + q6 + q7 * 5, 4);
    *oq5 = ROUND_POWER_OF_TWO(
        p1 + p0 + q0 + q1 + q2 + q3 + q4 + q5 * 2 + q6 + q7 * 6, 4);
    *oq6 = ROUND_POWER_OF_TWO(
        p0 + q0 + q1 + q2 + q3 + q4 + q5 + q6 * 2 + q7 * 7, 4);
542
  } else {
543
    filter8(mask, thresh, flat, op3, op2, op1, op0, oq0, oq1, oq2, oq3);
544 545 546
  }
}

547 548 549
static void mb_lpf_horizontal_edge_w(uint8_t *s, int p, const uint8_t *blimit,
                                     const uint8_t *limit,
                                     const uint8_t *thresh, int count) {
550
  int i;
551
#if CONFIG_PARALLEL_DEBLOCKING
552 553 554 555
  int step = 4;
#else
  int step = 8;
#endif
556

Dmitry Kovalev's avatar
Dmitry Kovalev committed
557 558
  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
559
  for (i = 0; i < step * count; ++i) {
560 561 562 563 564
    const uint8_t p7 = s[-8 * p], p6 = s[-7 * p], p5 = s[-6 * p],
                  p4 = s[-5 * p], p3 = s[-4 * p], p2 = s[-3 * p],
                  p1 = s[-2 * p], p0 = s[-p];
    const uint8_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p],
                  q4 = s[4 * p], q5 = s[5 * p], q6 = s[6 * p], q7 = s[7 * p];
clang-format's avatar
clang-format committed
565 566
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
567
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
568

Ola Hugosson's avatar
Ola Hugosson committed
569 570 571 572 573 574 575 576 577 578
#if PARALLEL_DEBLOCKING_13_TAP
    (void)p7;
    (void)q7;
    const int8_t flat2 = flat_mask4(1, p6, p5, p4, p0, q0, q4, q5, q6);

    filter14(mask, *thresh, flat, flat2, s - 7 * p, s - 6 * p, s - 5 * p,
             s - 4 * p, s - 3 * p, s - 2 * p, s - 1 * p, s, s + 1 * p,
             s + 2 * p, s + 3 * p, s + 4 * p, s + 5 * p, s + 6 * p);

#elif PARALLEL_DEBLOCKING_11_TAP
579 580 581 582 583 584 585 586 587 588 589 590 591 592
    const int8_t flat2 = flat_mask3(1, p5, p4, p0, q0, q4, q5);

    filter12(mask, *thresh, flat, flat2, s - 6 * p, s - 5 * p, s - 4 * p,
             s - 3 * p, s - 2 * p, s - 1 * p, s, s + 1 * p, s + 2 * p,
             s + 3 * p, s + 4 * p, s + 5 * p);

#elif PARALLEL_DEBLOCKING_9_TAP
    const int8_t flat2 = flat_mask2(1, p4, p0, q0, q4);

    filter10(mask, *thresh, flat, flat2, s - 5 * p, s - 4 * p, s - 3 * p,
             s - 2 * p, s - 1 * p, s, s + 1 * p, s + 2 * p, s + 3 * p,
             s + 4 * p);
#else
    const int8_t flat2 = flat_mask5(1, p7, p6, p5, p4, p0, q0, q4, q5, q6, q7);
clang-format's avatar
clang-format committed
593 594 595 596 597

    filter16(mask, *thresh, flat, flat2, s - 8 * p, s - 7 * p, s - 6 * p,
             s - 5 * p, s - 4 * p, s - 3 * p, s - 2 * p, s - 1 * p, s,
             s + 1 * p, s + 2 * p, s + 3 * p, s + 4 * p, s + 5 * p, s + 6 * p,
             s + 7 * p);
598 599
#endif

600
    ++s;
601
  }
602
}
603

Yaowu Xu's avatar
Yaowu Xu committed
604
void aom_lpf_horizontal_edge_8_c(uint8_t *s, int p, const uint8_t *blimit,
605 606 607 608
                                 const uint8_t *limit, const uint8_t *thresh) {
  mb_lpf_horizontal_edge_w(s, p, blimit, limit, thresh, 1);
}

Yaowu Xu's avatar
Yaowu Xu committed
609
void aom_lpf_horizontal_edge_16_c(uint8_t *s, int p, const uint8_t *blimit,
610
                                  const uint8_t *limit, const uint8_t *thresh) {
611
#if CONFIG_PARALLEL_DEBLOCKING
612 613
  mb_lpf_horizontal_edge_w(s, p, blimit, limit, thresh, 1);
#else
614
  mb_lpf_horizontal_edge_w(s, p, blimit, limit, thresh, 2);
615
#endif
616 617
}

clang-format's avatar
clang-format committed
618 619
static void mb_lpf_vertical_edge_w(uint8_t *s, int p, const uint8_t *blimit,
                                   const uint8_t *limit, const uint8_t *thresh,
620
                                   int count) {
621 622
  int i;

623
  for (i = 0; i < count; ++i) {
624 625 626 627
    const uint8_t p7 = s[-8], p6 = s[-7], p5 = s[-6], p4 = s[-5], p3 = s[-4],
                  p2 = s[-3], p1 = s[-2], p0 = s[-1];
    const uint8_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3], q4 = s[4],
                  q5 = s[5], q6 = s[6], q7 = s[7];
clang-format's avatar
clang-format committed
628 629
    const int8_t mask =
        filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3);
630
    const int8_t flat = flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3);
631

Ola Hugosson's avatar
Ola Hugosson committed
632 633 634 635 636 637 638 639
#if PARALLEL_DEBLOCKING_13_TAP
    (void)p7;
    (void)q7;
    const int8_t flat2 = flat_mask4(1, p6, p5, p4, p0, q0, q4, q5, q6);

    filter14(mask, *thresh, flat, flat2, s - 7, s - 6, s - 5, s - 4, s - 3,
             s - 2, s - 1, s, s + 1, s + 2, s + 3, s + 4, s + 5, s + 6);
#elif PARALLEL_DEBLOCKING_11_TAP
640 641 642 643 644 645 646 647 648 649 650 651
    const int8_t flat2 = flat_mask3(1, p5, p4, p0, q0, q4, q5);

    filter12(mask, *thresh, flat, flat2, s - 6, s - 5, s - 4, s - 3, s - 2,
             s - 1, s, s + 1, s + 2, s + 3, s + 4, s + 5);
#elif PARALLEL_DEBLOCKING_9_TAP
    const int8_t flat2 = flat_mask2(1, p4, p0, q0, q4);

    filter10(mask, *thresh, flat, flat2, s - 5, s - 4, s - 3, s - 2, s - 1, s,
             s + 1, s + 2, s + 3, s + 4);

#else
    const int8_t flat2 = flat_mask5(1, p7, p6, p5, p4, p0, q0, q4, q5, q6, q7);
652

clang-format's avatar
clang-format committed
653 654 655
    filter16(mask, *thresh, flat, flat2, s - 8, s - 7, s - 6, s - 5, s - 4,
             s - 3, s - 2, s - 1, s, s + 1, s + 2, s + 3, s + 4, s + 5, s + 6,
             s + 7);
656 657
#endif

658
    s += p;
659
  }
660
}
661

Yaowu Xu's avatar
Yaowu Xu committed
662
void aom_lpf_vertical_16_c(uint8_t *s, int p, const uint8_t *blimit,
Jim Bankoski's avatar
Jim Bankoski committed
663
                           const uint8_t *limit, const uint8_t *thresh) {
664
#if CONFIG_PARALLEL_DEBLOCKING
665 666
  mb_lpf_vertical_edge_w(s, p, blimit, limit, thresh, 4);
#else
667
  mb_lpf_vertical_edge_w(s, p, blimit, limit, thresh, 8);
668
#endif
669 670
}

Yaowu Xu's avatar
Yaowu Xu committed
671
void aom_lpf_vertical_16_dual_c(uint8_t *s, int p, const uint8_t *blimit,
Jim Bankoski's avatar
Jim Bankoski committed
672
                                const uint8_t *limit, const uint8_t *thresh) {
673
  mb_lpf_vertical_edge_w(s, p, blimit, limit, thresh, 16);
674
}
675

676
#if CONFIG_HIGHBITDEPTH
677 678 679 680 681 682 683 684 685 686 687 688 689 690 691
#if CONFIG_PARALLEL_DEBLOCKING
// Should we apply any filter at all: 11111111 yes, 00000000 no ?
static INLINE int8_t highbd_filter_mask2(uint8_t limit, uint8_t blimit,
                                         uint16_t p1, uint16_t p0, uint16_t q0,
                                         uint16_t q1, int bd) {
  int8_t mask = 0;
  int16_t limit16 = (uint16_t)limit << (bd - 8);
  int16_t blimit16 = (uint16_t)blimit << (bd - 8);
  mask |= (abs(p1 - p0) > limit16) * -1;
  mask |= (abs(q1 - q0) > limit16) * -1;
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit16) * -1;
  return ~mask;
}
#endif  // CONFIG_PARALLEL_DEBLOCKING

692
// Should we apply any filter at all: 11111111 yes, 00000000 no ?
693
static INLINE int8_t highbd_filter_mask(uint8_t limit, uint8_t blimit,
clang-format's avatar
clang-format committed
694 695
                                        uint16_t p3, uint16_t p2, uint16_t p1,
                                        uint16_t p0, uint16_t q0, uint16_t q1,
696
                                        uint16_t q2, uint16_t q3, int bd) {
697 698 699 700 701 702 703 704 705
  int8_t mask = 0;
  int16_t limit16 = (uint16_t)limit << (bd - 8);
  int16_t blimit16 = (uint16_t)blimit << (bd - 8);
  mask |= (abs(p3 - p2) > limit16) * -1;
  mask |= (abs(p2 - p1) > limit16) * -1;
  mask |= (abs(p1 - p0) > limit16) * -1;
  mask |= (abs(q1 - q0) > limit16) * -1;
  mask |= (abs(q2 - q1) > limit16) * -1;
  mask |= (abs(q3 - q2) > limit16) * -1;
clang-format's avatar
clang-format committed
706
  mask |= (abs(p0 - q0) * 2 + abs(p1 - q1) / 2 > blimit16) * -1;
707 708 709
  return ~mask;
}

Ola Hugosson's avatar
Ola Hugosson committed
710 711 712 713 714 715 716 717 718 719 720 721 722 723 724
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE int8_t highbd_flat_mask3_chroma(uint8_t thresh, uint16_t p2,
                                              uint16_t p1, uint16_t p0,
                                              uint16_t q0, uint16_t q1,
                                              uint16_t q2, int bd) {
  int8_t mask = 0;
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  mask |= (abs(p1 - p0) > thresh16) * -1;
  mask |= (abs(q1 - q0) > thresh16) * -1;
  mask |= (abs(p2 - p0) > thresh16) * -1;
  mask |= (abs(q2 - q0) > thresh16) * -1;
  return ~mask;
}
#endif

clang-format's avatar
clang-format committed
725 726 727 728
static INLINE int8_t highbd_flat_mask4(uint8_t thresh, uint16_t p3, uint16_t p2,
                                       uint16_t p1, uint16_t p0, uint16_t q0,
                                       uint16_t q1, uint16_t q2, uint16_t q3,
                                       int bd) {
729 730 731 732 733 734 735 736 737 738 739
  int8_t mask = 0;
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  mask |= (abs(p1 - p0) > thresh16) * -1;
  mask |= (abs(q1 - q0) > thresh16) * -1;
  mask |= (abs(p2 - p0) > thresh16) * -1;
  mask |= (abs(q2 - q0) > thresh16) * -1;
  mask |= (abs(p3 - p0) > thresh16) * -1;
  mask |= (abs(q3 - q0) > thresh16) * -1;
  return ~mask;
}

clang-format's avatar
clang-format committed
740 741 742
static INLINE int8_t highbd_flat_mask5(uint8_t thresh, uint16_t p4, uint16_t p3,
                                       uint16_t p2, uint16_t p1, uint16_t p0,
                                       uint16_t q0, uint16_t q1, uint16_t q2,
743 744
                                       uint16_t q3, uint16_t q4, int bd) {
  int8_t mask = ~highbd_flat_mask4(thresh, p3, p2, p1, p0, q0, q1, q2, q3, bd);
745 746 747 748 749 750 751 752
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  mask |= (abs(p4 - p0) > thresh16) * -1;
  mask |= (abs(q4 - q0) > thresh16) * -1;
  return ~mask;
}

// Is there high edge variance internal edge:
// 11111111_11111111 yes, 00000000_00000000 no ?
753 754
static INLINE int16_t highbd_hev_mask(uint8_t thresh, uint16_t p1, uint16_t p0,
                                      uint16_t q0, uint16_t q1, int bd) {
755 756 757 758 759 760 761
  int16_t hev = 0;
  int16_t thresh16 = (uint16_t)thresh << (bd - 8);
  hev |= (abs(p1 - p0) > thresh16) * -1;
  hev |= (abs(q1 - q0) > thresh16) * -1;
  return hev;
}

762 763 764
static INLINE void highbd_filter4(int8_t mask, uint8_t thresh, uint16_t *op1,
                                  uint16_t *op0, uint16_t *oq0, uint16_t *oq1,
                                  int bd) {
765 766 767 768 769 770 771 772
  int16_t filter1, filter2;
  // ^0x80 equivalent to subtracting 0x80 from the values to turn them
  // into -128 to +127 instead of 0 to 255.
  int shift = bd - 8;
  const int16_t ps1 = (int16_t)*op1 - (0x80 << shift);
  const int16_t ps0 = (int16_t)*op0 - (0x80 << shift);
  const int16_t qs0 = (int16_t)*oq0 - (0x80 << shift);
  const int16_t qs1 = (int16_t)*oq1 - (0x80 << shift);
773
  const uint16_t hev = highbd_hev_mask(thresh, *op1, *op0, *oq0, *oq1, bd);
774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796

  // Add outer taps if we have high edge variance.
  int16_t filter = signed_char_clamp_high(ps1 - qs1, bd) & hev;

  // Inner taps.
  filter = signed_char_clamp_high(filter + 3 * (qs0 - ps0), bd) & mask;

  // Save bottom 3 bits so that we round one side +4 and the other +3
  // if it equals 4 we'll set to adjust by -1 to account for the fact
  // we'd round 3 the other way.
  filter1 = signed_char_clamp_high(filter + 4, bd) >> 3;
  filter2 = signed_char_clamp_high(filter + 3, bd) >> 3;

  *oq0 = signed_char_clamp_high(qs0 - filter1, bd) + (0x80 << shift);
  *op0 = signed_char_clamp_high(ps0 + filter2, bd) + (0x80 << shift);

  // Outer tap adjustments.
  filter = ROUND_POWER_OF_TWO(filter1, 1) & ~hev;

  *oq1 = signed_char_clamp_high(qs1 - filter, bd) + (0x80 << shift);
  *op1 = signed_char_clamp_high(ps1 + filter, bd) + (0x80 << shift);
}

Yaowu Xu's avatar
Yaowu Xu committed
797
void aom_highbd_lpf_horizontal_4_c(uint16_t *s, int p /* pitch */,
798
                                   const uint8_t *blimit, const uint8_t *limit,
799
                                   const uint8_t *thresh, int bd) {
800
  int i;
801
#if CONFIG_PARALLEL_DEBLOCKING
802 803 804 805
  int count = 4;
#else
  int count = 8;
#endif
806 807 808

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
809
  for (i = 0; i < count; ++i) {
810
#if !CONFIG_PARALLEL_DEBLOCKING
811 812 813 814 815 816 817 818
    const uint16_t p3 = s[-4 * p];
    const uint16_t p2 = s[-3 * p];
    const uint16_t p1 = s[-2 * p];
    const uint16_t p0 = s[-p];
    const uint16_t q0 = s[0 * p];
    const uint16_t q1 = s[1 * p];
    const uint16_t q2 = s[2 * p];
    const uint16_t q3 = s[3 * p];
clang-format's avatar
clang-format committed
819 820
    const int8_t mask =
        highbd_filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3, bd);
821 822 823 824 825 826 827 828
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint16_t p1 = s[-2 * p];
    const uint16_t p0 = s[-p];
    const uint16_t q0 = s[0 * p];
    const uint16_t q1 = s[1 * p];
    const int8_t mask =
        highbd_filter_mask2(*limit, *blimit, p1, p0, q0, q1, bd);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
829
    highbd_filter4(mask, *thresh, s - 2 * p, s - 1 * p, s, s + 1 * p, bd);
830 831 832 833
    ++s;
  }
}

Yaowu Xu's avatar
Yaowu Xu committed
834
void aom_highbd_lpf_horizontal_4_dual_c(
clang-format's avatar
clang-format committed
835 836 837
    uint16_t *s, int p, const uint8_t *blimit0, const uint8_t *limit0,
    const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
    const uint8_t *thresh1, int bd) {
Yaowu Xu's avatar
Yaowu Xu committed
838 839
  aom_highbd_lpf_horizontal_4_c(s, p, blimit0, limit0, thresh0, bd);
  aom_highbd_lpf_horizontal_4_c(s + 8, p, blimit1, limit1, thresh1, bd);
840 841
}

Yaowu Xu's avatar
Yaowu Xu committed
842
void aom_highbd_lpf_vertical_4_c(uint16_t *s, int pitch, const uint8_t *blimit,
843
                                 const uint8_t *limit, const uint8_t *thresh,
844
                                 int bd) {
845
  int i;
846
#if CONFIG_PARALLEL_DEBLOCKING
847 848 849 850
  int count = 4;
#else
  int count = 8;
#endif
851 852 853

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
854
  for (i = 0; i < count; ++i) {
855
#if !CONFIG_PARALLEL_DEBLOCKING
856
    const uint16_t p3 = s[-4], p2 = s[-3], p1 = s[-2], p0 = s[-1];
clang-format's avatar
clang-format committed
857 858 859
    const uint16_t q0 = s[0], q1 = s[1], q2 = s[2], q3 = s[3];
    const int8_t mask =
        highbd_filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3, bd);
860 861 862 863 864 865
#else   // CONFIG_PARALLEL_DEBLOCKING
    const uint16_t p1 = s[-2], p0 = s[-1];
    const uint16_t q0 = s[0], q1 = s[1];
    const int8_t mask =
        highbd_filter_mask2(*limit, *blimit, p1, p0, q0, q1, bd);
#endif  // !CONFIG_PARALLEL_DEBLOCKING
866
    highbd_filter4(mask, *thresh, s - 2, s - 1, s, s + 1, bd);
867 868 869 870
    s += pitch;
  }
}

Yaowu Xu's avatar
Yaowu Xu committed
871
void aom_highbd_lpf_vertical_4_dual_c(
clang-format's avatar
clang-format committed
872 873 874
    uint16_t *s, int pitch, const uint8_t *blimit0, const uint8_t *limit0,
    const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
    const uint8_t *thresh1, int bd) {
Yaowu Xu's avatar
Yaowu Xu committed
875 876
  aom_highbd_lpf_vertical_4_c(s, pitch, blimit0, limit0, thresh0, bd);
  aom_highbd_lpf_vertical_4_c(s + 8 * pitch, pitch, blimit1, limit1, thresh1,
clang-format's avatar
clang-format committed
877
                              bd);
878 879
}

Ola Hugosson's avatar
Ola Hugosson committed
880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
static INLINE void highbd_filter6(int8_t mask, uint8_t thresh, int8_t flat,
                                  uint16_t *op2, uint16_t *op1, uint16_t *op0,
                                  uint16_t *oq0, uint16_t *oq1, uint16_t *oq2,
                                  int bd) {
  if (flat && mask) {
    const uint16_t p2 = *op2, p1 = *op1, p0 = *op0;
    const uint16_t q0 = *oq0, q1 = *oq1, q2 = *oq2;

    // 5-tap filter [1, 2, 2, 2, 1]
    *op1 = ROUND_POWER_OF_TWO(p2 * 3 + p1 * 2 + p0 * 2 + q0, 3);
    *op0 = ROUND_POWER_OF_TWO(p2 + p1 * 2 + p0 * 2 + q0 * 2 + q1, 3);
    *oq0 = ROUND_POWER_OF_TWO(p1 + p0 * 2 + q0 * 2 + q1 * 2 + q2, 3);
    *oq1 = ROUND_POWER_OF_TWO(p0 + q0 * 2 + q1 * 2 + q2 * 3, 3);
  } else {
    highbd_filter4(mask, thresh, op1, op0, oq0, oq1, bd);
  }
}
#endif

900
static INLINE void highbd_filter8(int8_t mask, uint8_t thresh, int8_t flat,
clang-format's avatar
clang-format committed
901 902
                                  uint16_t *op3, uint16_t *op2, uint16_t *op1,
                                  uint16_t *op0, uint16_t *oq0, uint16_t *oq1,
903
                                  uint16_t *oq2, uint16_t *oq3, int bd) {
904 905 906 907 908 909 910 911 912 913 914 915
  if (flat && mask) {
    const uint16_t p3 = *op3, p2 = *op2, p1 = *op1, p0 = *op0;
    const uint16_t q0 = *oq0, q1 = *oq1, q2 = *oq2, q3 = *oq3;

    // 7-tap filter [1, 1, 1, 2, 1, 1, 1]
    *op2 = ROUND_POWER_OF_TWO(p3 + p3 + p3 + 2 * p2 + p1 + p0 + q0, 3);
    *op1 = ROUND_POWER_OF_TWO(p3 + p3 + p2 + 2 * p1 + p0 + q0 + q1, 3);
    *op0 = ROUND_POWER_OF_TWO(p3 + p2 + p1 + 2 * p0 + q0 + q1 + q2, 3);
    *oq0 = ROUND_POWER_OF_TWO(p2 + p1 + p0 + 2 * q0 + q1 + q2 + q3, 3);
    *oq1 = ROUND_POWER_OF_TWO(p1 + p0 + q0 + 2 * q1 + q2 + q3 + q3, 3);
    *oq2 = ROUND_POWER_OF_TWO(p0 + q0 + q1 + 2 * q2 + q3 + q3 + q3, 3);
  } else {
clang-format's avatar
clang-format committed
916
    highbd_filter4(mask, thresh, op1, op0, oq0, oq1, bd);
917 918 919
  }
}

Yaowu Xu's avatar
Yaowu Xu committed
920
void aom_highbd_lpf_horizontal_8_c(uint16_t *s, int p, const uint8_t *blimit,
921
                                   const uint8_t *limit, const uint8_t *thresh,
922
                                   int bd) {
923
  int i;
924
#if CONFIG_PARALLEL_DEBLOCKING
925 926 927 928
  int count = 4;
#else
  int count = 8;
#endif
929 930 931

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
932
  for (i = 0; i < count; ++i) {
933 934 935
    const uint16_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
    const uint16_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];

clang-format's avatar
clang-format committed
936 937 938 939 940 941
    const int8_t mask =
        highbd_filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3, bd);
    const int8_t flat =
        highbd_flat_mask4(1, p3, p2, p1, p0, q0, q1, q2, q3, bd);
    highbd_filter8(mask, *thresh, flat, s - 4 * p, s - 3 * p, s - 2 * p,
                   s - 1 * p, s, s + 1 * p, s + 2 * p, s + 3 * p, bd);
942 943 944 945
    ++s;
  }
}

Ola Hugosson's avatar
Ola Hugosson committed
946 947 948 949 950
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
void aom_highbd_lpf_horizontal_6_c(uint16_t *s, int p, const uint8_t *blimit,
                                   const uint8_t *limit, const uint8_t *thresh,
                                   int bd) {
  int i;
951
#if CONFIG_PARALLEL_DEBLOCKING
Ola Hugosson's avatar
Ola Hugosson committed
952 953 954 955 956 957 958 959 960 961 962 963 964 965 966 967 968 969 970 971 972
  int count = 4;
#else
  int count = 8;
#endif

  // loop filter designed to work using chars so that we can make maximum use
  // of 8 bit simd instructions.
  for (i = 0; i < count; ++i) {
    const uint16_t p3 = s[-4 * p], p2 = s[-3 * p], p1 = s[-2 * p], p0 = s[-p];
    const uint16_t q0 = s[0 * p], q1 = s[1 * p], q2 = s[2 * p], q3 = s[3 * p];

    const int8_t mask =
        highbd_filter_mask(*limit, *blimit, p3, p2, p1, p0, q0, q1, q2, q3, bd);
    const int8_t flat = highbd_flat_mask3_chroma(1, p2, p1, p0, q0, q1, q2, bd);
    highbd_filter6(mask, *thresh, flat, s - 3 * p, s - 2 * p, s - 1 * p, s,
                   s + 1 * p, s + 2 * p, bd);
    ++s;
  }
}
#endif

Yaowu Xu's avatar
Yaowu Xu committed
973
void aom_highbd_lpf_horizontal_8_dual_c(
clang-format's avatar
clang-format committed
974 975 976
    uint16_t *s, int p, const uint8_t *blimit0, const uint8_t *limit0,
    const uint8_t *thresh0, const uint8_t *blimit1, const uint8_t *limit1,
    const uint8_t *thresh1, int bd) {
Yaowu Xu's avatar
Yaowu Xu committed
977 978
  aom_highbd_lpf_horizontal_8_c(s, p, blimit0, limit0, thresh0, bd);
  aom_highbd_lpf_horizontal_8_c(s + 8, p, blimit1, limit1, thresh1, bd);
979 980
}

Ola Hugosson's avatar
Ola Hugosson committed
981 982 983 984 985
#if PARALLEL_DEBLOCKING_5_TAP_CHROMA
void aom_highbd_lpf_vertical_6_c(uint16_t *s, int pitch, const uint8_t *blimit,
                                 const uint8_t *limit, const uint8_t *thresh,
                                 int bd) {
  int i;