Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 3 additions & 10 deletions libvmaf/src/feature/arm64/motion_neon.c
Original file line number Diff line number Diff line change
Expand Up @@ -5,13 +5,6 @@

#include "feature/integer_motion.h"

static inline int mirror(int idx, int size)
{
if (idx < 0) return -idx;
if (idx >= size) return 2 * size - idx - 2;
return idx;
}

// Phase 2: x_conv + abs + SAD for one row of int32 y_row.
// Processes 4 int32 columns at a time via widening s32->s64 multiply-accumulate.
static inline uint32_t
Expand All @@ -24,7 +17,7 @@ x_conv_row_sad_neon(const int32_t *y_row, unsigned w)
for (j = 0; j < 2 && j < w; j++) {
int64_t accum = 0;
for (int k = 0; k < 5; k++) {
int col = mirror((int)j - 2 + k, (int)w);
int col = motion_mirror((int)j - 2 + k, (int)w);
accum += (int64_t)filter[k] * y_row[col];
}
int32_t val = (int32_t)((accum + (1 << 15)) >> 16);
Expand Down Expand Up @@ -68,7 +61,7 @@ x_conv_row_sad_neon(const int32_t *y_row, unsigned w)
for (; j < w; j++) {
int64_t accum = 0;
for (int k = 0; k < 5; k++) {
int col = mirror((int)j - 2 + k, (int)w);
int col = motion_mirror((int)j - 2 + k, (int)w);
accum += (int64_t)filter[k] * y_row[col];
}
int32_t val = (int32_t)((accum + (1 << 15)) >> 16);
Expand All @@ -89,7 +82,7 @@ uint64_t motion_score_pipeline_8_neon(const uint8_t *prev, ptrdiff_t prev_stride
for (unsigned i = 0; i < h; i++) {
const uint8_t *p[5], *c[5];
for (int k = 0; k < 5; k++) {
int r = mirror((int)i - 2 + k, (int)h);
int r = motion_mirror((int)i - 2 + k, (int)h);
p[k] = prev + r * prev_stride;
c[k] = cur + r * cur_stride;
}
Expand Down
15 changes: 15 additions & 0 deletions libvmaf/src/feature/common/convolution.c
Original file line number Diff line number Diff line change
Expand Up @@ -25,12 +25,21 @@
extern int vmaf_floorn(int, int);
extern int vmaf_ceiln(int, int);

#define MIN(x, y) (((x) < (y)) ? (x) : (y))
#define MAX(x, y) (((x) > (y)) ? (x) : (y))

void convolution_x_c_s(const float *filter, int filter_width, const float *src, float *dst, int width, int height, int src_stride, int dst_stride, int step)
{
int radius = filter_width / 2;
int borders_left = vmaf_ceiln(radius, step);
int borders_right = vmaf_floorn(width - (filter_width - radius), step);

// Clamp so tiny widths (< filter_width) cannot produce negative loop bounds,
// which would make the trailing edge loop start at a negative index and write
// outside dst. No-op for width >= filter_width.
borders_left = MIN(borders_left, width);
borders_right = MAX(borders_right, borders_left);

for (int i = 0; i < height; ++i) {
for (int j = 0; j < borders_left; j += step) {
dst[i * dst_stride + j / step] = convolution_edge_s(true, filter, filter_width, src, width, height, src_stride, i, j);
Expand All @@ -56,6 +65,12 @@ void convolution_y_c_s(const float *filter, int filter_width, const float *src,
int borders_top = vmaf_ceiln(radius, step);
int borders_bottom = vmaf_floorn(height - (filter_width - radius), step);

// Clamp so tiny heights (< filter_width) cannot produce negative loop bounds,
// which would make the trailing edge loop start at a negative index and write
// outside dst. No-op for height >= filter_width.
borders_top = MIN(borders_top, height);
borders_bottom = MAX(borders_bottom, borders_top);

for (int i = 0; i < borders_top; i += step) {
for (int j = 0; j < width; ++j) {
dst[(i / step) * dst_stride + j] = convolution_edge_s(false, filter, filter_width, src, width, height, src_stride, i, j);
Expand Down
66 changes: 57 additions & 9 deletions libvmaf/src/feature/common/convolution_avx.c
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,9 @@

#define AVX_STEP (8)

#define MIN(x, y) (((x) < (y)) ? (x) : (y))
#define MAX(x, y) (((x) > (y)) ? (x) : (y))


void convolution_f32_avx_s_1d_h_scanline(const float * RESTRICT filter, int filter_width, const float * RESTRICT src, float * RESTRICT dst, int j_end) {
int radius = filter_width / 2;
Expand Down Expand Up @@ -131,8 +134,23 @@ void convolution_f32_avx_s(const float * RESTRICT filter, int filter_width, cons
int i_vec_end = height - radius;
int j_vec_end = vmaf_floorn(width - radius, AVX_STEP);

// Clamp edge-loop bounds so tiny width/height (< radius) cannot write outside
// the tmp/dst buffers. i_vec_end is safe to raise in place: its two uses are the
// vector loop's end bound and the trailing loop's start, and the raise only fires
// when the raised value is <= radius, which leaves the vector loop
// [radius, i_vec_end) empty. j_vec_end must NOT be raised
// in place: it is also passed unmodified to convolution_f32_avx_s_1d_h_scanline,
// whose j_end argument must stay <= vmaf_floorn(width - radius, AVX_STEP) or the
// scanline itself writes out of bounds (it stores dst[j + radius .. j + radius +
// AVX_STEP) for each j < j_end). So the horizontal trailing loop's clamped start
// goes in a SEPARATE variable, j_trailing_start.
int i_edge_end = MIN(radius, height);
i_vec_end = MAX(i_vec_end, i_edge_end);
int j_edge_end = MIN(radius, width);
int j_trailing_start = MAX(j_vec_end, j_edge_end);

// Vertical pass.
for (int i = 0; i < radius; ++i) {
for (int i = 0; i < i_edge_end; ++i) {
for (int j = 0; j < width; ++j) {
tmp[i * tmp_stride + j] = convolution_edge_s(false, filter, filter_width, src, width, height, src_stride, i, j);
}
Expand All @@ -152,13 +170,13 @@ void convolution_f32_avx_s(const float * RESTRICT filter, int filter_width, cons

// Horizontal pass.
for (int i = 0; i < height; ++i) {
for (int j = 0; j < radius; ++j) {
for (int j = 0; j < j_edge_end; ++j) {
dst[i * dst_stride + j] = convolution_edge_s(true, filter, filter_width, tmp, width, height, tmp_stride, i, j);
}

convolution_f32_avx_s_1d_h_scanline(filter, filter_width, tmp + i * tmp_stride, dst + i * dst_stride, j_vec_end);

for (int j = j_vec_end; j < width; ++j) {
for (int j = j_trailing_start; j < width; ++j) {
dst[i * dst_stride + j] = convolution_edge_s(true, filter, filter_width, tmp, width, height, tmp_stride, i, j);
}
}
Expand All @@ -172,8 +190,23 @@ void convolution_f32_avx_sq_s(const float * RESTRICT filter, int filter_width, c
int i_vec_end = height - radius;
int j_vec_end = vmaf_floorn(width - radius, AVX_STEP);

// Clamp edge-loop bounds so tiny width/height (< radius) cannot write outside
// the tmp/dst buffers. i_vec_end is safe to raise in place: its two uses are the
// vector loop's end bound and the trailing loop's start, and the raise only fires
// when the raised value is <= radius, which leaves the vector loop
// [radius, i_vec_end) empty. j_vec_end must NOT be raised
// in place: it is also passed unmodified to convolution_f32_avx_s_1d_h_scanline,
// whose j_end argument must stay <= vmaf_floorn(width - radius, AVX_STEP) or the
// scanline itself writes out of bounds (it stores dst[j + radius .. j + radius +
// AVX_STEP) for each j < j_end). So the horizontal trailing loop's clamped start
// goes in a SEPARATE variable, j_trailing_start.
int i_edge_end = MIN(radius, height);
i_vec_end = MAX(i_vec_end, i_edge_end);
int j_edge_end = MIN(radius, width);
int j_trailing_start = MAX(j_vec_end, j_edge_end);

// Vertical pass.
for (int i = 0; i < radius; ++i) {
for (int i = 0; i < i_edge_end; ++i) {
for (int j = 0; j < width; ++j) {
tmp[i * tmp_stride + j] = convolution_edge_sq_s(false, filter, filter_width, src, width, height, src_stride, i, j);
}
Expand All @@ -193,13 +226,13 @@ void convolution_f32_avx_sq_s(const float * RESTRICT filter, int filter_width, c

// Horizontal pass.
for (int i = 0; i < height; ++i) {
for (int j = 0; j < radius; ++j) {
for (int j = 0; j < j_edge_end; ++j) {
dst[i * dst_stride + j] = convolution_edge_s(true, filter, filter_width, tmp, width, height, tmp_stride, i, j);
}

convolution_f32_avx_s_1d_h_scanline(filter, filter_width, tmp + i * tmp_stride, dst + i * dst_stride, j_vec_end);

for (int j = j_vec_end; j < width; ++j) {
for (int j = j_trailing_start; j < width; ++j) {
dst[i * dst_stride + j] = convolution_edge_s(true, filter, filter_width, tmp, width, height, tmp_stride, i, j);
}
}
Expand All @@ -214,8 +247,23 @@ void convolution_f32_avx_xy_s(const float * RESTRICT filter, int filter_width, c
int i_vec_end = height - radius;
int j_vec_end = vmaf_floorn(width - radius, AVX_STEP);

// Clamp edge-loop bounds so tiny width/height (< radius) cannot write outside
// the tmp/dst buffers. i_vec_end is safe to raise in place: its two uses are the
// vector loop's end bound and the trailing loop's start, and the raise only fires
// when the raised value is <= radius, which leaves the vector loop
// [radius, i_vec_end) empty. j_vec_end must NOT be raised
// in place: it is also passed unmodified to convolution_f32_avx_s_1d_h_scanline,
// whose j_end argument must stay <= vmaf_floorn(width - radius, AVX_STEP) or the
// scanline itself writes out of bounds (it stores dst[j + radius .. j + radius +
// AVX_STEP) for each j < j_end). So the horizontal trailing loop's clamped start
// goes in a SEPARATE variable, j_trailing_start.
int i_edge_end = MIN(radius, height);
i_vec_end = MAX(i_vec_end, i_edge_end);
int j_edge_end = MIN(radius, width);
int j_trailing_start = MAX(j_vec_end, j_edge_end);

// Vertical pass.
for (int i = 0; i < radius; ++i) {
for (int i = 0; i < i_edge_end; ++i) {
for (int j = 0; j < width; ++j) {
tmp[i * tmp_stride + j] = convolution_edge_xy_s(false, filter, filter_width, src1, src2, width, height, src1_stride, src2_stride, i, j);
}
Expand All @@ -235,13 +283,13 @@ void convolution_f32_avx_xy_s(const float * RESTRICT filter, int filter_width, c

// Horizontal pass.
for (int i = 0; i < height; ++i) {
for (int j = 0; j < radius; ++j) {
for (int j = 0; j < j_edge_end; ++j) {
dst[i * dst_stride + j] = convolution_edge_s(true, filter, filter_width, tmp, width, height, tmp_stride, i, j);
}

convolution_f32_avx_s_1d_h_scanline(filter, filter_width, tmp + i * tmp_stride, dst + i * dst_stride, j_vec_end);

for (int j = j_vec_end; j < width; ++j) {
for (int j = j_trailing_start; j < width; ++j) {
dst[i * dst_stride + j] = convolution_edge_s(true, filter, filter_width, tmp, width, height, tmp_stride, i, j);
}
}
Expand Down
48 changes: 24 additions & 24 deletions libvmaf/src/feature/common/convolution_internal.h
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,24 @@
#include "macros.h"
#include <stdbool.h>

/* Whole-sample symmetric ("reflect-101") boundary index; same semantics as
* motion_mirror() in feature/integer_motion.h. Reflects repeatedly so any size >= 1
* is safe; for in-contract offsets (|tap - i or j| within the filter's radius) at
* most one reflection is taken, so behavior is bit-identical to the historical
* single-bounce version for size >= radius+1. size <= 1 is special-cased: the
* reflection period 2*(size-1) is 0 (or negative) and the loop would not
* terminate. size is always >= 1 in the real call graph, so covering size <= 0
* as well only makes the function total; it changes no reachable result. */
FORCE_INLINE int convolution_mirror(int tap, int size)
{
if (size <= 1) return 0;
while (tap < 0 || tap >= size) {
if (tap < 0) tap = -tap;
else tap = 2 * size - tap - 2;
}
return tap;
}

FORCE_INLINE float convolution_edge_s(bool horizontal, const float *filter, int filter_width, const float *src, int width, int height, int stride, int i, int j)
{
int radius = filter_width / 2;
Expand All @@ -35,15 +53,9 @@ FORCE_INLINE float convolution_edge_s(bool horizontal, const float *filter, int

// Handle edges by mirroring.
if (horizontal) {
if (j_tap < 0)
j_tap = -j_tap;
else if (j_tap >= width)
j_tap = width - (j_tap - width + 2);
j_tap = convolution_mirror(j_tap, width);
} else {
if (i_tap < 0)
i_tap = -i_tap;
else if (i_tap >= height)
i_tap = height - (i_tap - height + 2);
i_tap = convolution_mirror(i_tap, height);
}

accum += filter[k] * src[i_tap * stride + j_tap];
Expand All @@ -63,16 +75,10 @@ FORCE_INLINE float convolution_edge_sq_s(bool horizontal, const float *filter, i

// Handle edges by mirroring.
if (horizontal) {
if (j_tap < 0)
j_tap = -j_tap;
else if (j_tap >= width)
j_tap = width - (j_tap - width + 2);
j_tap = convolution_mirror(j_tap, width);
}
else {
if (i_tap < 0)
i_tap = -i_tap;
else if (i_tap >= height)
i_tap = height - (i_tap - height + 2);
i_tap = convolution_mirror(i_tap, height);
}
src_val = src[i_tap * stride + j_tap];
accum += filter[k] * (src_val * src_val);
Expand All @@ -92,16 +98,10 @@ FORCE_INLINE float convolution_edge_xy_s(bool horizontal, const float *filter, i

// Handle edges by mirroring.
if (horizontal) {
if (j_tap < 0)
j_tap = -j_tap;
else if (j_tap >= width)
j_tap = width - (j_tap - width + 2);
j_tap = convolution_mirror(j_tap, width);
}
else {
if (i_tap < 0)
i_tap = -i_tap;
else if (i_tap >= height)
i_tap = height - (i_tap - height + 2);
i_tap = convolution_mirror(i_tap, height);
}
src_val1 = src1[i_tap * stride1 + j_tap];
src_val2 = src2[i_tap * stride2 + j_tap];
Expand Down
1 change: 1 addition & 0 deletions libvmaf/src/feature/cuda/integer_motion/motion_score.cu
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@ __constant__ int radius = (sizeof(filter_d) / sizeof(filter_d[0])) / 2;
// Device function that mirrors an idx along its valid [0,sup) range
__device__ __forceinline__ int mirror(const int idx, const int sup)
{
if (sup == 1) return 0;
int out = abs(idx);
return (out < sup) ? out : (sup - (out - sup + 1));
}
Expand Down
15 changes: 4 additions & 11 deletions libvmaf/src/feature/integer_motion.c
Original file line number Diff line number Diff line change
Expand Up @@ -146,13 +146,6 @@ static const VmafOption options[] = {
{ 0 }
};

static inline int mirror(int idx, int size)
{
if (idx < 0) return -idx;
if (idx >= size) return 2 * size - idx - 2;
return idx;
}

static uint64_t
motion_score_pipeline_8(const uint8_t *prev, ptrdiff_t prev_stride,
const uint8_t *cur, ptrdiff_t cur_stride,
Expand All @@ -172,7 +165,7 @@ motion_score_pipeline_8(const uint8_t *prev, ptrdiff_t prev_stride,
for (unsigned j = 0; j < w; j++) {
int32_t accum = 0;
for (int k = 0; k < filter_width; k++) {
const int row = mirror((int)i - radius + k, (int)h);
const int row = motion_mirror((int)i - radius + k, (int)h);
int32_t diff = prev[row * prev_stride + j]
- cur[row * cur_stride + j];
accum += (int32_t)filter[k] * diff;
Expand All @@ -188,7 +181,7 @@ motion_score_pipeline_8(const uint8_t *prev, ptrdiff_t prev_stride,
for (unsigned j = 0; j < w; j++) {
int64_t accum = 0;
for (int k = 0; k < filter_width; k++) {
const int col = mirror((int)j - radius + k, (int)w);
const int col = motion_mirror((int)j - radius + k, (int)w);
accum += (int64_t)filter[k] * y_row[col];
}
int32_t val = (int32_t)((accum + x_round) >> 16);
Expand Down Expand Up @@ -223,7 +216,7 @@ motion_score_pipeline_16(const uint8_t *prev_u8, ptrdiff_t prev_stride,
for (unsigned j = 0; j < w; j++) {
int64_t accum = 0;
for (int k = 0; k < filter_width; k++) {
const int row = mirror((int)i - radius + k, (int)h);
const int row = motion_mirror((int)i - radius + k, (int)h);
int32_t diff = prev[row * p_stride + j]
- cur[row * c_stride + j];
accum += (int64_t)filter[k] * diff;
Expand All @@ -239,7 +232,7 @@ motion_score_pipeline_16(const uint8_t *prev_u8, ptrdiff_t prev_stride,
for (unsigned j = 0; j < w; j++) {
int64_t accum = 0;
for (int k = 0; k < filter_width; k++) {
const int col = mirror((int)j - radius + k, (int)w);
const int col = motion_mirror((int)j - radius + k, (int)w);
accum += (int64_t)filter[k] * y_row[col];
}
int32_t val = (int32_t)((accum + x_round) >> 16);
Expand Down
38 changes: 14 additions & 24 deletions libvmaf/src/feature/integer_motion.h
Original file line number Diff line number Diff line change
Expand Up @@ -25,32 +25,22 @@
static const uint16_t filter[5] = { 3571, 16004, 26386, 16004, 3571 };
static const int filter_width = sizeof(filter) / sizeof(filter[0]);

static inline uint32_t
edge_16(bool horizontal, const uint16_t *src, int width,
int height, int stride, int i, int j)
/* Whole-sample symmetric ("reflect-101") boundary index.
* Reflects repeatedly so that any size >= 1 is safe; for size >= 3 and
* |overshoot| <= 2 (the 5-tap/radius-2 call contract) at most one
* reflection is taken, so behavior is bit-identical to the historical
* single-bounce version. size <= 1 must be special-cased: the reflection
* period 2*(size-1) is 0 (or negative) and the loop would not terminate.
* size is always >= 1 in the real call graph, so covering size <= 0 as well
* only makes the function total; it changes no reachable result. */
static inline int motion_mirror(int idx, int size)
{
const int radius = filter_width / 2;
uint32_t accum = 0;

// MIRROR | ЯOЯЯIM
for (int k = 0; k < filter_width; ++k) {
int i_tap = horizontal ? i : i - radius + k;
int j_tap = horizontal ? j - radius + k : j;

if (horizontal) {
if (j_tap < 0)
j_tap = -j_tap;
else if (j_tap >= width)
j_tap = width - (j_tap - width + 2);
} else {
if (i_tap < 0)
i_tap = -i_tap;
else if (i_tap >= height)
i_tap = height - (i_tap - height + 2);
}
accum += filter[k] * src[i_tap * stride + j_tap];
if (size <= 1) return 0;
while (idx < 0 || idx >= size) {
if (idx < 0) idx = -idx;
else idx = 2 * size - idx - 2;
}
return accum;
return idx;
}

#endif /* _FEATURE_MOTION_H_ */
Loading
Loading