@@ -60,6 +60,13 @@ av_cold void ff_vp78dsp_init_riscv(VP8DSPContext *c)
c->put_vp8_bilinear_pixels_tab[1][0][2] = ff_put_vp8_bilin8_h_rvv;
c->put_vp8_bilinear_pixels_tab[2][0][1] = ff_put_vp8_bilin4_h_rvv;
c->put_vp8_bilinear_pixels_tab[2][0][2] = ff_put_vp8_bilin4_h_rvv;
+
+ c->put_vp8_bilinear_pixels_tab[0][1][0] = ff_put_vp8_bilin16_v_rvv;
+ c->put_vp8_bilinear_pixels_tab[0][2][0] = ff_put_vp8_bilin16_v_rvv;
+ c->put_vp8_bilinear_pixels_tab[1][1][0] = ff_put_vp8_bilin8_v_rvv;
+ c->put_vp8_bilinear_pixels_tab[1][2][0] = ff_put_vp8_bilin8_v_rvv;
+ c->put_vp8_bilinear_pixels_tab[2][1][0] = ff_put_vp8_bilin4_v_rvv;
+ c->put_vp8_bilinear_pixels_tab[2][2][0] = ff_put_vp8_bilin4_v_rvv;
}
#endif
#endif
@@ -113,6 +113,31 @@ func ff_put_vp8_bilin\len\()_h_rvv, zve32x
endfunc
.endm
+.macro put_vp8_bilin_v len
+func ff_put_vp8_bilin\len\()_v_rvv, zve32x
+ vsetvlstatic8 \len
+ li t1, 8
+ li t4, 4
+ sub t1, t1, a6
+1:
+ add t2, a2, a3
+ addi a4, a4, -1
+ vle8.v v0, (a2)
+ vle8.v v2, (t2)
+ vwmulu.vx v28, v0, t1
+ vwmaccu.vx v28, a6, v2
+ vwaddu.wx v24, v28, t4
+ vnsra.wi v0, v24, 3
+ vse8.v v0, (a0)
+ add a2, a2, a3
+ add a0, a0, a1
+ bnez a4, 1b
+
+ ret
+endfunc
+.endm
+
.irp len 16,8,4
put_vp8_bilin_h \len
+put_vp8_bilin_v \len
.endr
From: sunyuechi <sunyuechi@iscas.ac.cn> C908: vp8_put_bilin4_v_c: 383.5 vp8_put_bilin4_v_rvv_i32: 139.7 vp8_put_bilin8_v_c: 1455.7 vp8_put_bilin8_v_rvv_i32: 299.7 vp8_put_bilin16_v_c: 2863.7 vp8_put_bilin16_v_rvv_i32: 347.7 --- libavcodec/riscv/vp8dsp_init.c | 7 +++++++ libavcodec/riscv/vp8dsp_rvv.S | 25 +++++++++++++++++++++++++ 2 files changed, 32 insertions(+)