From patchwork Tue Apr 13 12:00:11 2021 Content-Type: text/plain; charset="utf-8" MIME-Version: 1.0 Content-Transfer-Encoding: 7bit X-Patchwork-Submitter: Mikhail Nitenko X-Patchwork-Id: 26892 Return-Path: X-Original-To: patchwork@ffaux-bg.ffmpeg.org Delivered-To: patchwork@ffaux-bg.ffmpeg.org Received: from ffbox0-bg.mplayerhq.hu (ffbox0-bg.ffmpeg.org [79.124.17.100]) by ffaux.localdomain (Postfix) with ESMTP id DE058448BE9 for ; Tue, 13 Apr 2021 15:00:29 +0300 (EEST) Received: from [127.0.1.1] (localhost [127.0.0.1]) by ffbox0-bg.mplayerhq.hu (Postfix) with ESMTP id BAA5C689D57; Tue, 13 Apr 2021 15:00:29 +0300 (EEST) X-Original-To: ffmpeg-devel@ffmpeg.org Delivered-To: ffmpeg-devel@ffmpeg.org Received: from mail-lf1-f44.google.com (mail-lf1-f44.google.com [209.85.167.44]) by ffbox0-bg.mplayerhq.hu (Postfix) with ESMTPS id DF8D86806EB for ; Tue, 13 Apr 2021 15:00:22 +0300 (EEST) Received: by mail-lf1-f44.google.com with SMTP id b4so26867401lfi.6 for ; Tue, 13 Apr 2021 05:00:22 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=gmail.com; s=20161025; h=from:to:cc:subject:date:message-id:in-reply-to:references :mime-version:content-transfer-encoding; bh=wfRLKENLYh8hBi8FvthWhkvPCVtvF6t2HAHpct9LqEA=; b=QzskE7HApgT8XDnjII7Y7EI8Ymg4YoW2kAfklnLhMy4hOeNuXGqWK+kwi+HeVD50kj UyX9MKfNe8ipc+puKSZVEXWbrffXiSiuVJhjrZKdMIFMkjBjOPDv807knPMr569lvVCn IauCFyRbQP0C7t9a5CMsrTQIC38XA5x7piovmpg9poRM/amOMImdNGUPSgQzx7Ii4jZ8 pbHtS04PwODNzNxCDrx5xUY0Ek/BCuYfNnRPoVv1iRxMwna7ibL5OXm/yL5icbFiBj2s xuO6QG1xol4DH/w/pTyCLjtY0LJHu3cUntQ956+gA309O+C/VVZMAeBVrklgkIlN9+ms nY/Q== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20161025; h=x-gm-message-state:from:to:cc:subject:date:message-id:in-reply-to :references:mime-version:content-transfer-encoding; bh=wfRLKENLYh8hBi8FvthWhkvPCVtvF6t2HAHpct9LqEA=; b=P52pLNtgMV2UZwr/Za5RAVZxlYW3//9otK1u56WvxxbINeH5zwOcVkCD69Nqle9/a+ i9y8tShnyDZ2CL6JKLs8oNMzAfweIwg1b29j/aqrK/ZvYfiHd16f/df9/tFR499cXgyz kZCKrZjUhQQtujuLQoKNyKNLlrMiJJzJ60tfWzSRl3Fs0pNxGQaWrN1Gz77eqK7CG+8U q6KBa06Vvjafx9gXVM8ASTSn71srk18LqQFfCFcSemWamqDdSsPNP4ErCn6/Qe+vYV9j V8dlTltyYn1zhG7znkmMrqb9hOyHVRlsX1zxS7+52c77/7nBWdHkDkWjTLciNGyhDMvB l+oA== X-Gm-Message-State: AOAM531glregqHR2w5S9G4Zdb1EfVU0+36sW2eiPuBY1XRi1+iHp0Qub ZR3/GCbLb9/Ay2C3kDv4W2oEoLwcx5ATcpaw X-Google-Smtp-Source: ABdhPJxGfDBLg2dRQc/2QGbsWuGAH8Sjp70i9Nq5xYt7mSVQBD1vJIjAZkPZZ3DB+N9gPd+zENk2CQ== X-Received: by 2002:a05:6512:2182:: with SMTP id b2mr23987047lft.284.1618315220750; Tue, 13 Apr 2021 05:00:20 -0700 (PDT) Received: from localhost.localdomain (broadband-109-173-73-178.ip.moscow.rt.ru. [109.173.73.178]) by smtp.gmail.com with ESMTPSA id g9sm1000031lfb.233.2021.04.13.05.00.20 (version=TLS1_3 cipher=TLS_AES_256_GCM_SHA384 bits=256/256); Tue, 13 Apr 2021 05:00:20 -0700 (PDT) From: Mikhail Nitenko To: ffmpeg-devel@ffmpeg.org Date: Tue, 13 Apr 2021 15:00:11 +0300 Message-Id: <20210413120012.91790-1-mnitenko@gmail.com> X-Mailer: git-send-email 2.31.1 In-Reply-To: References: MIME-Version: 1.0 Subject: [FFmpeg-devel] [PATCH v3 1/2] lavc/aarch64: change h264pred_init structure X-BeenThere: ffmpeg-devel@ffmpeg.org X-Mailman-Version: 2.1.20 Precedence: list List-Id: FFmpeg development discussions and patches List-Unsubscribe: , List-Archive: List-Post: List-Help: List-Subscribe: , Reply-To: FFmpeg development discussions and patches Cc: Mikhail Nitenko Errors-To: ffmpeg-devel-bounces@ffmpeg.org Sender: "ffmpeg-devel" Change structure to allow addition of other bit depths. --- libavcodec/aarch64/h264pred_init.c | 57 ++++++++++++++---------------- 1 file changed, 27 insertions(+), 30 deletions(-) diff --git a/libavcodec/aarch64/h264pred_init.c b/libavcodec/aarch64/h264pred_init.c index b144376f90..fc8989ae0d 100644 --- a/libavcodec/aarch64/h264pred_init.c +++ b/libavcodec/aarch64/h264pred_init.c @@ -49,38 +49,35 @@ static av_cold void h264_pred_init_neon(H264PredContext *h, int codec_id, const int bit_depth, const int chroma_format_idc) { - const int high_depth = bit_depth > 8; - - if (high_depth) - return; - - if (chroma_format_idc <= 1) { - h->pred8x8[VERT_PRED8x8 ] = ff_pred8x8_vert_neon; - h->pred8x8[HOR_PRED8x8 ] = ff_pred8x8_hor_neon; - if (codec_id != AV_CODEC_ID_VP7 && codec_id != AV_CODEC_ID_VP8) - h->pred8x8[PLANE_PRED8x8] = ff_pred8x8_plane_neon; - h->pred8x8[DC_128_PRED8x8 ] = ff_pred8x8_128_dc_neon; - if (codec_id != AV_CODEC_ID_RV40 && codec_id != AV_CODEC_ID_VP7 && - codec_id != AV_CODEC_ID_VP8) { - h->pred8x8[DC_PRED8x8 ] = ff_pred8x8_dc_neon; - h->pred8x8[LEFT_DC_PRED8x8] = ff_pred8x8_left_dc_neon; - h->pred8x8[TOP_DC_PRED8x8 ] = ff_pred8x8_top_dc_neon; - h->pred8x8[ALZHEIMER_DC_L0T_PRED8x8] = ff_pred8x8_l0t_dc_neon; - h->pred8x8[ALZHEIMER_DC_0LT_PRED8x8] = ff_pred8x8_0lt_dc_neon; - h->pred8x8[ALZHEIMER_DC_L00_PRED8x8] = ff_pred8x8_l00_dc_neon; - h->pred8x8[ALZHEIMER_DC_0L0_PRED8x8] = ff_pred8x8_0l0_dc_neon; + if (bit_depth == 8) { + if (chroma_format_idc <= 1) { + h->pred8x8[VERT_PRED8x8 ] = ff_pred8x8_vert_neon; + h->pred8x8[HOR_PRED8x8 ] = ff_pred8x8_hor_neon; + if (codec_id != AV_CODEC_ID_VP7 && codec_id != AV_CODEC_ID_VP8) + h->pred8x8[PLANE_PRED8x8] = ff_pred8x8_plane_neon; + h->pred8x8[DC_128_PRED8x8 ] = ff_pred8x8_128_dc_neon; + if (codec_id != AV_CODEC_ID_RV40 && codec_id != AV_CODEC_ID_VP7 && + codec_id != AV_CODEC_ID_VP8) { + h->pred8x8[DC_PRED8x8 ] = ff_pred8x8_dc_neon; + h->pred8x8[LEFT_DC_PRED8x8] = ff_pred8x8_left_dc_neon; + h->pred8x8[TOP_DC_PRED8x8 ] = ff_pred8x8_top_dc_neon; + h->pred8x8[ALZHEIMER_DC_L0T_PRED8x8] = ff_pred8x8_l0t_dc_neon; + h->pred8x8[ALZHEIMER_DC_0LT_PRED8x8] = ff_pred8x8_0lt_dc_neon; + h->pred8x8[ALZHEIMER_DC_L00_PRED8x8] = ff_pred8x8_l00_dc_neon; + h->pred8x8[ALZHEIMER_DC_0L0_PRED8x8] = ff_pred8x8_0l0_dc_neon; + } } - } - h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_neon; - h->pred16x16[VERT_PRED8x8 ] = ff_pred16x16_vert_neon; - h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_hor_neon; - h->pred16x16[LEFT_DC_PRED8x8] = ff_pred16x16_left_dc_neon; - h->pred16x16[TOP_DC_PRED8x8 ] = ff_pred16x16_top_dc_neon; - h->pred16x16[DC_128_PRED8x8 ] = ff_pred16x16_128_dc_neon; - if (codec_id != AV_CODEC_ID_SVQ3 && codec_id != AV_CODEC_ID_RV40 && - codec_id != AV_CODEC_ID_VP7 && codec_id != AV_CODEC_ID_VP8) - h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_plane_neon; + h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_neon; + h->pred16x16[VERT_PRED8x8 ] = ff_pred16x16_vert_neon; + h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_hor_neon; + h->pred16x16[LEFT_DC_PRED8x8] = ff_pred16x16_left_dc_neon; + h->pred16x16[TOP_DC_PRED8x8 ] = ff_pred16x16_top_dc_neon; + h->pred16x16[DC_128_PRED8x8 ] = ff_pred16x16_128_dc_neon; + if (codec_id != AV_CODEC_ID_SVQ3 && codec_id != AV_CODEC_ID_RV40 && + codec_id != AV_CODEC_ID_VP7 && codec_id != AV_CODEC_ID_VP8) + h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_plane_neon; + } } av_cold void ff_h264_pred_init_aarch64(H264PredContext *h, int codec_id, From patchwork Tue Apr 13 12:00:12 2021 Content-Type: text/plain; charset="utf-8" MIME-Version: 1.0 Content-Transfer-Encoding: 7bit X-Patchwork-Submitter: Mikhail Nitenko X-Patchwork-Id: 26893 Return-Path: X-Original-To: patchwork@ffaux-bg.ffmpeg.org Delivered-To: patchwork@ffaux-bg.ffmpeg.org Received: from ffbox0-bg.mplayerhq.hu (ffbox0-bg.ffmpeg.org [79.124.17.100]) by ffaux.localdomain (Postfix) with ESMTP id C7A30448BE9 for ; Tue, 13 Apr 2021 15:00:35 +0300 (EEST) Received: from [127.0.1.1] (localhost [127.0.0.1]) by ffbox0-bg.mplayerhq.hu (Postfix) with ESMTP id ADF1A689DC3; Tue, 13 Apr 2021 15:00:35 +0300 (EEST) X-Original-To: ffmpeg-devel@ffmpeg.org Delivered-To: ffmpeg-devel@ffmpeg.org Received: from mail-lf1-f42.google.com (mail-lf1-f42.google.com [209.85.167.42]) by ffbox0-bg.mplayerhq.hu (Postfix) with ESMTPS id F0DC1689BEC for ; Tue, 13 Apr 2021 15:00:28 +0300 (EEST) Received: by mail-lf1-f42.google.com with SMTP id g8so26877858lfv.12 for ; Tue, 13 Apr 2021 05:00:28 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=gmail.com; s=20161025; h=from:to:cc:subject:date:message-id:in-reply-to:references :mime-version:content-transfer-encoding; bh=6cFe0+e/9hI+kz9uCyFPQYVeF2gW4L4DRZMVbQO66jg=; b=WIpHgMmIGu5ZWq1pSg5v9+8xHlLBQxFMvPAQok2VCXyYn5K5l6I73VA8A2vfowKAPf odCrpZEv9TFOIjKA75Vfikq8bn721nJFe1HSYeZotzupyVUTUALR1zhNeSSObE4bGKeU yt3NLX4XbK/FlFQ+q0ImyqRieKv7ibNKMwdvsunXG/VDPx6+KB8phDonlmr6TaS+sfut mp0K9omJOjI8c8XtTweq/be50kEJ91676pZc7mE4oJgaxEJJ3yxJLBzNfCvLBAKm5XNK 7IXQFLR2TZTDRy2edyteOurQm6GOVHuTtYEexioUGI+aFImXUVN13psYGikfwTJNTsKw nuCw== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20161025; h=x-gm-message-state:from:to:cc:subject:date:message-id:in-reply-to :references:mime-version:content-transfer-encoding; bh=6cFe0+e/9hI+kz9uCyFPQYVeF2gW4L4DRZMVbQO66jg=; b=eP5SjFRcj9fuHoSriq0FAahLt4pnYP27thhIq9I3XZ3FQESjHUswjc6vBWMZj8uQmY HbMnTsqqDot8pGBHhMqs2ecU4z4Jq3A4UhbSpkJCtBoQXmZmGvhe48Jdd6c/hLlfwUiT ymH10ouHkfTi8LUiaIU5UaEedfLM6KKZYxLIcFb0jXqdae+AovjFeykinnxxSkmvcuD9 WaA158loMOfJhjl/fa++2IXyqavKyUTdZQVbNi+beNJglMiLeLVz04GMNO6rgDRysnlJ 3R/yw3P14lBdyycOH+Aon0+4sAe90Ul3QLs4EVcgt6rR+uQeVEEV3QjkzBupX9tOTkCT FzSQ== X-Gm-Message-State: AOAM530WQSaey/BtE1sATcaA9zk2vK2XJ7wRpsjKDT2dJEKRL3YN3yM3 D0B96tnAvwE0h75F7y6MbdVnVt7l+gJb4uoS X-Google-Smtp-Source: ABdhPJw5qB7hcmMpBRq54vL602IQkCwKgxbDMmHV0f+6eCGqjgdOokFijeHsIP+GHTetvAL8MgcYxQ== X-Received: by 2002:a05:6512:a86:: with SMTP id m6mr23271834lfu.314.1618315227956; Tue, 13 Apr 2021 05:00:27 -0700 (PDT) Received: from localhost.localdomain (broadband-109-173-73-178.ip.moscow.rt.ru. [109.173.73.178]) by smtp.gmail.com with ESMTPSA id g9sm1000031lfb.233.2021.04.13.05.00.27 (version=TLS1_3 cipher=TLS_AES_256_GCM_SHA384 bits=256/256); Tue, 13 Apr 2021 05:00:27 -0700 (PDT) From: Mikhail Nitenko To: ffmpeg-devel@ffmpeg.org Date: Tue, 13 Apr 2021 15:00:12 +0300 Message-Id: <20210413120012.91790-2-mnitenko@gmail.com> X-Mailer: git-send-email 2.31.1 In-Reply-To: <20210413120012.91790-1-mnitenko@gmail.com> References: <20210413120012.91790-1-mnitenko@gmail.com> MIME-Version: 1.0 Subject: [FFmpeg-devel] [PATCH v3 2/2] lavc/aarch64: add pred16x16 10-bit functions X-BeenThere: ffmpeg-devel@ffmpeg.org X-Mailman-Version: 2.1.20 Precedence: list List-Id: FFmpeg development discussions and patches List-Unsubscribe: , List-Archive: List-Post: List-Help: List-Subscribe: , Reply-To: FFmpeg development discussions and patches Cc: Mikhail Nitenko Errors-To: ffmpeg-devel-bounces@ffmpeg.org Sender: "ffmpeg-devel" Benchmarks: pred16x16_dc_10_c: 124.0 pred16x16_dc_10_neon: 97.2 pred16x16_horizontal_10_c: 71.7 pred16x16_horizontal_10_neon: 66.2 pred16x16_top_dc_10_c: 90.7 pred16x16_top_dc_10_neon: 71.5 pred16x16_vertical_10_c: 64.7 pred16x16_vertical_10_neon: 61.7 Some functions work slower than C and are left commented out. --- libavcodec/aarch64/h264pred_init.c | 12 +++ libavcodec/aarch64/h264pred_neon.S | 117 +++++++++++++++++++++++++++++ 2 files changed, 129 insertions(+) diff --git a/libavcodec/aarch64/h264pred_init.c b/libavcodec/aarch64/h264pred_init.c index fc8989ae0d..9a1f13910d 100644 --- a/libavcodec/aarch64/h264pred_init.c +++ b/libavcodec/aarch64/h264pred_init.c @@ -45,6 +45,11 @@ void ff_pred8x8_0lt_dc_neon(uint8_t *src, ptrdiff_t stride); void ff_pred8x8_l00_dc_neon(uint8_t *src, ptrdiff_t stride); void ff_pred8x8_0l0_dc_neon(uint8_t *src, ptrdiff_t stride); +void ff_pred16x16_top_dc_neon_10(uint8_t *src, ptrdiff_t stride); +void ff_pred16x16_dc_neon_10(uint8_t *src, ptrdiff_t stride); +void ff_pred16x16_hor_neon_10(uint8_t *src, ptrdiff_t stride); +void ff_pred16x16_vert_neon_10(uint8_t *src, ptrdiff_t stride); + static av_cold void h264_pred_init_neon(H264PredContext *h, int codec_id, const int bit_depth, const int chroma_format_idc) @@ -78,6 +83,12 @@ static av_cold void h264_pred_init_neon(H264PredContext *h, int codec_id, codec_id != AV_CODEC_ID_VP7 && codec_id != AV_CODEC_ID_VP8) h->pred16x16[PLANE_PRED8x8 ] = ff_pred16x16_plane_neon; } + if (bit_depth == 10) { + h->pred16x16[DC_PRED8x8 ] = ff_pred16x16_dc_neon_10; + h->pred16x16[VERT_PRED8x8 ] = ff_pred16x16_vert_neon_10; + h->pred16x16[HOR_PRED8x8 ] = ff_pred16x16_hor_neon_10; + h->pred16x16[TOP_DC_PRED8x8 ] = ff_pred16x16_top_dc_neon_10; + } } av_cold void ff_h264_pred_init_aarch64(H264PredContext *h, int codec_id, @@ -88,3 +99,4 @@ av_cold void ff_h264_pred_init_aarch64(H264PredContext *h, int codec_id, if (have_neon(cpu_flags)) h264_pred_init_neon(h, codec_id, bit_depth, chroma_format_idc); } + diff --git a/libavcodec/aarch64/h264pred_neon.S b/libavcodec/aarch64/h264pred_neon.S index 213b40b3e7..5ce50323f8 100644 --- a/libavcodec/aarch64/h264pred_neon.S +++ b/libavcodec/aarch64/h264pred_neon.S @@ -359,3 +359,120 @@ function ff_pred8x8_0l0_dc_neon, export=1 dup v1.8b, v1.b[0] b .L_pred8x8_dc_end endfunc + +.macro ldcol.16 rd, rs, rt, n=4, hi=0 +.if \n >= 4 || \hi == 0 + ld1 {\rd\().h}[0], [\rs], \rt + ld1 {\rd\().h}[1], [\rs], \rt +.endif +.if \n >= 4 || \hi == 1 + ld1 {\rd\().h}[2], [\rs], \rt + ld1 {\rd\().h}[3], [\rs], \rt +.endif +.if \n == 8 + ld1 {\rd\().h}[4], [\rs], \rt + ld1 {\rd\().h}[5], [\rs], \rt + ld1 {\rd\().h}[6], [\rs], \rt + ld1 {\rd\().h}[7], [\rs], \rt +.endif +.endm + +// slower than C +/* +function ff_pred16x16_128_dc_neon_10, export=1 + movi v0.8h, #2, lsl #8 // 512, 1 << (bit_depth - 1) + + b .L_pred16x16_dc_10_end +endfunc +*/ + +function ff_pred16x16_top_dc_neon_10, export=1 + sub x2, x0, x1 + + ld1 {v0.8h, v1.8h}, [x2] + + addv h0, v0.8h + addv h1, v1.8h + + add v0.4h, v0.4h, v1.4h + + rshrn v0.4h, v0.4s, #4 + dup v0.8h, v0.h[0] + b .L_pred16x16_dc_10_end +endfunc + +// slower than C +/* +function ff_pred16x16_left_dc_neon_10, export=1 + sub x2, x0, #2 // access to the "left" column + ldcol.16 v0, x2, x1, 8 + ldcol.16 v1, x2, x1, 8 // load "left" column + + addv h0, v0.8h + addv h1, v1.8h + + add v0.4h, v0.4h, v1.4h + + rshrn v0.4h, v0.4s, #4 + dup v0.8h, v0.h[0] + b .L_pred16x16_dc_10_end +endfunc +*/ + +function ff_pred16x16_dc_neon_10, export=1 + sub x2, x0, x1 // access to the "top" row + sub x3, x0, #2 // access to the "left" column + + ld1 {v0.8h, v1.8h}, [x2] + ldcol.16 v2, x3, x1, 8 + ldcol.16 v3, x3, x1, 8 // load pixels in "top" row and "left" col + + addv h0, v0.8h + addv h1, v1.8h + addv h2, v2.8h // sum all pixels in the "top" row and "left" col + addv h3, v3.8h // (sum stays in h0-h3 registers) + + add v0.4h, v0.4h, v1.4h + add v0.4h, v0.4h, v2.4h + add v0.4h, v0.4h, v3.4h // sum registers h0-h3 + + rshrn v0.4h, v0.4s, #5 + dup v0.8h, v0.h[0] +.L_pred16x16_dc_10_end: + mov v1.16b, v0.16b + mov w3, #8 +6: st1 {v0.8h, v1.8h}, [x0], x1 + subs w3, w3, #1 + st1 {v0.8h, v1.8h}, [x0], x1 + b.ne 6b + ret +endfunc + +function ff_pred16x16_hor_neon_10, export=1 + sub x2, x0, #2 + add x3, x0, #16 + + mov w4, #16 +1: ld1r {v0.8h}, [x2], x1 + subs w4, w4, #1 + st1 {v0.8h}, [x0], x1 + st1 {v0.8h}, [x3], x1 + b.ne 1b + ret +endfunc + +function ff_pred16x16_vert_neon_10, export=1 + sub x2, x0, x1 + add x1, x1, x1 + + ld1 {v0.8h, v1.8h}, [x2], x1 + + mov w3, #8 +1: subs w3, w3, #1 + st1 {v0.8h, v1.8h}, [x0], x1 + st1 {v0.8h, v1.8h}, [x2], x1 + + b.ne 1b + ret +endfunc +