From patchwork Sat Mar 10 15:22:03 2018
Content-Type: text/plain; charset="utf-8"
MIME-Version: 1.0
Content-Transfer-Encoding: 7bit
X-Patchwork-Submitter: Ard Biesheuvel <ard.biesheuvel@linaro.org>
X-Patchwork-Id: 10273635
X-Patchwork-Delegate: herbert@gondor.apana.org.au
Return-Path: <linux-crypto-owner@kernel.org>
Received: from mail.wl.linuxfoundation.org (pdx-wl-mail.web.codeaurora.org
	[172.30.200.125])
	by pdx-korg-patchwork.web.codeaurora.org (Postfix) with ESMTP id
	CF1DF602BD for <patchwork-linux-crypto@patchwork.kernel.org>;
	Sat, 10 Mar 2018 15:23:30 +0000 (UTC)
Received: from mail.wl.linuxfoundation.org (localhost [127.0.0.1])
	by mail.wl.linuxfoundation.org (Postfix) with ESMTP id BF099294FE
	for <patchwork-linux-crypto@patchwork.kernel.org>;
	Sat, 10 Mar 2018 15:23:30 +0000 (UTC)
Received: by mail.wl.linuxfoundation.org (Postfix, from userid 486)
	id B22C229577; Sat, 10 Mar 2018 15:23:30 +0000 (UTC)
X-Spam-Checker-Version: SpamAssassin 3.3.1 (2010-03-16) on
	pdx-wl-mail.web.codeaurora.org
X-Spam-Level: 
X-Spam-Status: No, score=-7.0 required=2.0 tests=BAYES_00,DKIM_SIGNED,
	DKIM_VALID, DKIM_VALID_AU,
	RCVD_IN_DNSWL_HI autolearn=ham version=3.3.1
Received: from vger.kernel.org (vger.kernel.org [209.132.180.67])
	by mail.wl.linuxfoundation.org (Postfix) with ESMTP id 4CF4A294FE
	for <patchwork-linux-crypto@patchwork.kernel.org>;
	Sat, 10 Mar 2018 15:23:30 +0000 (UTC)
Received: (majordomo@vger.kernel.org) by vger.kernel.org via listexpand
	id S932365AbeCJPX1 (ORCPT
	<rfc822;patchwork-linux-crypto@patchwork.kernel.org>);
	Sat, 10 Mar 2018 10:23:27 -0500
Received: from mail-wm0-f66.google.com ([74.125.82.66]:54585 "EHLO
	mail-wm0-f66.google.com" rhost-flags-OK-OK-OK-OK) by vger.kernel.org
	with ESMTP id S932359AbeCJPXY (ORCPT
	<rfc822;linux-crypto@vger.kernel.org>);
	Sat, 10 Mar 2018 10:23:24 -0500
Received: by mail-wm0-f66.google.com with SMTP id z81so8854718wmb.4
	for <linux-crypto@vger.kernel.org>;
	Sat, 10 Mar 2018 07:23:24 -0800 (PST)
DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linaro.org; s=google;
	h=from:to:cc:subject:date:message-id:in-reply-to:references;
	bh=/m7DRDk722apMGW0nG/T2Et7TigRr637a0J0fTnO3LM=;
	b=BP37ZWGOAxd4CxCz4/MEMVsa363tbICpcGqYIV+YjziWEAgRkhZOUGHPU9mlSngTIU
	9+TCSKDIIUjpr4R9393lhzMghJfd0BvqDyUiX/Fi3c/ldF9uTqqVF9359CLN0GAHzVxO
	ocn3tx/ysyGE01XSvelPzvwX08mF1m9DfP6hI=
X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed;
	d=1e100.net; s=20161025;
	h=x-gm-message-state:from:to:cc:subject:date:message-id:in-reply-to
	:references;
	bh=/m7DRDk722apMGW0nG/T2Et7TigRr637a0J0fTnO3LM=;
	b=EM84w15TwfFa9QZSmrO0edcQT4W2yrD8WcOgLN0oFGtOAomoHp18xRHVrJHMlBfcLN
	xmSTNywpw5pE6tBtcaH4/pZ8shcPfp/DougmsGFRXdRTD+gqz4+PA59Tj5AtxBqgUD+d
	8DbRb4VXZNBeOsKakp0TH1Ix5DL6VKFXMgV/Uc0OshNDq12cRG3LivHr68sWbrpazEqp
	es/ZwyEreqMVKwnFlPpcvJkpsK/Lqe6fOvdxm8o1p6F5Z1hUiPk8cuOF1wiYYuEObOi6
	NEMdOhLGYp7/iGQ4DmMlA3Y0G6GP5xcp6LWvvAMrMjlZBcmKkYEbudYDiQrzvxVtI1ja
	jSxw==
X-Gm-Message-State: AElRT7ExlnoEBKhEjRkvO4vV0bU4iDJ+i/+FxANWhyGbcAIj+bX0DFp4
	p12dvDbV2UyTjrDlt12Iid5UaG++N6o=
X-Google-Smtp-Source: 
 AG47ELs3gb+Tz1ITx2vrOME6uUbIVlThIxW5UZgSxA6MiUmZk4fx4xzMJ1hslgF6kFMkjCyem9Eg3A==
X-Received: by 10.28.193.134 with SMTP id r128mr1328900wmf.85.1520695403388;
	Sat, 10 Mar 2018 07:23:23 -0800 (PST)
Received: from localhost.localdomain ([105.148.128.186])
	by smtp.gmail.com with ESMTPSA id
	m9sm7027531wrf.13.2018.03.10.07.23.20
	(version=TLS1_2 cipher=ECDHE-RSA-AES128-GCM-SHA256 bits=128/128);
	Sat, 10 Mar 2018 07:23:22 -0800 (PST)
From: Ard Biesheuvel <ard.biesheuvel@linaro.org>
To: linux-crypto@vger.kernel.org
Cc: herbert@gondor.apana.org.au, linux-arm-kernel@lists.infradead.org,
	Ard Biesheuvel <ard.biesheuvel@linaro.org>,
	Dave Martin <Dave.Martin@arm.com>,
	Russell King - ARM Linux <linux@armlinux.org.uk>,
	Sebastian Andrzej Siewior <bigeasy@linutronix.de>,
	Mark Rutland <mark.rutland@arm.com>, linux-rt-users@vger.kernel.org,
	Peter Zijlstra <peterz@infradead.org>,
	Catalin Marinas <catalin.marinas@arm.com>,
	Will Deacon <will.deacon@arm.com>, Steven Rostedt <rostedt@goodmis.org>,
	Thomas Gleixner <tglx@linutronix.de>
Subject: [PATCH v5 18/23] crypto: arm64/crc32-ce - yield NEON after every
	block of input
Date: Sat, 10 Mar 2018 15:22:03 +0000
Message-Id: <20180310152208.10369-19-ard.biesheuvel@linaro.org>
X-Mailer: git-send-email 2.15.1
In-Reply-To: <20180310152208.10369-1-ard.biesheuvel@linaro.org>
References: <20180310152208.10369-1-ard.biesheuvel@linaro.org>
Sender: linux-crypto-owner@vger.kernel.org
Precedence: bulk
List-ID: <linux-crypto.vger.kernel.org>
X-Mailing-List: linux-crypto@vger.kernel.org
X-Virus-Scanned: ClamAV using ClamSMTP

Avoid excessive scheduling delays under a preemptible kernel by
conditionally yielding the NEON after every block of input.

Signed-off-by: Ard Biesheuvel <ard.biesheuvel@linaro.org>
---
 arch/arm64/crypto/crc32-ce-core.S | 40 +++++++++++++++-----
 1 file changed, 30 insertions(+), 10 deletions(-)

diff --git a/arch/arm64/crypto/crc32-ce-core.S b/arch/arm64/crypto/crc32-ce-core.S
index 16ed3c7ebd37..8061bf0f9c66 100644
--- a/arch/arm64/crypto/crc32-ce-core.S
+++ b/arch/arm64/crypto/crc32-ce-core.S
@@ -100,9 +100,10 @@
 	dCONSTANT	.req	d0
 	qCONSTANT	.req	q0
 
-	BUF		.req	x0
-	LEN		.req	x1
-	CRC		.req	x2
+	BUF		.req	x19
+	LEN		.req	x20
+	CRC		.req	x21
+	CONST		.req	x22
 
 	vzr		.req	v9
 
@@ -123,7 +124,14 @@ ENTRY(crc32_pmull_le)
 ENTRY(crc32c_pmull_le)
 	adr_l		x3, .Lcrc32c_constants
 
-0:	bic		LEN, LEN, #15
+0:	frame_push	4, 64
+
+	mov		BUF, x0
+	mov		LEN, x1
+	mov		CRC, x2
+	mov		CONST, x3
+
+	bic		LEN, LEN, #15
 	ld1		{v1.16b-v4.16b}, [BUF], #0x40
 	movi		vzr.16b, #0
 	fmov		dCONSTANT, CRC
@@ -132,7 +140,7 @@ ENTRY(crc32c_pmull_le)
 	cmp		LEN, #0x40
 	b.lt		less_64
 
-	ldr		qCONSTANT, [x3]
+	ldr		qCONSTANT, [CONST]
 
 loop_64:		/* 64 bytes Full cache line folding */
 	sub		LEN, LEN, #0x40
@@ -162,10 +170,21 @@ loop_64:		/* 64 bytes Full cache line folding */
 	eor		v4.16b, v4.16b, v8.16b
 
 	cmp		LEN, #0x40
-	b.ge		loop_64
+	b.lt		less_64
+
+	if_will_cond_yield_neon
+	stp		q1, q2, [sp, #.Lframe_local_offset]
+	stp		q3, q4, [sp, #.Lframe_local_offset + 32]
+	do_cond_yield_neon
+	ldp		q1, q2, [sp, #.Lframe_local_offset]
+	ldp		q3, q4, [sp, #.Lframe_local_offset + 32]
+	ldr		qCONSTANT, [CONST]
+	movi		vzr.16b, #0
+	endif_yield_neon
+	b		loop_64
 
 less_64:		/* Folding cache line into 128bit */
-	ldr		qCONSTANT, [x3, #16]
+	ldr		qCONSTANT, [CONST, #16]
 
 	pmull2		v5.1q, v1.2d, vCONSTANT.2d
 	pmull		v1.1q, v1.1d, vCONSTANT.1d
@@ -204,8 +223,8 @@ fold_64:
 	eor		v1.16b, v1.16b, v2.16b
 
 	/* final 32-bit fold */
-	ldr		dCONSTANT, [x3, #32]
-	ldr		d3, [x3, #40]
+	ldr		dCONSTANT, [CONST, #32]
+	ldr		d3, [CONST, #40]
 
 	ext		v2.16b, v1.16b, vzr.16b, #4
 	and		v1.16b, v1.16b, v3.16b
@@ -213,7 +232,7 @@ fold_64:
 	eor		v1.16b, v1.16b, v2.16b
 
 	/* Finish up with the bit-reversed barrett reduction 64 ==> 32 bits */
-	ldr		qCONSTANT, [x3, #48]
+	ldr		qCONSTANT, [CONST, #48]
 
 	and		v2.16b, v1.16b, v3.16b
 	ext		v2.16b, vzr.16b, v2.16b, #8
@@ -223,6 +242,7 @@ fold_64:
 	eor		v1.16b, v1.16b, v2.16b
 	mov		w0, v1.s[1]
 
+	frame_pop
 	ret
 ENDPROC(crc32_pmull_le)
 ENDPROC(crc32c_pmull_le)