From 346b384d1c6f0cb00170c1bf8ad3e8302f142237 Mon Sep 17 00:00:00 2001 From: yancheng Date: Thu, 7 Dec 2023 11:30:02 +0800 Subject: [PATCH] loongarch64: Add optimization for max. --- kernel/loongarch64/KERNEL.LOONGSON2K1000 | 3 + kernel/loongarch64/KERNEL.LOONGSON3R5 | 3 + kernel/loongarch64/dmax_lasx.S | 175 +++++++++++++++++++ kernel/loongarch64/dmax_lsx.S | 141 ++++++++++++++++ kernel/loongarch64/smax_lasx.S | 205 +++++++++++++++++++++++ kernel/loongarch64/smax_lsx.S | 171 +++++++++++++++++++ 6 files changed, 698 insertions(+) create mode 100644 kernel/loongarch64/dmax_lasx.S create mode 100644 kernel/loongarch64/dmax_lsx.S create mode 100644 kernel/loongarch64/smax_lasx.S create mode 100644 kernel/loongarch64/smax_lsx.S diff --git a/kernel/loongarch64/KERNEL.LOONGSON2K1000 b/kernel/loongarch64/KERNEL.LOONGSON2K1000 index 279ff6a9c..e00893b72 100644 --- a/kernel/loongarch64/KERNEL.LOONGSON2K1000 +++ b/kernel/loongarch64/KERNEL.LOONGSON2K1000 @@ -13,4 +13,7 @@ DAMAXKERNEL = damax_lsx.S SAMINKERNEL = samin_lsx.S DAMINKERNEL = damin_lsx.S +SMAXKERNEL = smax_lsx.S +DMAXKERNEL = dmax_lsx.S + endif diff --git a/kernel/loongarch64/KERNEL.LOONGSON3R5 b/kernel/loongarch64/KERNEL.LOONGSON3R5 index 83db79050..f238436f5 100644 --- a/kernel/loongarch64/KERNEL.LOONGSON3R5 +++ b/kernel/loongarch64/KERNEL.LOONGSON3R5 @@ -13,6 +13,9 @@ DAMAXKERNEL = damax_lasx.S SAMINKERNEL = samin_lasx.S DAMINKERNEL = damin_lasx.S +SMAXKERNEL = smax_lasx.S +DMAXKERNEL = dmax_lasx.S + DGEMMKERNEL = dgemm_kernel_16x4.S DGEMMINCOPY = dgemm_ncopy_16.S DGEMMITCOPY = dgemm_tcopy_16.S diff --git a/kernel/loongarch64/dmax_lasx.S b/kernel/loongarch64/dmax_lasx.S new file mode 100644 index 000000000..46366d2ec --- /dev/null +++ b/kernel/loongarch64/dmax_lasx.S @@ -0,0 +1,175 @@ +#define ASSEMBLER + +#include "common.h" + +#define N $r4 +#define X $r5 +#define INCX $r6 +#define I $r12 +#define J $r13 +#define t1 $r14 +#define t2 $r18 +#define t3 $r15 +#define t4 $r17 +#define TEMP $r16 +#define m0 $xr8 +#define x1 $xr9 +#define x2 $xr10 +#define x3 $xr11 +#define x4 $xr12 +#define VX0 $xr20 +#define VX1 $xr21 +#define VM0 $xr22 +#define VM1 $xr23 +#define VM2 $xr19 + + PROLOGUE + + bge $r0, N, .L999 + bge $r0, INCX, .L999 + li.d TEMP, 1 + slli.d TEMP, TEMP, BASE_SHIFT + slli.d INCX, INCX, BASE_SHIFT + bne INCX, TEMP, .L20 + xvld VM0, X, 0 + srai.d I, N, 3 + bge $r0, I, .L12 + .align 3 + +.L10: + xvld VX0, X, 0 * SIZE + xvld VX1, X, 4 * SIZE + addi.d I, I, -1 + xvfmax.d VM1, VX1, VX0 + addi.d X, X, 8 * SIZE + xvfmax.d VM0, VM0, VM1 + blt $r0, I, .L10 + .align 3 + +.L11: + xvpickve.d x1, VM0, 0 + xvpickve.d x2, VM0, 1 + xvpickve.d x3, VM0, 2 + xvpickve.d x4, VM0, 3 + xvfmax.d VM1, x1, x2 + xvfmax.d VM2, x3, x4 + xvfmax.d VM0, VM1, VM2 + .align 3 + +.L12: //INCX==1 and N<8 + andi I, N, 7 + li.d J, 4 + bge J, I, .L13 // 4