blob: 666b0015fd448535be0ec1be9c06edaf8f941f96 [file]
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc < %s -mtriple=x86_64-- -mattr=+gfni,+avx2 | FileCheck %s
declare <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8>, <16 x i8>, i8)
define <16 x i8> @test_const_splat_on_const_matrix(<16 x i8> %src) nounwind {
; CHECK-LABEL: test_const_splat_on_const_matrix:
; CHECK: # %bb.0:
; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0]
; CHECK-NEXT: retq
%and = and <16 x i8> %src, splat(i8 15)
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128>, i8 0)
ret <16 x i8> %gfni
}
define <32 x i8> @test_const_splat_on_const_matrix256(<32 x i8> %src) nounwind {
; CHECK-LABEL: test_const_splat_on_const_matrix256:
; CHECK: # %bb.0:
; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm0, %ymm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0]
; CHECK-NEXT: retq
%and = and <32 x i8> %src, splat(i8 15)
%gfni = call <32 x i8> @llvm.x86.vgf2p8affineqb.256(<32 x i8> %and, <32 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128>, i8 0)
ret <32 x i8> %gfni
}
;; Same situation as above but with a diffrent matrix
define <16 x i8> @test_const_splat_alternative_const_matrix(<16 x i8> %src) nounwind {
; CHECK-LABEL: test_const_splat_alternative_const_matrix:
; CHECK: # %bb.0:
; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [1,1,5,5,21,21,85,85,1,1,5,5,21,21,85,85]
; CHECK-NEXT: retq
%and = and <16 x i8> %src, splat(i8 85)
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 1, i8 3, i8 7, i8 15, i8 31, i8 63, i8 127, i8 255, i8 1, i8 3, i8 7, i8 15, i8 31, i8 63, i8 127, i8 255>, i8 0)
ret <16 x i8> %gfni
}
define <16 x i8> @test_const_splat_on_const_matrix_nonzero_imm(<16 x i8> %src) nounwind {
; CHECK-LABEL: test_const_splat_on_const_matrix_nonzero_imm:
; CHECK: # %bb.0:
; CHECK-NEXT: vgf2p8affineqb $127, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0]
; CHECK-NEXT: retq
%and = and <16 x i8> %src, splat(i8 15)
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128>, i8 127)
ret <16 x i8> %gfni
}
define <16 x i8> @test_var_splat_on_const_matrix(<16 x i8> %src, i8 %scalar) nounwind {
; CHECK-LABEL: test_var_splat_on_const_matrix:
; CHECK: # %bb.0:
; CHECK-NEXT: vmovd %edi, %xmm1
; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1
; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm1, %xmm1
; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0
; CHECK-NEXT: retq
%inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0
%varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer
%and = and <16 x i8> %src, %varsplat
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> splat(i8 2), i8 0)
ret <16 x i8> %gfni
}
define <16 x i8> @test_var_splat_parity_fill_matrix(<16 x i8> %src, i8 %scalar) nounwind {
; CHECK-LABEL: test_var_splat_parity_fill_matrix:
; CHECK: # %bb.0:
; CHECK-NEXT: vmovd %edi, %xmm1
; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1
; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0
; CHECK-NEXT: retq
%inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0
%varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer
%and = and <16 x i8> %src, %varsplat
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> splat(i8 -1), i8 0)
ret <16 x i8> %gfni
}
define <16 x i8> @test_var_splat_on_const_matrix_nonzero_imm(<16 x i8> %src, i8 %scalar) nounwind {
; CHECK-LABEL: test_var_splat_on_const_matrix_nonzero_imm:
; CHECK: # %bb.0:
; CHECK-NEXT: vmovd %edi, %xmm1
; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1
; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm1, %xmm1
; CHECK-NEXT: vgf2p8affineqb $170, %xmm1, %xmm0, %xmm0
; CHECK-NEXT: retq
%inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0
%varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer
%and = and <16 x i8> %src, %varsplat
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> splat(i8 2), i8 170)
ret <16 x i8> %gfni
}
;; Multi use still shortens dependency chain
define <16 x i8> @test_const_splat_on_const_matrix_multi_use(<16 x i8> %src, ptr %sink) nounwind {
; CHECK-LABEL: test_const_splat_on_const_matrix_multi_use:
; CHECK: # %bb.0:
; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm1
; CHECK-NEXT: vmovdqa %xmm1, (%rdi)
; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0]
; CHECK-NEXT: retq
%and = and <16 x i8> %src, splat(i8 15)
store <16 x i8> %and, ptr %sink
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0)
ret <16 x i8> %gfni
}
;; Negative test: multi use would generate an additional AND
define <16 x i8> @test_var_splat_on_const_matrix_multi_use(<16 x i8> %src, i8 %scalar, ptr %sink) nounwind {
; CHECK-LABEL: test_var_splat_on_const_matrix_multi_use:
; CHECK: # %bb.0:
; CHECK-NEXT: vmovd %edi, %xmm1
; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1
; CHECK-NEXT: vpand %xmm1, %xmm0, %xmm1
; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm1, %xmm0 # [64,32,16,8,4,2,1,128,64,32,16,8,4,2,1,128]
; CHECK-NEXT: vmovdqa %xmm1, (%rsi)
; CHECK-NEXT: retq
%inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0
%varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer
%and = and <16 x i8> %src, %varsplat
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0)
store <16 x i8> %and, ptr %sink
ret <16 x i8> %gfni
}
;; Negative test: don't fold as it may increase dependency chain
define <16 x i8> @test_const_splat_on_var_operands(<16 x i8> %src, <16 x i8> %matrix) nounwind {
; CHECK-LABEL: test_const_splat_on_var_operands:
; CHECK: # %bb.0:
; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0
; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0
; CHECK-NEXT: retq
%and = and <16 x i8> %src, splat(i8 1)
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> %matrix, i8 0)
ret <16 x i8> %gfni
}
;; Negative test: don't fold as it may increase dependency chain
define <16 x i8> @test_var_splat_on_var_operands(<16 x i8> %src, <16 x i8> %matrix, i8 %scalar) nounwind {
; CHECK-LABEL: test_var_splat_on_var_operands:
; CHECK: # %bb.0:
; CHECK-NEXT: vmovd %edi, %xmm2
; CHECK-NEXT: vpbroadcastb %xmm2, %xmm2
; CHECK-NEXT: vpand %xmm2, %xmm0, %xmm0
; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0
; CHECK-NEXT: retq
%inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0
%varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer
%and = and <16 x i8> %src, %varsplat
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> %matrix, i8 0)
ret <16 x i8> %gfni
}
;; Negative test: larger splats should not fold
define <16 x i8> @test_const_splat16_on_const_matrix(<16 x i8> %src) nounwind {
; CHECK-LABEL: test_const_splat16_on_const_matrix:
; CHECK: # %bb.0:
; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0
; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [64,32,16,8,4,2,1,128,64,32,16,8,4,2,1,128]
; CHECK-NEXT: retq
%splat = bitcast <8 x i16> splat(i16 1) to <16 x i8>
%and = and <16 x i8> %src, %splat
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0)
ret <16 x i8> %gfni
}
;; Negative test: larger splats should not fold
define <16 x i8> @test_var_splat16_on_const_matrix(<16 x i8> %src, i16 %scalar) nounwind {
; CHECK-LABEL: test_var_splat16_on_const_matrix:
; CHECK: # %bb.0:
; CHECK-NEXT: vmovd %edi, %xmm1
; CHECK-NEXT: vpbroadcastw %xmm1, %xmm1
; CHECK-NEXT: vpand %xmm1, %xmm0, %xmm0
; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [64,32,16,8,4,2,1,128,64,32,16,8,4,2,1,128]
; CHECK-NEXT: retq
%inlo = insertelement <8 x i16> poison, i16 %scalar, i64 0
%varsplat16 = shufflevector <8 x i16> %inlo, <8 x i16> poison, <8 x i32> zeroinitializer
%varsplat = bitcast <8 x i16> %varsplat16 to <16 x i8>
%and = and <16 x i8> %src, %varsplat
%gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0)
ret <16 x i8> %gfni
}