| ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6 |
| ; RUN: llc < %s -mtriple=x86_64-- -mattr=+gfni,+avx2 | FileCheck %s |
| |
| declare <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8>, <16 x i8>, i8) |
| |
| define <16 x i8> @test_const_splat_on_const_matrix(<16 x i8> %src) nounwind { |
| ; CHECK-LABEL: test_const_splat_on_const_matrix: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0] |
| ; CHECK-NEXT: retq |
| %and = and <16 x i8> %src, splat(i8 15) |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128>, i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| define <32 x i8> @test_const_splat_on_const_matrix256(<32 x i8> %src) nounwind { |
| ; CHECK-LABEL: test_const_splat_on_const_matrix256: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm0, %ymm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0] |
| ; CHECK-NEXT: retq |
| %and = and <32 x i8> %src, splat(i8 15) |
| %gfni = call <32 x i8> @llvm.x86.vgf2p8affineqb.256(<32 x i8> %and, <32 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128>, i8 0) |
| ret <32 x i8> %gfni |
| } |
| |
| ;; Same situation as above but with a diffrent matrix |
| define <16 x i8> @test_const_splat_alternative_const_matrix(<16 x i8> %src) nounwind { |
| ; CHECK-LABEL: test_const_splat_alternative_const_matrix: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [1,1,5,5,21,21,85,85,1,1,5,5,21,21,85,85] |
| ; CHECK-NEXT: retq |
| %and = and <16 x i8> %src, splat(i8 85) |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 1, i8 3, i8 7, i8 15, i8 31, i8 63, i8 127, i8 255, i8 1, i8 3, i8 7, i8 15, i8 31, i8 63, i8 127, i8 255>, i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| define <16 x i8> @test_const_splat_on_const_matrix_nonzero_imm(<16 x i8> %src) nounwind { |
| ; CHECK-LABEL: test_const_splat_on_const_matrix_nonzero_imm: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vgf2p8affineqb $127, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0] |
| ; CHECK-NEXT: retq |
| %and = and <16 x i8> %src, splat(i8 15) |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 -128>, i8 127) |
| ret <16 x i8> %gfni |
| } |
| |
| define <16 x i8> @test_var_splat_on_const_matrix(<16 x i8> %src, i8 %scalar) nounwind { |
| ; CHECK-LABEL: test_var_splat_on_const_matrix: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vmovd %edi, %xmm1 |
| ; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1 |
| ; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm1, %xmm1 |
| ; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0 |
| ; CHECK-NEXT: retq |
| %inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0 |
| %varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer |
| %and = and <16 x i8> %src, %varsplat |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> splat(i8 2), i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| define <16 x i8> @test_var_splat_parity_fill_matrix(<16 x i8> %src, i8 %scalar) nounwind { |
| ; CHECK-LABEL: test_var_splat_parity_fill_matrix: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vmovd %edi, %xmm1 |
| ; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1 |
| ; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0 |
| ; CHECK-NEXT: retq |
| %inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0 |
| %varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer |
| %and = and <16 x i8> %src, %varsplat |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> splat(i8 -1), i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| define <16 x i8> @test_var_splat_on_const_matrix_nonzero_imm(<16 x i8> %src, i8 %scalar) nounwind { |
| ; CHECK-LABEL: test_var_splat_on_const_matrix_nonzero_imm: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vmovd %edi, %xmm1 |
| ; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1 |
| ; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm1, %xmm1 |
| ; CHECK-NEXT: vgf2p8affineqb $170, %xmm1, %xmm0, %xmm0 |
| ; CHECK-NEXT: retq |
| %inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0 |
| %varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer |
| %and = and <16 x i8> %src, %varsplat |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> splat(i8 2), i8 170) |
| ret <16 x i8> %gfni |
| } |
| |
| ;; Multi use still shortens dependency chain |
| define <16 x i8> @test_const_splat_on_const_matrix_multi_use(<16 x i8> %src, ptr %sink) nounwind { |
| ; CHECK-LABEL: test_const_splat_on_const_matrix_multi_use: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm1 |
| ; CHECK-NEXT: vmovdqa %xmm1, (%rdi) |
| ; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [0,0,0,8,4,2,1,0,0,0,0,8,4,2,1,0] |
| ; CHECK-NEXT: retq |
| %and = and <16 x i8> %src, splat(i8 15) |
| store <16 x i8> %and, ptr %sink |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| ;; Negative test: multi use would generate an additional AND |
| define <16 x i8> @test_var_splat_on_const_matrix_multi_use(<16 x i8> %src, i8 %scalar, ptr %sink) nounwind { |
| ; CHECK-LABEL: test_var_splat_on_const_matrix_multi_use: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vmovd %edi, %xmm1 |
| ; CHECK-NEXT: vpbroadcastb %xmm1, %xmm1 |
| ; CHECK-NEXT: vpand %xmm1, %xmm0, %xmm1 |
| ; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm1, %xmm0 # [64,32,16,8,4,2,1,128,64,32,16,8,4,2,1,128] |
| ; CHECK-NEXT: vmovdqa %xmm1, (%rsi) |
| ; CHECK-NEXT: retq |
| %inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0 |
| %varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer |
| %and = and <16 x i8> %src, %varsplat |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0) |
| store <16 x i8> %and, ptr %sink |
| ret <16 x i8> %gfni |
| } |
| |
| ;; Negative test: don't fold as it may increase dependency chain |
| define <16 x i8> @test_const_splat_on_var_operands(<16 x i8> %src, <16 x i8> %matrix) nounwind { |
| ; CHECK-LABEL: test_const_splat_on_var_operands: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 |
| ; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0 |
| ; CHECK-NEXT: retq |
| %and = and <16 x i8> %src, splat(i8 1) |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> %matrix, i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| ;; Negative test: don't fold as it may increase dependency chain |
| define <16 x i8> @test_var_splat_on_var_operands(<16 x i8> %src, <16 x i8> %matrix, i8 %scalar) nounwind { |
| ; CHECK-LABEL: test_var_splat_on_var_operands: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vmovd %edi, %xmm2 |
| ; CHECK-NEXT: vpbroadcastb %xmm2, %xmm2 |
| ; CHECK-NEXT: vpand %xmm2, %xmm0, %xmm0 |
| ; CHECK-NEXT: vgf2p8affineqb $0, %xmm1, %xmm0, %xmm0 |
| ; CHECK-NEXT: retq |
| %inlo = insertelement <16 x i8> poison, i8 %scalar, i64 0 |
| %varsplat = shufflevector <16 x i8> %inlo, <16 x i8> poison, <16 x i32> zeroinitializer |
| %and = and <16 x i8> %src, %varsplat |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> %matrix, i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| ;; Negative test: larger splats should not fold |
| define <16 x i8> @test_const_splat16_on_const_matrix(<16 x i8> %src) nounwind { |
| ; CHECK-LABEL: test_const_splat16_on_const_matrix: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vpand {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 |
| ; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [64,32,16,8,4,2,1,128,64,32,16,8,4,2,1,128] |
| ; CHECK-NEXT: retq |
| %splat = bitcast <8 x i16> splat(i16 1) to <16 x i8> |
| %and = and <16 x i8> %src, %splat |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0) |
| ret <16 x i8> %gfni |
| } |
| |
| ;; Negative test: larger splats should not fold |
| define <16 x i8> @test_var_splat16_on_const_matrix(<16 x i8> %src, i16 %scalar) nounwind { |
| ; CHECK-LABEL: test_var_splat16_on_const_matrix: |
| ; CHECK: # %bb.0: |
| ; CHECK-NEXT: vmovd %edi, %xmm1 |
| ; CHECK-NEXT: vpbroadcastw %xmm1, %xmm1 |
| ; CHECK-NEXT: vpand %xmm1, %xmm0, %xmm0 |
| ; CHECK-NEXT: vgf2p8affineqb $0, {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm0, %xmm0 # [64,32,16,8,4,2,1,128,64,32,16,8,4,2,1,128] |
| ; CHECK-NEXT: retq |
| %inlo = insertelement <8 x i16> poison, i16 %scalar, i64 0 |
| %varsplat16 = shufflevector <8 x i16> %inlo, <8 x i16> poison, <8 x i32> zeroinitializer |
| %varsplat = bitcast <8 x i16> %varsplat16 to <16 x i8> |
| %and = and <16 x i8> %src, %varsplat |
| %gfni = call <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8> %and, <16 x i8> <i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128, i8 64, i8 32, i8 16, i8 8, i8 4, i8 2, i8 1, i8 128>, i8 0) |
| ret <16 x i8> %gfni |
| } |