| ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter "br i1|= select |!prof|!llvm.loop|^.*:" --version 6 |
| ; RUN: opt -passes=loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s --check-prefix=VF4 |
| ; RUN: opt -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=2 -force-ordered-reductions -tail-folding-policy=prefer-fold-tail -S %s | FileCheck %s --check-prefix=VF1IC2 |
| |
| ; A uniform condition stays scalar, so it describes the same probability. |
| define void @widened_select_uniform_cond(ptr %a, i1 %c, i64 %n) !prof !0 { |
| ; VF4-LABEL: define void @widened_select_uniform_cond( |
| ; VF4-SAME: ptr [[A:%.*]], i1 [[C:%.*]], i64 [[N:%.*]]) !prof [[PROF0:![0-9]+]] { |
| ; VF4: [[ENTRY:.*:]] |
| ; VF4: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]] |
| ; VF4: [[VECTOR_PH]]: |
| ; VF4: [[VECTOR_BODY:.*]]: |
| ; VF4: [[TMP2:%.*]] = select i1 [[C]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer |
| ; VF4: br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP1:![0-9]+]] |
| ; VF4: [[MIDDLE_BLOCK]]: |
| ; VF4: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]] |
| ; VF4: [[SCALAR_PH]]: |
| ; VF4: [[LOOP:.*]]: |
| ; VF4: [[SEL:%.*]] = select i1 [[C]], i32 [[L:%.*]], i32 0, !prof [[PROF4:![0-9]+]] |
| ; VF4: br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]] |
| ; VF4: [[EXIT]]: |
| ; |
| ; VF1IC2-LABEL: define void @widened_select_uniform_cond( |
| ; VF1IC2-SAME: ptr [[A:%.*]], i1 [[C:%.*]], i64 [[N:%.*]]) !prof [[PROF0:![0-9]+]] { |
| ; VF1IC2: [[ENTRY:.*:]] |
| ; VF1IC2: [[VECTOR_PH:.*:]] |
| ; VF1IC2: [[VECTOR_BODY:.*]]: |
| ; VF1IC2: br i1 [[TMP2:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]] |
| ; VF1IC2: [[PRED_STORE_IF]]: |
| ; VF1IC2: [[TMP6:%.*]] = select i1 [[C]], i32 [[TMP5:%.*]], i32 0, !prof [[PROF1:![0-9]+]] |
| ; VF1IC2: [[PRED_STORE_CONTINUE]]: |
| ; VF1IC2: br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]] |
| ; VF1IC2: [[PRED_STORE_IF1]]: |
| ; VF1IC2: [[TMP9:%.*]] = select i1 [[C]], i32 [[TMP8:%.*]], i32 0, !prof [[PROF1]] |
| ; VF1IC2: [[PRED_STORE_CONTINUE2]]: |
| ; VF1IC2: br i1 [[TMP10:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP2:![0-9]+]] |
| ; VF1IC2: [[MIDDLE_BLOCK]]: |
| ; VF1IC2: [[EXIT:.*:]] |
| ; |
| entry: |
| br label %loop |
| |
| loop: |
| %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] |
| %gep = getelementptr inbounds i32, ptr %a, i64 %iv |
| %l = load i32, ptr %gep, align 4 |
| %sel = select i1 %c, i32 %l, i32 0, !prof !1 |
| store i32 %sel, ptr %gep, align 4 |
| %iv.next = add nuw nsw i64 %iv, 1 |
| %ec = icmp eq i64 %iv.next, %n |
| br i1 %ec, label %exit, label %loop |
| |
| exit: |
| ret void |
| } |
| |
| define void @widened_select_varying_cond(ptr %a, i64 %n) !prof !0 { |
| ; VF4-LABEL: define void @widened_select_varying_cond( |
| ; VF4-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF4: [[ENTRY:.*:]] |
| ; VF4: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]] |
| ; VF4: [[VECTOR_PH]]: |
| ; VF4: [[VECTOR_BODY:.*]]: |
| ; VF4: [[TMP3:%.*]] = select <4 x i1> [[TMP2:%.*]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer |
| ; VF4: br i1 [[TMP4:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]] |
| ; VF4: [[MIDDLE_BLOCK]]: |
| ; VF4: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]] |
| ; VF4: [[SCALAR_PH]]: |
| ; VF4: [[LOOP:.*]]: |
| ; VF4: [[SEL:%.*]] = select i1 [[C:%.*]], i32 [[L:%.*]], i32 0, !prof [[PROF4]] |
| ; VF4: br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]] |
| ; VF4: [[EXIT]]: |
| ; |
| ; VF1IC2-LABEL: define void @widened_select_varying_cond( |
| ; VF1IC2-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF1IC2: [[ENTRY:.*:]] |
| ; VF1IC2: [[VECTOR_PH:.*:]] |
| ; VF1IC2: [[VECTOR_BODY:.*]]: |
| ; VF1IC2: br i1 [[TMP2:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]] |
| ; VF1IC2: [[PRED_STORE_IF]]: |
| ; VF1IC2: [[TMP7:%.*]] = select i1 [[TMP6:%.*]], i32 [[TMP5:%.*]], i32 0, !prof [[PROF1]] |
| ; VF1IC2: [[PRED_STORE_CONTINUE]]: |
| ; VF1IC2: br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]] |
| ; VF1IC2: [[PRED_STORE_IF1]]: |
| ; VF1IC2: [[TMP11:%.*]] = select i1 [[TMP10:%.*]], i32 [[TMP9:%.*]], i32 0, !prof [[PROF1]] |
| ; VF1IC2: [[PRED_STORE_CONTINUE2]]: |
| ; VF1IC2: br i1 [[TMP12:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]] |
| ; VF1IC2: [[MIDDLE_BLOCK]]: |
| ; VF1IC2: [[EXIT:.*:]] |
| ; |
| entry: |
| br label %loop |
| |
| loop: |
| %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] |
| %gep = getelementptr inbounds i32, ptr %a, i64 %iv |
| %l = load i32, ptr %gep, align 4 |
| %c = icmp sgt i32 %l, 0 |
| %sel = select i1 %c, i32 %l, i32 0, !prof !1 |
| store i32 %sel, ptr %gep, align 4 |
| %iv.next = add nuw nsw i64 %iv, 1 |
| %ec = icmp eq i64 %iv.next, %n |
| br i1 %ec, label %exit, label %loop |
| |
| exit: |
| ret void |
| } |
| |
| ; Folding the not into the compare swaps the operands of the select it feeds. |
| define void @swapped_by_folding_not_into_cmp(ptr %a, ptr %b, i32 %x, i32 %y, i64 %n) !prof !0 { |
| ; VF4-LABEL: define void @swapped_by_folding_not_into_cmp( |
| ; VF4-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF4: [[ENTRY:.*:]] |
| ; VF4: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]] |
| ; VF4: [[VECTOR_MEMCHECK]]: |
| ; VF4: br i1 [[DIFF_CHECK:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]] |
| ; VF4: [[VECTOR_PH]]: |
| ; VF4: [[TMP4:%.*]] = select i1 [[TMP3:%.*]], <4 x i32> splat (i32 20), <4 x i32> splat (i32 10) |
| ; VF4: [[VECTOR_BODY:.*]]: |
| ; VF4: br i1 [[TMP8:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]] |
| ; VF4: [[MIDDLE_BLOCK]]: |
| ; VF4: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]] |
| ; VF4: [[SCALAR_PH]]: |
| ; VF4: [[LOOP:.*]]: |
| ; VF4: [[SEL:%.*]] = select i1 [[C:%.*]], i32 10, i32 20, !prof [[PROF4]] |
| ; VF4: br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP9:![0-9]+]] |
| ; VF4: [[EXIT]]: |
| ; |
| ; VF1IC2-LABEL: define void @swapped_by_folding_not_into_cmp( |
| ; VF1IC2-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF1IC2: [[ENTRY:.*:]] |
| ; VF1IC2: [[VECTOR_MEMCHECK:.*:]] |
| ; VF1IC2: br i1 [[DIFF_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]] |
| ; VF1IC2: [[VECTOR_PH]]: |
| ; VF1IC2: [[VECTOR_BODY:.*]]: |
| ; VF1IC2: br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]] |
| ; VF1IC2: [[PRED_STORE_IF]]: |
| ; VF1IC2: [[TMP8:%.*]] = select i1 [[TMP3:%.*]], i32 20, i32 10, !prof [[PROF1]] |
| ; VF1IC2: [[PRED_STORE_CONTINUE]]: |
| ; VF1IC2: br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]] |
| ; VF1IC2: [[PRED_STORE_IF3]]: |
| ; VF1IC2: [[TMP12:%.*]] = select i1 [[TMP3]], i32 20, i32 10, !prof [[PROF1]] |
| ; VF1IC2: [[PRED_STORE_CONTINUE4]]: |
| ; VF1IC2: br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]] |
| ; VF1IC2: [[MIDDLE_BLOCK]]: |
| ; VF1IC2: [[SCALAR_PH]]: |
| ; VF1IC2: [[LOOP:.*]]: |
| ; VF1IC2: [[SEL:%.*]] = select i1 [[C:%.*]], i32 10, i32 20, !prof [[PROF1]] |
| ; VF1IC2: br i1 [[EC:%.*]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]] |
| ; VF1IC2: [[EXIT]]: |
| ; |
| entry: |
| br label %loop |
| |
| loop: |
| %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] |
| %c = icmp slt i32 %x, %y |
| %nc = xor i1 %c, true |
| %sel = select i1 %c, i32 10, i32 20, !prof !1 |
| %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv |
| store i32 %sel, ptr %gep.a, align 4 |
| %z = zext i1 %nc to i32 |
| %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv |
| store i32 %z, ptr %gep.b, align 4 |
| %iv.next = add nuw nsw i64 %iv, 1 |
| %ec = icmp eq i64 %iv.next, %n |
| br i1 %ec, label %exit, label %loop |
| |
| exit: |
| ret void |
| } |
| |
| ; Same, for folding the not into the select directly. |
| define void @swapped_by_folding_not_into_select(ptr %a, i32 %x, i64 %n) !prof !0 { |
| ; VF4-LABEL: define void @swapped_by_folding_not_into_select( |
| ; VF4-SAME: ptr [[A:%.*]], i32 [[X:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF4: [[ENTRY:.*:]] |
| ; VF4: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]] |
| ; VF4: [[VECTOR_PH]]: |
| ; VF4: [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF4]] |
| ; VF4: [[VECTOR_BODY:.*]]: |
| ; VF4: br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]] |
| ; VF4: [[MIDDLE_BLOCK]]: |
| ; VF4: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]] |
| ; VF4: [[SCALAR_PH]]: |
| ; VF4: [[LOOP:.*]]: |
| ; VF4: [[SEL:%.*]] = select i1 [[NC:%.*]], i32 10, i32 20, !prof [[PROF4]] |
| ; VF4: br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP11:![0-9]+]] |
| ; VF4: [[EXIT]]: |
| ; |
| ; VF1IC2-LABEL: define void @swapped_by_folding_not_into_select( |
| ; VF1IC2-SAME: ptr [[A:%.*]], i32 [[X:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF1IC2: [[ENTRY:.*:]] |
| ; VF1IC2: [[VECTOR_PH:.*:]] |
| ; VF1IC2: [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF1]] |
| ; VF1IC2: [[VECTOR_BODY:.*]]: |
| ; VF1IC2: br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]] |
| ; VF1IC2: [[PRED_STORE_IF]]: |
| ; VF1IC2: [[PRED_STORE_CONTINUE]]: |
| ; VF1IC2: br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]] |
| ; VF1IC2: [[PRED_STORE_IF1]]: |
| ; VF1IC2: [[PRED_STORE_CONTINUE2]]: |
| ; VF1IC2: br i1 [[TMP7:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]] |
| ; VF1IC2: [[MIDDLE_BLOCK]]: |
| ; VF1IC2: [[EXIT:.*:]] |
| ; |
| entry: |
| %c = trunc i32 %x to i1 |
| br label %loop |
| |
| loop: |
| %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] |
| %nc = xor i1 %c, true |
| %sel = select i1 %nc, i32 10, i32 20, !prof !1 |
| %gep = getelementptr inbounds i32, ptr %a, i64 %iv |
| store i32 %sel, ptr %gep, align 4 |
| %iv.next = add nuw nsw i64 %iv, 1 |
| %ec = icmp eq i64 %iv.next, %n |
| br i1 %ec, label %exit, label %loop |
| |
| exit: |
| ret void |
| } |
| |
| define float @ordered_reduction_tail_folded(ptr %src) !prof !0 { |
| ; VF4-LABEL: define float @ordered_reduction_tail_folded( |
| ; VF4-SAME: ptr [[SRC:%.*]]) !prof [[PROF0]] { |
| ; VF4: [[ENTRY:.*:]] |
| ; VF4: [[VECTOR_PH:.*:]] |
| ; VF4: [[VECTOR_BODY:.*]]: |
| ; VF4: br i1 [[TMP1:%.*]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]] |
| ; VF4: [[PRED_LOAD_IF]]: |
| ; VF4: [[PRED_LOAD_CONTINUE]]: |
| ; VF4: br i1 [[TMP6:%.*]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2:.*]] |
| ; VF4: [[PRED_LOAD_IF1]]: |
| ; VF4: [[PRED_LOAD_CONTINUE2]]: |
| ; VF4: br i1 [[TMP12:%.*]], label %[[PRED_LOAD_IF3:.*]], label %[[PRED_LOAD_CONTINUE4:.*]] |
| ; VF4: [[PRED_LOAD_IF3]]: |
| ; VF4: [[PRED_LOAD_CONTINUE4]]: |
| ; VF4: br i1 [[TMP18:%.*]], label %[[PRED_LOAD_IF5:.*]], label %[[PRED_LOAD_CONTINUE6:.*]] |
| ; VF4: [[PRED_LOAD_IF5]]: |
| ; VF4: [[PRED_LOAD_CONTINUE6]]: |
| ; VF4: br i1 [[TMP25:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF12:![0-9]+]], !llvm.loop [[LOOP13:![0-9]+]] |
| ; VF4: [[MIDDLE_BLOCK]]: |
| ; VF4: [[TMP26:%.*]] = select contract <4 x i1> [[TMP0:%.*]], <4 x float> [[TMP24:%.*]], <4 x float> [[VEC_PHI:%.*]] |
| ; VF4: [[EXIT:.*:]] |
| ; |
| ; VF1IC2-LABEL: define float @ordered_reduction_tail_folded( |
| ; VF1IC2-SAME: ptr [[SRC:%.*]]) !prof [[PROF0]] { |
| ; VF1IC2: [[ENTRY:.*:]] |
| ; VF1IC2: [[VECTOR_PH:.*:]] |
| ; VF1IC2: [[VECTOR_BODY:.*]]: |
| ; VF1IC2: br i1 [[TMP2:%.*]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]] |
| ; VF1IC2: [[PRED_LOAD_IF]]: |
| ; VF1IC2: [[PRED_LOAD_CONTINUE]]: |
| ; VF1IC2: br i1 [[TMP3:%.*]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2:.*]] |
| ; VF1IC2: [[PRED_LOAD_IF1]]: |
| ; VF1IC2: [[PRED_LOAD_CONTINUE2]]: |
| ; VF1IC2: [[TMP10:%.*]] = select contract i1 [[TMP2]], float [[TMP6:%.*]], float -0.000000e+00 |
| ; VF1IC2: [[TMP12:%.*]] = select contract i1 [[TMP3]], float [[TMP9:%.*]], float -0.000000e+00 |
| ; VF1IC2: br i1 [[TMP14:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF9:![0-9]+]], !llvm.loop [[LOOP10:![0-9]+]] |
| ; VF1IC2: [[MIDDLE_BLOCK]]: |
| ; VF1IC2: [[EXIT:.*:]] |
| ; |
| entry: |
| br label %loop |
| |
| loop: |
| %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ] |
| %rdx = phi float [ 0.000000e+00, %entry ], [ %rdx.next, %loop ] |
| %iv.next = add nuw nsw i32 %iv, 1 |
| %gep = getelementptr float, ptr %src, i32 %iv |
| %l = load float, ptr %gep, align 4 |
| %rdx.next = fadd contract float %rdx, %l |
| %ec = icmp ult i32 %iv.next, 15 |
| br i1 %ec, label %loop, label %exit |
| |
| exit: |
| %res = phi float [ %rdx.next, %loop ] |
| ret float %res |
| } |
| |
| ; A logical and of a block mask and a condition combines two probabilities. |
| define i32 @logical_and_of_mask_and_condition(ptr noalias %src1, ptr noalias %src2, i64 %n) !prof !0 { |
| ; VF4-LABEL: define i32 @logical_and_of_mask_and_condition( |
| ; VF4-SAME: ptr noalias [[SRC1:%.*]], ptr noalias [[SRC2:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF4: [[ENTRY:.*:]] |
| ; VF4: br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]] |
| ; VF4: [[VECTOR_PH]]: |
| ; VF4: [[VECTOR_BODY:.*]]: |
| ; VF4: br i1 [[TMP3:%.*]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]] |
| ; VF4: [[PRED_LOAD_IF]]: |
| ; VF4: [[PRED_LOAD_CONTINUE]]: |
| ; VF4: br i1 [[TMP8:%.*]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2:.*]] |
| ; VF4: [[PRED_LOAD_IF1]]: |
| ; VF4: [[PRED_LOAD_CONTINUE2]]: |
| ; VF4: br i1 [[TMP14:%.*]], label %[[PRED_LOAD_IF3:.*]], label %[[PRED_LOAD_CONTINUE4:.*]] |
| ; VF4: [[PRED_LOAD_IF3]]: |
| ; VF4: [[PRED_LOAD_CONTINUE4]]: |
| ; VF4: br i1 [[TMP20:%.*]], label %[[PRED_LOAD_IF5:.*]], label %[[PRED_LOAD_CONTINUE6:.*]] |
| ; VF4: [[PRED_LOAD_IF5]]: |
| ; VF4: [[PRED_LOAD_CONTINUE6]]: |
| ; VF4: [[TMP27:%.*]] = select <4 x i1> [[TMP2:%.*]], <4 x i1> [[TMP26:%.*]], <4 x i1> zeroinitializer |
| ; VF4: br i1 [[TMP29:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]] |
| ; VF4: [[MIDDLE_BLOCK]]: |
| ; VF4: [[RDX_SELECT:%.*]] = select i1 [[TMP31:%.*]], i32 1, i32 0 |
| ; VF4: br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]] |
| ; VF4: [[SCALAR_PH]]: |
| ; VF4: [[LOOP_HEADER:.*]]: |
| ; VF4: br i1 [[C_1:%.*]], label %[[THEN:.*]], label %[[LOOP_LATCH:.*]] |
| ; VF4: [[THEN]]: |
| ; VF4: [[SEL:%.*]] = select i1 [[C_2:%.*]], i32 1, i32 [[RES:%.*]] |
| ; VF4: [[LOOP_LATCH]]: |
| ; VF4: br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP16:![0-9]+]] |
| ; VF4: [[EXIT]]: |
| ; |
| ; VF1IC2-LABEL: define i32 @logical_and_of_mask_and_condition( |
| ; VF1IC2-SAME: ptr noalias [[SRC1:%.*]], ptr noalias [[SRC2:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] { |
| ; VF1IC2: [[ENTRY:.*:]] |
| ; VF1IC2: [[VECTOR_PH:.*:]] |
| ; VF1IC2: [[VECTOR_BODY:.*]]: |
| ; VF1IC2: br i1 [[TMP3:%.*]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]] |
| ; VF1IC2: [[PRED_LOAD_IF]]: |
| ; VF1IC2: [[PRED_LOAD_CONTINUE]]: |
| ; VF1IC2: br i1 [[TMP4:%.*]], label %[[PRED_LOAD_IF2:.*]], label %[[PRED_LOAD_CONTINUE3:.*]] |
| ; VF1IC2: [[PRED_LOAD_IF2]]: |
| ; VF1IC2: [[PRED_LOAD_CONTINUE3]]: |
| ; VF1IC2: [[TMP13:%.*]] = select i1 [[TMP3]], i1 [[TMP11:%.*]], i1 false |
| ; VF1IC2: [[TMP14:%.*]] = select i1 [[TMP4]], i1 [[TMP12:%.*]], i1 false |
| ; VF1IC2: br i1 [[TMP13]], label %[[PRED_LOAD_IF4:.*]], label %[[PRED_LOAD_CONTINUE5:.*]] |
| ; VF1IC2: [[PRED_LOAD_IF4]]: |
| ; VF1IC2: [[PRED_LOAD_CONTINUE5]]: |
| ; VF1IC2: br i1 [[TMP14]], label %[[PRED_LOAD_IF6:.*]], label %[[PRED_LOAD_CONTINUE7:.*]] |
| ; VF1IC2: [[PRED_LOAD_IF6]]: |
| ; VF1IC2: [[PRED_LOAD_CONTINUE7]]: |
| ; VF1IC2: [[TMP23:%.*]] = select i1 [[TMP13]], i1 [[TMP21:%.*]], i1 false |
| ; VF1IC2: [[TMP25:%.*]] = select i1 [[TMP14]], i1 [[TMP22:%.*]], i1 false |
| ; VF1IC2: br i1 [[TMP27:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]] |
| ; VF1IC2: [[MIDDLE_BLOCK]]: |
| ; VF1IC2: [[RDX_SELECT:%.*]] = select i1 [[TMP28:%.*]], i32 1, i32 0 |
| ; VF1IC2: [[EXIT:.*:]] |
| ; |
| entry: |
| br label %loop.header |
| |
| loop.header: |
| %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] |
| %res = phi i32 [ 0, %entry ], [ %res.next, %loop.latch ] |
| %gep.1 = getelementptr inbounds i32, ptr %src1, i64 %iv |
| %l.1 = load i32, ptr %gep.1, align 4 |
| %c.1 = icmp sgt i32 %l.1, 35 |
| br i1 %c.1, label %then, label %loop.latch |
| |
| then: |
| %gep.2 = getelementptr inbounds i32, ptr %src2, i64 %iv |
| %l.2 = load i32, ptr %gep.2, align 4 |
| %c.2 = icmp eq i32 %l.2, 2 |
| %sel = select i1 %c.2, i32 1, i32 %res |
| br label %loop.latch |
| |
| loop.latch: |
| %res.next = phi i32 [ %res, %loop.header ], [ %sel, %then ] |
| %iv.next = add nuw nsw i64 %iv, 1 |
| %ec = icmp eq i64 %iv.next, %n |
| br i1 %ec, label %exit, label %loop.header |
| |
| exit: |
| ret i32 %res.next |
| } |
| |
| !0 = !{!"function_entry_count", i64 1000} |
| !1 = !{!"branch_weights", i32 3, i32 5} |
| ;. |
| ; VF4: [[PROF0]] = !{!"function_entry_count", i64 1000} |
| ; VF4: [[LOOP1]] = distinct !{[[LOOP1]], [[META2:![0-9]+]], [[META3:![0-9]+]]} |
| ; VF4: [[META2]] = !{!"llvm.loop.isvectorized", i32 1} |
| ; VF4: [[META3]] = !{!"llvm.loop.unroll.runtime.disable"} |
| ; VF4: [[PROF4]] = !{!"branch_weights", i32 3, i32 5} |
| ; VF4: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]], [[META2]]} |
| ; VF4: [[LOOP6]] = distinct !{[[LOOP6]], [[META2]], [[META3]]} |
| ; VF4: [[LOOP7]] = distinct !{[[LOOP7]], [[META3]], [[META2]]} |
| ; VF4: [[LOOP8]] = distinct !{[[LOOP8]], [[META2]], [[META3]]} |
| ; VF4: [[LOOP9]] = distinct !{[[LOOP9]], [[META2]]} |
| ; VF4: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META3]]} |
| ; VF4: [[LOOP11]] = distinct !{[[LOOP11]], [[META3]], [[META2]]} |
| ; VF4: [[PROF12]] = !{!"branch_weights", i32 1, i32 3} |
| ; VF4: [[LOOP13]] = distinct !{[[LOOP13]], [[META2]], [[META3]], [[META14:![0-9]+]]} |
| ; VF4: [[META14]] = !{!"llvm.loop.estimated_trip_count", i32 4} |
| ; VF4: [[LOOP15]] = distinct !{[[LOOP15]], [[META2]], [[META3]]} |
| ; VF4: [[LOOP16]] = distinct !{[[LOOP16]], [[META3]], [[META2]]} |
| ;. |
| ; VF1IC2: [[PROF0]] = !{!"function_entry_count", i64 1000} |
| ; VF1IC2: [[PROF1]] = !{!"branch_weights", i32 3, i32 5} |
| ; VF1IC2: [[LOOP2]] = distinct !{[[LOOP2]], [[META3:![0-9]+]], [[META4:![0-9]+]]} |
| ; VF1IC2: [[META3]] = !{!"llvm.loop.isvectorized", i32 1} |
| ; VF1IC2: [[META4]] = !{!"llvm.loop.unroll.runtime.disable"} |
| ; VF1IC2: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]], [[META4]]} |
| ; VF1IC2: [[LOOP6]] = distinct !{[[LOOP6]], [[META3]], [[META4]]} |
| ; VF1IC2: [[LOOP7]] = distinct !{[[LOOP7]], [[META3]]} |
| ; VF1IC2: [[LOOP8]] = distinct !{[[LOOP8]], [[META3]], [[META4]]} |
| ; VF1IC2: [[PROF9]] = !{!"branch_weights", i32 1, i32 7} |
| ; VF1IC2: [[LOOP10]] = distinct !{[[LOOP10]], [[META3]], [[META4]], [[META11:![0-9]+]]} |
| ; VF1IC2: [[META11]] = !{!"llvm.loop.estimated_trip_count", i32 8} |
| ; VF1IC2: [[LOOP12]] = distinct !{[[LOOP12]], [[META3]], [[META4]]} |
| ;. |