blob: d3766ca8850661538c87e6e275775a90d3a91d39 [file]
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=loop-interchange -cache-line-size=4 -S | FileCheck %s
@b = external global [512 x [4 x i32]]
@c = global [2 x [4 x i32]] zeroinitializer, align 1
; Check that the outermost and the middle loops are not interchanged since
; the innermost loop has a reduction operation which is however not in a form
; that loop interchange can handle. Interchanging the outermost and the
; middle loops would intervene with the reduction and cause miscompile.
define i32 @test7() {
; CHECK-LABEL: define i32 @test7() {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: br label %[[FOR_COND1_PREHEADER_I:.*]]
; CHECK: [[FOR_COND1_PREHEADER_I]]:
; CHECK-NEXT: [[I_011_I:%.*]] = phi i16 [ 0, %[[ENTRY]] ], [ [[INC20_I:%.*]], %[[FOR_INC19_I:.*]] ]
; CHECK-NEXT: br label %[[FOR_COND4_PREHEADER_I:.*]]
; CHECK: [[FOR_COND4_PREHEADER_I]]:
; CHECK-NEXT: [[J_010_I:%.*]] = phi i16 [ 0, %[[FOR_COND1_PREHEADER_I]] ], [ [[INC17_I:%.*]], %[[MIDDLE_BLOCK:.*]] ]
; CHECK-NEXT: [[ARRAYIDX14_I:%.*]] = getelementptr inbounds [2 x [4 x i32]], ptr @c, i16 0, i16 [[I_011_I]], i16 [[J_010_I]]
; CHECK-NEXT: [[ARRAYIDX14_PROMOTED_I:%.*]] = load i32, ptr [[ARRAYIDX14_I]], align 1
; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> <i32 poison, i32 0, i32 0, i32 0>, i32 [[ARRAYIDX14_PROMOTED_I]], i64 0
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i16 [ 0, %[[FOR_COND4_PREHEADER_I]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i32> [ [[TMP0]], %[[FOR_COND4_PREHEADER_I]] ], [ [[TMP16:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP1:%.*]] = or i16 [[INDEX]], 1
; CHECK-NEXT: [[TMP2:%.*]] = or i16 [[INDEX]], 2
; CHECK-NEXT: [[TMP3:%.*]] = or i16 [[INDEX]], 3
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 [[INDEX]], i16 [[J_010_I]]
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 [[TMP1]], i16 [[J_010_I]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 [[TMP2]], i16 [[J_010_I]]
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 [[TMP3]], i16 [[J_010_I]]
; CHECK-NEXT: [[TMP8:%.*]] = load i32, ptr [[TMP4]], align 1
; CHECK-NEXT: [[TMP9:%.*]] = load i32, ptr [[TMP5]], align 1
; CHECK-NEXT: [[TMP10:%.*]] = load i32, ptr [[TMP6]], align 1
; CHECK-NEXT: [[TMP11:%.*]] = load i32, ptr [[TMP7]], align 1
; CHECK-NEXT: [[TMP12:%.*]] = insertelement <4 x i32> poison, i32 [[TMP8]], i64 0
; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x i32> [[TMP12]], i32 [[TMP9]], i64 1
; CHECK-NEXT: [[TMP14:%.*]] = insertelement <4 x i32> [[TMP13]], i32 [[TMP10]], i64 2
; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x i32> [[TMP14]], i32 [[TMP11]], i64 3
; CHECK-NEXT: [[TMP16]] = add <4 x i32> [[TMP15]], [[VEC_PHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i16 [[INDEX]], 4
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i16 [[INDEX_NEXT]], 512
; CHECK-NEXT: br i1 [[TMP17]], label %[[MIDDLE_BLOCK]], label %[[VECTOR_BODY]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[DOTLCSSA:%.*]] = phi <4 x i32> [ [[TMP16]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP18:%.*]] = tail call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[DOTLCSSA]])
; CHECK-NEXT: store i32 [[TMP18]], ptr [[ARRAYIDX14_I]], align 1
; CHECK-NEXT: [[INC17_I]] = add nuw nsw i16 [[J_010_I]], 1
; CHECK-NEXT: [[EXITCOND12_NOT_I:%.*]] = icmp eq i16 [[INC17_I]], 4
; CHECK-NEXT: br i1 [[EXITCOND12_NOT_I]], label %[[FOR_INC19_I]], label %[[FOR_COND4_PREHEADER_I]]
; CHECK: [[FOR_INC19_I]]:
; CHECK-NEXT: [[INC20_I]] = add nuw nsw i16 [[I_011_I]], 1
; CHECK-NEXT: [[EXITCOND13_NOT_I:%.*]] = icmp eq i16 [[INC20_I]], 2
; CHECK-NEXT: br i1 [[EXITCOND13_NOT_I]], label %[[TEST_EXIT:.*]], label %[[FOR_COND1_PREHEADER_I]]
; CHECK: [[TEST_EXIT]]:
; CHECK-NEXT: [[TMP19:%.*]] = load i32, ptr @c, align 1
; CHECK-NEXT: ret i32 [[TMP19]]
;
entry:
br label %for.cond1.preheader.i
for.cond1.preheader.i: ; preds = %for.inc19.i, %entry
%i.011.i = phi i16 [ 0, %entry ], [ %inc20.i, %for.inc19.i ]
br label %for.cond4.preheader.i
for.cond4.preheader.i: ; preds = %middle.block, %for.cond1.preheader.i
%j.010.i = phi i16 [ 0, %for.cond1.preheader.i ], [ %inc17.i, %middle.block ]
%arrayidx14.i = getelementptr inbounds [2 x [4 x i32]], ptr @c, i16 0, i16 %i.011.i, i16 %j.010.i
%arrayidx14.promoted.i = load i32, ptr %arrayidx14.i, align 1
%0 = insertelement <4 x i32> <i32 poison, i32 0, i32 0, i32 0>, i32 %arrayidx14.promoted.i, i64 0
br label %vector.body
vector.body: ; preds = %vector.body, %for.cond4.preheader.i
%index = phi i16 [ 0, %for.cond4.preheader.i ], [ %index.next, %vector.body ]
%vec.phi = phi <4 x i32> [ %0, %for.cond4.preheader.i ], [ %16, %vector.body ]
%1 = or i16 %index, 1
%2 = or i16 %index, 2
%3 = or i16 %index, 3
%4 = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 %index, i16 %j.010.i
%5 = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 %1, i16 %j.010.i
%6 = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 %2, i16 %j.010.i
%7 = getelementptr inbounds [512 x [4 x i32]], ptr @b, i16 0, i16 %3, i16 %j.010.i
%8 = load i32, ptr %4, align 1
%9 = load i32, ptr %5, align 1
%10 = load i32, ptr %6, align 1
%11 = load i32, ptr %7, align 1
%12 = insertelement <4 x i32> poison, i32 %8, i64 0
%13 = insertelement <4 x i32> %12, i32 %9, i64 1
%14 = insertelement <4 x i32> %13, i32 %10, i64 2
%15 = insertelement <4 x i32> %14, i32 %11, i64 3
%16 = add <4 x i32> %15, %vec.phi
%index.next = add nuw i16 %index, 4
%17 = icmp eq i16 %index.next, 512
br i1 %17, label %middle.block, label %vector.body
middle.block: ; preds = %vector.body
%18 = tail call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %16)
store i32 %18, ptr %arrayidx14.i, align 1
%inc17.i = add nuw nsw i16 %j.010.i, 1
%exitcond12.not.i = icmp eq i16 %inc17.i, 4
br i1 %exitcond12.not.i, label %for.inc19.i, label %for.cond4.preheader.i
for.inc19.i: ; preds = %middle.block
%inc20.i = add nuw nsw i16 %i.011.i, 1
%exitcond13.not.i = icmp eq i16 %inc20.i, 2
br i1 %exitcond13.not.i, label %test.exit, label %for.cond1.preheader.i
test.exit: ; preds = %for.inc19.i
%19 = load i32, ptr @c, align 1
ret i32 %19
}
declare i32 @llvm.vector.reduce.add.v4i32(<4 x i32>)