blob: b1337924219d49b06cfeb8b2f0610d8314362647 [file]
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph" --version 6
; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -force-vector-width=4 -S %s | FileCheck %s
; The inner loop only reads, and the outer loop stores to a distinct object at a
; unit-stride address.
define void @inner_reads_outer_store(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @inner_reads_outer_store(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; Each of the two stores writes a distinct object.
define void @two_distinct_stores(ptr noalias %A, ptr noalias %B, ptr noalias %C, i64 %N, i64 %M) {
; CHECK-LABEL: define void @two_distinct_stores(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds float, ptr [[C]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP8]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%C.ptr = getelementptr inbounds float, ptr %C, i64 %outer.iv
store float %sum.next, ptr %C.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The store advances by more than the bytes it writes, so the lanes write
; disjoint elements.
define void @store_strided(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @store_strided(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[TMP7:%.*]] = shl nsw <4 x i64> [[VEC_IND]], splat (i64 1)
; CHECK-NEXT: [[WIDE_GEP5:%.*]] = getelementptr inbounds float, ptr [[B]], <4 x i64> [[TMP7]]
; CHECK-NEXT: call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP3]], <4 x ptr> align 4 [[WIDE_GEP5]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%outer.iv.mul.2 = mul nsw i64 %outer.iv, 2
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv.mul.2
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The store may alias the load, but the memory both access over the whole nest
; can be bounded.
define void @may_alias_load(ptr %A, ptr %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @may_alias_load(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The store may alias both loads.
define void @may_alias_two_loads(ptr %A, ptr %B, ptr %C, i64 %N, i64 %M) {
; CHECK-LABEL: define void @may_alias_two_loads(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH6:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH6]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[WIDE_GEP4:%.*]] = getelementptr inbounds float, ptr [[C]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER5:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP4]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3:%.*]] = fadd <4 x float> [[WIDE_MASKED_GATHER]], [[WIDE_MASKED_GATHER5]]
; CHECK-NEXT: [[TMP4]] = fadd <4 x float> [[SUM3]], [[TMP3]]
; CHECK-NEXT: [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
; CHECK-NEXT: br i1 [[TMP7]], label %[[OUTER_LATCH6]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH6]]:
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[TMP8]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%C.ptr = getelementptr inbounds float, ptr %C, i64 %idx
%C.val = load float, ptr %C.ptr, align 4
%add = fadd float %A.val, %C.val
%sum.next = fadd float %sum, %add
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The load may alias the store, and only dereferences the pointer of the last
; inner iteration, so the inbounds GEPs of earlier iterations may be poison.
define void @may_alias_load_after_inner_loop(ptr %A, ptr %B) {
; CHECK-LABEL: define void @may_alias_load_after_inner_loop(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP2:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP0:%.*]] = mul <4 x i64> [[INNER_IV2]], splat (i64 9223372036854775807)
; CHECK-NEXT: [[TMP1:%.*]] = add <4 x i64> <i64 0, i64 1, i64 2, i64 3>, [[TMP0]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i8, ptr [[A]], <4 x i64> [[TMP1]]
; CHECK-NEXT: [[TMP2]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <4 x i64> [[TMP2]], splat (i64 3)
; CHECK-NEXT: [[TMP4:%.*]] = extractelement <4 x i1> [[TMP3]], i64 0
; CHECK-NEXT: br i1 [[TMP4]], label %[[OUTER_LATCH3:.*]], label %[[INNER_BODY1]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: [[OUTER_LATCH3]]:
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x i8> @llvm.masked.gather.v4i8.v4p0(<4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x i8> poison)
; CHECK-NEXT: [[WIDE_GEP4:%.*]] = getelementptr inbounds i8, ptr [[B]], <4 x i64> <i64 0, i64 -1, i64 -2, i64 -3>
; CHECK-NEXT: call void @llvm.masked.scatter.v4i8.v4p0(<4 x i8> [[WIDE_MASKED_GATHER]], <4 x ptr> align 1 [[WIDE_GEP4]], <4 x i1> splat (i1 true))
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%inner.iv.mul.smax = mul i64 %inner.iv, 9223372036854775807
%idx = add i64 %outer.iv, %inner.iv.mul.smax
%A.ptr = getelementptr inbounds i8, ptr %A, i64 %idx
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, 3
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%A.ptr.lcssa = phi ptr [ %A.ptr, %inner.body ]
%A.val = load i8, ptr %A.ptr.lcssa, align 1
%outer.iv.neg = sub i64 0, %outer.iv
%B.ptr = getelementptr inbounds i8, ptr %B, i64 %outer.iv.neg
store i8 %A.val, ptr %B.ptr, align 1
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, 4
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; Same as @may_alias_load_after_inner_loop, but with noalias arguments.
define void @load_after_inner_loop(ptr noalias %A, ptr noalias %B) {
; CHECK-LABEL: define void @load_after_inner_loop(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP2:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP0:%.*]] = mul <4 x i64> [[INNER_IV2]], splat (i64 9223372036854775807)
; CHECK-NEXT: [[TMP1:%.*]] = add <4 x i64> <i64 0, i64 1, i64 2, i64 3>, [[TMP0]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i8, ptr [[A]], <4 x i64> [[TMP1]]
; CHECK-NEXT: [[TMP2]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <4 x i64> [[TMP2]], splat (i64 3)
; CHECK-NEXT: [[TMP4:%.*]] = extractelement <4 x i1> [[TMP3]], i64 0
; CHECK-NEXT: br i1 [[TMP4]], label %[[OUTER_LATCH3:.*]], label %[[INNER_BODY1]], !llvm.loop [[LOOP13:![0-9]+]]
; CHECK: [[OUTER_LATCH3]]:
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x i8> @llvm.masked.gather.v4i8.v4p0(<4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x i8> poison)
; CHECK-NEXT: [[WIDE_GEP4:%.*]] = getelementptr inbounds i8, ptr [[B]], <4 x i64> <i64 0, i64 -1, i64 -2, i64 -3>
; CHECK-NEXT: call void @llvm.masked.scatter.v4i8.v4p0(<4 x i8> [[WIDE_MASKED_GATHER]], <4 x ptr> align 1 [[WIDE_GEP4]], <4 x i1> splat (i1 true))
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%inner.iv.mul.smax = mul i64 %inner.iv, 9223372036854775807
%idx = add i64 %outer.iv, %inner.iv.mul.smax
%A.ptr = getelementptr inbounds i8, ptr %A, i64 %idx
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, 3
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%A.ptr.lcssa = phi ptr [ %A.ptr, %inner.body ]
%A.val = load i8, ptr %A.ptr.lcssa, align 1
%outer.iv.neg = sub i64 0, %outer.iv
%B.ptr = getelementptr inbounds i8, ptr %B, i64 %outer.iv.neg
store i8 %A.val, ptr %B.ptr, align 1
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, 4
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The load may alias the store, and its address is not formed by an inbounds
; GEP.
define void @may_alias_load_not_inbounds(ptr %A, ptr %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @may_alias_load_not_inbounds(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The load and the store may alias, but are in different address spaces.
define void @may_alias_different_address_spaces(ptr addrspace(1) %A, ptr %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @may_alias_different_address_spaces(
; CHECK-SAME: ptr addrspace(1) [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p1(<4 x ptr addrspace(1)> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr addrspace(1) %A, i64 %idx
%A.val = load float, ptr addrspace(1) %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The load may alias the store, and is only executed conditionally.
define void @may_alias_load_not_executed_every_iteration(ptr %A, ptr %B, i64 %N, i64 %M, i1 %cond) {
; CHECK-LABEL: define void @may_alias_load_not_executed_every_iteration(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]], i1 [[COND:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH7:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH7]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_HEADER1:.*]]
; CHECK: [[INNER_HEADER1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_LATCH5:.*]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_LATCH5]] ]
; CHECK-NEXT: br i1 [[COND]], label %[[INNER_THEN4:.*]], label %[[INNER_LATCH5]]
; CHECK: [[INNER_THEN4]]:
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: br label %[[INNER_LATCH5]]
; CHECK: [[INNER_LATCH5]]:
; CHECK-NEXT: [[VAL6:%.*]] = phi <4 x float> [ [[WIDE_MASKED_GATHER]], %[[INNER_THEN4]] ], [ zeroinitializer, %[[INNER_HEADER1]] ]
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[VAL6]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH7]], label %[[INNER_HEADER1]]
; CHECK: [[OUTER_LATCH7]]:
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP18:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.header
inner.header:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.latch ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.latch ]
br i1 %cond, label %inner.then, label %inner.latch
inner.then:
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
br label %inner.latch
inner.latch:
%val = phi float [ %A.val, %inner.then ], [ 0.000000e+00, %inner.header ]
%sum.next = fadd float %sum, %val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.header
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The store address does not vary with the outer loop, so all lanes write the
; same location.
define void @store_outer_invariant(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @store_outer_invariant(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x ptr> poison, ptr [[B]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x ptr> [[BROADCAST_SPLATINSERT1]], <4 x ptr> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH6:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH6]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY3:.*]]
; CHECK: [[INNER_BODY3]]:
; CHECK-NEXT: [[INNER_IV4:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY3]] ]
; CHECK-NEXT: [[SUM5:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY3]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV4]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM5]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV4]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH6]], label %[[INNER_BODY3]]
; CHECK: [[OUTER_LATCH6]]:
; CHECK-NEXT: call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP3]], <4 x ptr> align 4 [[BROADCAST_SPLAT2]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
store float %sum.next, ptr %B, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The store address is not formed by an inbounds GEP, so it may wrap and two
; outer iterations may write the same location.
define void @store_not_inbounds(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @store_not_inbounds(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP22:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr float, ptr %B, i64 %outer.iv
store float %sum.next, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The store advances by 4 bytes but writes 8, so adjacent lanes overlap.
define void @store_wider_than_stride(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @store_wider_than_stride(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
; CHECK-NEXT: br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[WIDE_GEP5:%.*]] = getelementptr inbounds float, ptr [[B]], <4 x i64> [[VEC_IND]]
; CHECK-NEXT: [[TMP7:%.*]] = fpext <4 x float> [[TMP3]] to <4 x double>
; CHECK-NEXT: call void @llvm.masked.scatter.v4f64.v4p0(<4 x double> [[TMP7]], <4 x ptr> align 4 [[WIDE_GEP5]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP24:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%sum.next = fadd float %sum, %A.val
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
%sum.ext = fpext float %sum.next to double
store double %sum.ext, ptr %B.ptr, align 4
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The inner-loop store writes a distinct row on each outer iteration.
define void @store_in_inner_loop(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @store_in_inner_loop(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
; CHECK-NEXT: [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[WIDE_GEP3:%.*]] = getelementptr inbounds float, ptr [[B]], <4 x i64> [[TMP2]]
; CHECK-NEXT: call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[WIDE_MASKED_GATHER]], <4 x ptr> align 4 [[WIDE_GEP3]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[TMP3]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i64> [[TMP3]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x i1> [[TMP4]], i64 0
; CHECK-NEXT: br i1 [[TMP5]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH4]]:
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP26:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.M, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%B.ptr = getelementptr inbounds float, ptr %B, i64 %idx
store float %A.val, ptr %B.ptr, align 4
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The lanes of the column-major store A[i + j*8] are only distinct for VF <= 8.
define void @col_major_m8(ptr noalias %A, i64 %N) {
; CHECK-LABEL: define void @col_major_m8(
; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP1:%.*]] = shl nsw <4 x i64> [[INNER_IV2]], splat (i64 3)
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP1]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
; CHECK-NEXT: [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 8)
; CHECK-NEXT: [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
; CHECK-NEXT: br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH3]]:
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%inner.iv.mul.8 = mul nsw i64 %inner.iv, 8
%idx = add nsw i64 %outer.iv, %inner.iv.mul.8
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%outer.iv.fp = sitofp i64 %outer.iv to float
%sub = fsub float %outer.iv.fp, %A.val
store float %sub, ptr %A.ptr, align 4
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, 8
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The lanes of the column-major store A[i + j*2] are only distinct for VF <= 2.
define void @col_major_m2(ptr noalias %A, i64 %N) {
; CHECK-LABEL: define void @col_major_m2(
; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP1:%.*]] = shl nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP1]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
; CHECK-NEXT: [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 2)
; CHECK-NEXT: [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
; CHECK-NEXT: br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH3]]:
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%inner.iv.mul.2 = mul nsw i64 %inner.iv, 2
%idx = add nsw i64 %outer.iv, %inner.iv.mul.2
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%outer.iv.fp = sitofp i64 %outer.iv to float
%sub = fsub float %outer.iv.fp, %A.val
store float %sub, ptr %A.ptr, align 4
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, 2
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The row-major store A[i*8 + j] writes a distinct row on each outer iteration.
define void @row_major_m8(ptr noalias %A, i64 %N) {
; CHECK-LABEL: define void @row_major_m8(
; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
; CHECK-NEXT: [[TMP1:%.*]] = shl nsw <4 x i64> [[VEC_IND]], splat (i64 3)
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
; CHECK-NEXT: [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 8)
; CHECK-NEXT: [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
; CHECK-NEXT: br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH3]]:
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.8 = mul nsw i64 %outer.iv, 8
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%idx = add nsw i64 %outer.iv.mul.8, %inner.iv
%A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 4
%outer.iv.fp = sitofp i64 %outer.iv to float
%sub = fsub float %outer.iv.fp, %A.val
store float %sub, ptr %A.ptr, align 4
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, 8
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The 4-byte accesses to A[i + j*4] advance by one byte per outer iteration, so
; the lanes overlap.
define void @access_wider_than_outer_stride(ptr noalias %A, i64 %N) {
; CHECK-LABEL: define void @access_wider_than_outer_stride(
; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
; CHECK-NEXT: br label %[[INNER_BODY1:.*]]
; CHECK: [[INNER_BODY1]]:
; CHECK-NEXT: [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
; CHECK-NEXT: [[TMP1:%.*]] = shl nsw <4 x i64> [[INNER_IV2]], splat (i64 2)
; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP1]]
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i8, ptr [[A]], <4 x i64> [[TMP2]]
; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
; CHECK-NEXT: [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
; CHECK-NEXT: [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
; CHECK-NEXT: call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true))
; CHECK-NEXT: [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 4)
; CHECK-NEXT: [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
; CHECK-NEXT: br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
; CHECK: [[OUTER_LATCH3]]:
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%inner.iv.mul.4 = mul nsw i64 %inner.iv, 4
%idx = add nsw i64 %outer.iv, %inner.iv.mul.4
%A.ptr = getelementptr inbounds i8, ptr %A, i64 %idx
%A.val = load float, ptr %A.ptr, align 1
%outer.iv.fp = sitofp i64 %outer.iv to float
%sub = fsub float %outer.iv.fp, %A.val
store float %sub, ptr %A.ptr, align 1
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, 4
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
; The nest copies between distinct objects using a memory intrinsic.
define void @memcpy_in_nest(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
; CHECK-LABEL: define void @memcpy_in_nest(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
; CHECK: [[OUTER_HEADER]]:
; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
; CHECK-NEXT: [[OUTER_IV_MUL_M:%.*]] = mul nsw i64 [[OUTER_IV]], [[M]]
; CHECK-NEXT: br label %[[INNER_BODY:.*]]
; CHECK: [[INNER_BODY]]:
; CHECK-NEXT: [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER_BODY]] ]
; CHECK-NEXT: [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
; CHECK-NEXT: [[INNER_IV_CMP:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
; CHECK-NEXT: br i1 [[INNER_IV_CMP]], label %[[OUTER_LATCH]], label %[[INNER_BODY]]
; CHECK: [[OUTER_LATCH]]:
; CHECK-NEXT: [[A_PTR:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[OUTER_IV_MUL_M]]
; CHECK-NEXT: [[B_PTR:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[OUTER_IV]]
; CHECK-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr [[B_PTR]], ptr [[A_PTR]], i64 4, i1 false)
; CHECK-NEXT: [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
; CHECK-NEXT: [[OUTER_IV_CMP:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[OUTER_IV_CMP]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP36:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
entry:
br label %outer.header
outer.header:
%outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
%outer.iv.mul.M = mul nsw i64 %outer.iv, %M
br label %inner.body
inner.body:
%inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
%inner.iv.next = add nuw nsw i64 %inner.iv, 1
%inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
outer.latch:
%A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv.mul.M
%B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
call void @llvm.memcpy.p0.p0.i64(ptr %B.ptr, ptr %A.ptr, i64 4, i1 false)
%outer.iv.next = add nuw nsw i64 %outer.iv, 1
%outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
exit:
ret void
}
declare void @llvm.memcpy.p0.p0.i64(ptr, ptr, i64, i1)
!0 = distinct !{!0, !1}
!1 = !{!"llvm.loop.vectorize.enable"}