| ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py |
| ; RUN: opt -S -passes='default<O2>' -mtriple=x86_64-unknown-linux-gnu -data-layout="e-p:64:64" < %s | FileCheck %s |
| |
| ; For the cases below, the X86 cost model considers the original vector load |
| ; plus its lane-zero extract to have the same cost as a scalar load. The other |
| ; store overlaps the vector access but not the scalar access. Scalarizing the |
| ; single-use load therefore lets later passes forward the scalar store and |
| ; eliminate the remaining stack traffic. |
| |
| ; Original reproducer: forward an integer store through a floating-point load. |
| define float @PR217598(ptr %p, i32 %bits) { |
| ; CHECK-LABEL: @PR217598( |
| ; CHECK-NEXT: [[TMP1:%.*]] = bitcast i32 [[BITS:%.*]] to float |
| ; CHECK-NEXT: ret float [[TMP1]] |
| ; |
| %a = alloca [20 x i8], align 16 |
| store i32 %bits, ptr %a, align 4 |
| %q = getelementptr i8, ptr %a, i64 12 |
| store ptr %p, ptr %q, align 4 |
| %v = load <4 x float>, ptr %a, align 4 |
| %x = extractelement <4 x float> %v, i64 0 |
| ret float %x |
| } |
| |
| ; Forward a store with the same type as the extracted element. |
| define float @forward_same_type(ptr %p, float %value) { |
| ; CHECK-LABEL: @forward_same_type( |
| ; CHECK-NEXT: ret float [[VALUE:%.*]] |
| ; |
| %a = alloca [16 x i8], align 16 |
| store float %value, ptr %a, align 4 |
| %q = getelementptr inbounds i8, ptr %a, i64 8 |
| store ptr %p, ptr %q, align 8 |
| %v = load <4 x float>, ptr %a, align 4 |
| %x = extractelement <4 x float> %v, i64 0 |
| ret float %x |
| } |
| |
| ; Forward from a vector load whose base is at a nonzero alloca offset, and |
| ; preserve the integer-to-double bit interpretation. |
| define double @forward_from_offset(ptr %p, i64 %bits) { |
| ; CHECK-LABEL: @forward_from_offset( |
| ; CHECK-NEXT: [[TMP1:%.*]] = bitcast i64 [[BITS:%.*]] to double |
| ; CHECK-NEXT: ret double [[TMP1]] |
| ; |
| %a = alloca [32 x i8], align 16 |
| %base = getelementptr inbounds i8, ptr %a, i64 8 |
| store i64 %bits, ptr %base, align 8 |
| %q = getelementptr inbounds i8, ptr %base, i64 8 |
| store ptr %p, ptr %q, align 8 |
| %v = load <2 x double>, ptr %base, align 8 |
| %x = extractelement <2 x double> %v, i64 0 |
| ret double %x |
| } |
| |
| ; Promote stores from different predecessors after scalarization. |
| define float @forward_conditional_store(ptr %p, i1 %cond, float %x, float %y) { |
| ; CHECK-LABEL: @forward_conditional_store( |
| ; CHECK-NEXT: entry: |
| ; CHECK-NEXT: [[X_Y:%.*]] = select i1 [[COND:%.*]], float [[X:%.*]], float [[Y:%.*]] |
| ; CHECK-NEXT: ret float [[X_Y]] |
| ; |
| entry: |
| %a = alloca [16 x i8], align 16 |
| br i1 %cond, label %then, label %else |
| |
| then: |
| store float %x, ptr %a, align 4 |
| br label %merge |
| |
| else: |
| store float %y, ptr %a, align 4 |
| br label %merge |
| |
| merge: |
| %q = getelementptr inbounds i8, ptr %a, i64 8 |
| store ptr %p, ptr %q, align 8 |
| %v = load <4 x float>, ptr %a, align 4 |
| %value = extractelement <4 x float> %v, i64 0 |
| ret float %value |
| } |
| |
| ; Exercise a vector shorter than the native 128-bit vector width. The pointer |
| ; store only partially overlaps the original vector load. |
| define float @forward_v2f32(ptr %p, i32 %bits) { |
| ; CHECK-LABEL: @forward_v2f32( |
| ; CHECK-NEXT: [[TMP1:%.*]] = bitcast i32 [[BITS:%.*]] to float |
| ; CHECK-NEXT: ret float [[TMP1]] |
| ; |
| %a = alloca [12 x i8], align 8 |
| store i32 %bits, ptr %a, align 4 |
| %q = getelementptr inbounds i8, ptr %a, i64 4 |
| store ptr %p, ptr %q, align 4 |
| %v = load <2 x float>, ptr %a, align 4 |
| %x = extractelement <2 x float> %v, i64 0 |
| ret float %x |
| } |
| |
| ; Exercise a narrower floating-point element type. |
| define half @forward_f16(ptr %p, i16 %bits) { |
| ; CHECK-LABEL: @forward_f16( |
| ; CHECK-NEXT: [[TMP1:%.*]] = bitcast i16 [[BITS:%.*]] to half |
| ; CHECK-NEXT: ret half [[TMP1]] |
| ; |
| %a = alloca [16 x i8], align 16 |
| store i16 %bits, ptr %a, align 2 |
| %q = getelementptr inbounds i8, ptr %a, i64 8 |
| store ptr %p, ptr %q, align 8 |
| %v = load <8 x half>, ptr %a, align 2 |
| %x = extractelement <8 x half> %v, i64 0 |
| ret half %x |
| } |