blob: 9028b47200bccc4face9e7f502c4abe08d3b7978 [file]
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc -mtriple=amdgpu6.00 < %s | FileCheck -check-prefix=GCN-6 %s
; RUN: llc -mtriple=amdgpu8.02 -mattr=-flat-for-global < %s | FileCheck -check-prefix=GCN-802 %s
; The bitcast should be pushed through the bitcasts so the vectors can
; be broken down and the shared components can be CSEd
define amdgpu_kernel void @store_bitcast_constant_v8i32_to_v8f32(ptr addrspace(1) %out, <8 x i32> %vec) {
; GCN-6-LABEL: store_bitcast_constant_v8i32_to_v8f32:
; GCN-6: ; %bb.0:
; GCN-6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GCN-6-NEXT: s_mov_b32 s4, 7
; GCN-6-NEXT: v_mov_b32_e32 v0, 7
; GCN-6-NEXT: s_mov_b32 s3, 0xf000
; GCN-6-NEXT: s_mov_b32 s2, -1
; GCN-6-NEXT: v_mov_b32_e32 v3, 8
; GCN-6-NEXT: v_mov_b32_e32 v1, v0
; GCN-6-NEXT: v_mov_b32_e32 v2, v0
; GCN-6-NEXT: s_mov_b32 s5, s4
; GCN-6-NEXT: s_mov_b32 s6, s4
; GCN-6-NEXT: s_mov_b32 s7, s4
; GCN-6-NEXT: s_waitcnt lgkmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v4, s4
; GCN-6-NEXT: v_mov_b32_e32 v5, s5
; GCN-6-NEXT: v_mov_b32_e32 v6, s6
; GCN-6-NEXT: v_mov_b32_e32 v7, s7
; GCN-6-NEXT: s_waitcnt expcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v3, 9
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: s_endpgm
;
; GCN-802-LABEL: store_bitcast_constant_v8i32_to_v8f32:
; GCN-802: ; %bb.0:
; GCN-802-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GCN-802-NEXT: s_mov_b32 s4, 7
; GCN-802-NEXT: v_mov_b32_e32 v0, 7
; GCN-802-NEXT: s_mov_b32 s3, 0xf000
; GCN-802-NEXT: s_mov_b32 s2, -1
; GCN-802-NEXT: v_mov_b32_e32 v3, 8
; GCN-802-NEXT: v_mov_b32_e32 v1, v0
; GCN-802-NEXT: v_mov_b32_e32 v2, v0
; GCN-802-NEXT: s_mov_b32 s5, s4
; GCN-802-NEXT: s_mov_b32 s6, s4
; GCN-802-NEXT: s_mov_b32 s7, s4
; GCN-802-NEXT: s_waitcnt lgkmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: v_mov_b32_e32 v4, s4
; GCN-802-NEXT: v_mov_b32_e32 v5, s5
; GCN-802-NEXT: v_mov_b32_e32 v6, s6
; GCN-802-NEXT: v_mov_b32_e32 v7, s7
; GCN-802-NEXT: v_mov_b32_e32 v3, 9
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: s_endpgm
%vec0.bc = bitcast <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 8> to <8 x float>
store volatile <8 x float> %vec0.bc, ptr addrspace(1) %out
%vec1.bc = bitcast <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 9> to <8 x float>
store volatile <8 x float> %vec1.bc, ptr addrspace(1) %out
ret void
}
define amdgpu_kernel void @store_bitcast_constant_v4i64_to_v8f32(ptr addrspace(1) %out, <4 x i64> %vec) {
; GCN-6-LABEL: store_bitcast_constant_v4i64_to_v8f32:
; GCN-6: ; %bb.0:
; GCN-6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GCN-6-NEXT: s_mov_b32 s5, 0
; GCN-6-NEXT: s_mov_b32 s4, 7
; GCN-6-NEXT: v_mov_b32_e32 v1, 0
; GCN-6-NEXT: s_mov_b32 s3, 0xf000
; GCN-6-NEXT: s_mov_b32 s2, -1
; GCN-6-NEXT: v_mov_b32_e32 v0, 7
; GCN-6-NEXT: v_mov_b32_e32 v2, 8
; GCN-6-NEXT: v_mov_b32_e32 v3, v1
; GCN-6-NEXT: s_mov_b32 s6, s4
; GCN-6-NEXT: s_mov_b32 s7, s5
; GCN-6-NEXT: s_waitcnt lgkmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v4, s4
; GCN-6-NEXT: v_mov_b32_e32 v5, s5
; GCN-6-NEXT: v_mov_b32_e32 v6, s6
; GCN-6-NEXT: v_mov_b32_e32 v7, s7
; GCN-6-NEXT: s_waitcnt expcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v2, 9
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: s_endpgm
;
; GCN-802-LABEL: store_bitcast_constant_v4i64_to_v8f32:
; GCN-802: ; %bb.0:
; GCN-802-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GCN-802-NEXT: s_mov_b32 s5, 0
; GCN-802-NEXT: s_mov_b32 s4, 7
; GCN-802-NEXT: v_mov_b32_e32 v1, 0
; GCN-802-NEXT: s_mov_b32 s3, 0xf000
; GCN-802-NEXT: s_mov_b32 s2, -1
; GCN-802-NEXT: v_mov_b32_e32 v0, 7
; GCN-802-NEXT: v_mov_b32_e32 v2, 8
; GCN-802-NEXT: v_mov_b32_e32 v3, v1
; GCN-802-NEXT: s_mov_b32 s6, s4
; GCN-802-NEXT: s_mov_b32 s7, s5
; GCN-802-NEXT: s_waitcnt lgkmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: v_mov_b32_e32 v4, s4
; GCN-802-NEXT: v_mov_b32_e32 v5, s5
; GCN-802-NEXT: v_mov_b32_e32 v6, s6
; GCN-802-NEXT: v_mov_b32_e32 v7, s7
; GCN-802-NEXT: v_mov_b32_e32 v2, 9
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: s_endpgm
%vec0.bc = bitcast <4 x i64> <i64 7, i64 7, i64 7, i64 8> to <8 x float>
store volatile <8 x float> %vec0.bc, ptr addrspace(1) %out
%vec1.bc = bitcast <4 x i64> <i64 7, i64 7, i64 7, i64 9> to <8 x float>
store volatile <8 x float> %vec1.bc, ptr addrspace(1) %out
ret void
}
define amdgpu_kernel void @store_bitcast_constant_v4i64_to_v4f64(ptr addrspace(1) %out, <4 x i64> %vec) {
; GCN-6-LABEL: store_bitcast_constant_v4i64_to_v4f64:
; GCN-6: ; %bb.0:
; GCN-6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GCN-6-NEXT: s_mov_b32 s5, 0
; GCN-6-NEXT: s_mov_b32 s4, 7
; GCN-6-NEXT: v_mov_b32_e32 v1, 0
; GCN-6-NEXT: s_mov_b32 s3, 0xf000
; GCN-6-NEXT: s_mov_b32 s2, -1
; GCN-6-NEXT: v_mov_b32_e32 v0, 7
; GCN-6-NEXT: v_mov_b32_e32 v2, 8
; GCN-6-NEXT: v_mov_b32_e32 v3, v1
; GCN-6-NEXT: s_mov_b32 s6, s4
; GCN-6-NEXT: s_mov_b32 s7, s5
; GCN-6-NEXT: s_waitcnt lgkmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v4, s4
; GCN-6-NEXT: v_mov_b32_e32 v5, s5
; GCN-6-NEXT: v_mov_b32_e32 v6, s6
; GCN-6-NEXT: v_mov_b32_e32 v7, s7
; GCN-6-NEXT: s_waitcnt expcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v2, 9
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: s_endpgm
;
; GCN-802-LABEL: store_bitcast_constant_v4i64_to_v4f64:
; GCN-802: ; %bb.0:
; GCN-802-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GCN-802-NEXT: s_mov_b32 s5, 0
; GCN-802-NEXT: s_mov_b32 s4, 7
; GCN-802-NEXT: v_mov_b32_e32 v1, 0
; GCN-802-NEXT: s_mov_b32 s3, 0xf000
; GCN-802-NEXT: s_mov_b32 s2, -1
; GCN-802-NEXT: v_mov_b32_e32 v0, 7
; GCN-802-NEXT: v_mov_b32_e32 v2, 8
; GCN-802-NEXT: v_mov_b32_e32 v3, v1
; GCN-802-NEXT: s_mov_b32 s6, s4
; GCN-802-NEXT: s_mov_b32 s7, s5
; GCN-802-NEXT: s_waitcnt lgkmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: v_mov_b32_e32 v4, s4
; GCN-802-NEXT: v_mov_b32_e32 v5, s5
; GCN-802-NEXT: v_mov_b32_e32 v6, s6
; GCN-802-NEXT: v_mov_b32_e32 v7, s7
; GCN-802-NEXT: v_mov_b32_e32 v2, 9
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: s_endpgm
%vec0.bc = bitcast <4 x i64> <i64 7, i64 7, i64 7, i64 8> to <4 x double>
store volatile <4 x double> %vec0.bc, ptr addrspace(1) %out
%vec1.bc = bitcast <4 x i64> <i64 7, i64 7, i64 7, i64 9> to <4 x double>
store volatile <4 x double> %vec1.bc, ptr addrspace(1) %out
ret void
}
define amdgpu_kernel void @store_bitcast_constant_v8i32_to_v16i16(ptr addrspace(1) %out, <16 x i16> %vec) {
; GCN-6-LABEL: store_bitcast_constant_v8i32_to_v16i16:
; GCN-6: ; %bb.0:
; GCN-6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GCN-6-NEXT: s_mov_b32 s4, 0x70007
; GCN-6-NEXT: v_mov_b32_e32 v0, 0x70007
; GCN-6-NEXT: s_mov_b32 s3, 0xf000
; GCN-6-NEXT: s_mov_b32 s2, -1
; GCN-6-NEXT: v_mov_b32_e32 v3, 0x80007
; GCN-6-NEXT: v_mov_b32_e32 v1, v0
; GCN-6-NEXT: v_mov_b32_e32 v2, v0
; GCN-6-NEXT: s_mov_b32 s5, s4
; GCN-6-NEXT: s_mov_b32 s6, s4
; GCN-6-NEXT: s_mov_b32 s7, s4
; GCN-6-NEXT: s_waitcnt lgkmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v4, s4
; GCN-6-NEXT: v_mov_b32_e32 v5, s5
; GCN-6-NEXT: v_mov_b32_e32 v6, s6
; GCN-6-NEXT: v_mov_b32_e32 v7, s7
; GCN-6-NEXT: s_waitcnt expcnt(0)
; GCN-6-NEXT: v_mov_b32_e32 v3, 0x90007
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-6-NEXT: s_waitcnt vmcnt(0)
; GCN-6-NEXT: s_endpgm
;
; GCN-802-LABEL: store_bitcast_constant_v8i32_to_v16i16:
; GCN-802: ; %bb.0:
; GCN-802-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GCN-802-NEXT: s_mov_b32 s4, 0x70007
; GCN-802-NEXT: v_mov_b32_e32 v0, 0x70007
; GCN-802-NEXT: s_mov_b32 s3, 0xf000
; GCN-802-NEXT: s_mov_b32 s2, -1
; GCN-802-NEXT: v_mov_b32_e32 v3, 0x80007
; GCN-802-NEXT: v_mov_b32_e32 v1, v0
; GCN-802-NEXT: v_mov_b32_e32 v2, v0
; GCN-802-NEXT: s_mov_b32 s5, s4
; GCN-802-NEXT: s_mov_b32 s6, s4
; GCN-802-NEXT: s_mov_b32 s7, s4
; GCN-802-NEXT: s_waitcnt lgkmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: v_mov_b32_e32 v4, s4
; GCN-802-NEXT: v_mov_b32_e32 v5, s5
; GCN-802-NEXT: v_mov_b32_e32 v6, s6
; GCN-802-NEXT: v_mov_b32_e32 v7, s7
; GCN-802-NEXT: v_mov_b32_e32 v3, 0x90007
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[0:3], off, s[0:3], 0 offset:16
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 0
; GCN-802-NEXT: s_waitcnt vmcnt(0)
; GCN-802-NEXT: s_endpgm
%vec0.bc = bitcast <16 x i16> <i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 8> to <8 x float>
store volatile <8 x float> %vec0.bc, ptr addrspace(1) %out
%vec1.bc = bitcast <16 x i16> <i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 9> to <8 x float>
store volatile <8 x float> %vec1.bc, ptr addrspace(1) %out
ret void
}
attributes #0 = { nounwind }
attributes #1 = { nounwind readnone convergent }