167 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py2; RUN: llc -mcpu=gfx1010 -mtriple=amdgcn-- < %s | FileCheck -check-prefix=GFX10 %s3; RUN: llc -mcpu=gfx900 -mtriple=amdgcn-- < %s | FileCheck -check-prefix=GFX9 %s4; RUN: llc -mcpu=gfx810 -mtriple=amdgcn-- < %s | FileCheck -check-prefix=GFX8 %s5; RUN: llc -mcpu=gfx1100 -mattr=+real-true16 -mtriple=amdgcn-- < %s | FileCheck -check-prefixes=GFX11,GFX11-TRUE16 %s6; RUN: llc -mcpu=gfx1100 -mattr=-real-true16 -mtriple=amdgcn-- < %s | FileCheck -check-prefixes=GFX11,GFX11-FAKE16 %s7@esgs_ring = external addrspace(3) global [0 x i32], align 655368 9define amdgpu_gs void @main(ptr addrspace(8) %arg, i32 %arg1) {10; GFX10-LABEL: main:11; GFX10: ; %bb.0: ; %bb12; GFX10-NEXT: s_mov_b32 s1, exec_lo13; GFX10-NEXT: .LBB0_1: ; =>This Inner Loop Header: Depth=114; GFX10-NEXT: v_readfirstlane_b32 s4, v015; GFX10-NEXT: v_readfirstlane_b32 s5, v116; GFX10-NEXT: v_readfirstlane_b32 s6, v217; GFX10-NEXT: v_readfirstlane_b32 s7, v318; GFX10-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]19; GFX10-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[2:3]20; GFX10-NEXT: s_and_b32 s0, vcc_lo, s021; GFX10-NEXT: s_and_saveexec_b32 s0, s022; GFX10-NEXT: buffer_load_format_d16_xyz v[5:6], v4, s[4:7], 0 idxen23; GFX10-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr324; GFX10-NEXT: ; implicit-def: $vgpr425; GFX10-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)26; GFX10-NEXT: s_xor_b32 exec_lo, exec_lo, s027; GFX10-NEXT: s_cbranch_execnz .LBB0_128; GFX10-NEXT: ; %bb.2:29; GFX10-NEXT: s_mov_b32 exec_lo, s130; GFX10-NEXT: s_waitcnt vmcnt(0)31; GFX10-NEXT: v_lshrrev_b32_e32 v0, 16, v532; GFX10-NEXT: v_and_b32_e32 v1, 0xffff, v633; GFX10-NEXT: v_mov_b32_e32 v2, 034; GFX10-NEXT: s_waitcnt_vscnt null, 0x035; GFX10-NEXT: ds_write2_b32 v2, v0, v1 offset0:7 offset1:836;37; GFX9-LABEL: main:38; GFX9: ; %bb.0: ; %bb39; GFX9-NEXT: s_mov_b64 s[2:3], exec40; GFX9-NEXT: .LBB0_1: ; =>This Inner Loop Header: Depth=141; GFX9-NEXT: v_readfirstlane_b32 s4, v042; GFX9-NEXT: v_readfirstlane_b32 s5, v143; GFX9-NEXT: v_readfirstlane_b32 s6, v244; GFX9-NEXT: v_readfirstlane_b32 s7, v345; GFX9-NEXT: v_cmp_eq_u64_e32 vcc, s[4:5], v[0:1]46; GFX9-NEXT: v_cmp_eq_u64_e64 s[0:1], s[6:7], v[2:3]47; GFX9-NEXT: s_and_b64 s[0:1], vcc, s[0:1]48; GFX9-NEXT: s_and_saveexec_b64 s[0:1], s[0:1]49; GFX9-NEXT: s_nop 050; GFX9-NEXT: buffer_load_format_d16_xyz v[5:6], v4, s[4:7], 0 idxen51; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr352; GFX9-NEXT: ; implicit-def: $vgpr453; GFX9-NEXT: s_xor_b64 exec, exec, s[0:1]54; GFX9-NEXT: s_cbranch_execnz .LBB0_155; GFX9-NEXT: ; %bb.2:56; GFX9-NEXT: s_mov_b64 exec, s[2:3]57; GFX9-NEXT: s_waitcnt vmcnt(0)58; GFX9-NEXT: v_lshrrev_b32_e32 v0, 16, v559; GFX9-NEXT: v_and_b32_e32 v1, 0xffff, v660; GFX9-NEXT: v_mov_b32_e32 v2, 061; GFX9-NEXT: ds_write2_b32 v2, v0, v1 offset0:7 offset1:862;63; GFX8-LABEL: main:64; GFX8: ; %bb.0: ; %bb65; GFX8-NEXT: s_mov_b64 s[2:3], exec66; GFX8-NEXT: .LBB0_1: ; =>This Inner Loop Header: Depth=167; GFX8-NEXT: v_readfirstlane_b32 s4, v068; GFX8-NEXT: v_readfirstlane_b32 s5, v169; GFX8-NEXT: v_readfirstlane_b32 s6, v270; GFX8-NEXT: v_readfirstlane_b32 s7, v371; GFX8-NEXT: v_cmp_eq_u64_e32 vcc, s[4:5], v[0:1]72; GFX8-NEXT: v_cmp_eq_u64_e64 s[0:1], s[6:7], v[2:3]73; GFX8-NEXT: s_and_b64 s[0:1], vcc, s[0:1]74; GFX8-NEXT: s_and_saveexec_b64 s[0:1], s[0:1]75; GFX8-NEXT: s_nop 076; GFX8-NEXT: buffer_load_format_d16_xyz v[5:6], v4, s[4:7], 0 idxen77; GFX8-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr378; GFX8-NEXT: ; implicit-def: $vgpr479; GFX8-NEXT: s_xor_b64 exec, exec, s[0:1]80; GFX8-NEXT: s_cbranch_execnz .LBB0_181; GFX8-NEXT: ; %bb.2:82; GFX8-NEXT: s_mov_b64 exec, s[2:3]83; GFX8-NEXT: s_waitcnt vmcnt(0)84; GFX8-NEXT: v_lshrrev_b32_e32 v0, 16, v585; GFX8-NEXT: v_and_b32_e32 v1, 0xffff, v686; GFX8-NEXT: v_mov_b32_e32 v2, 087; GFX8-NEXT: s_mov_b32 m0, -188; GFX8-NEXT: ds_write2_b32 v2, v0, v1 offset0:7 offset1:889;90; GFX11-TRUE16-LABEL: main:91; GFX11-TRUE16: ; %bb.0: ; %bb92; GFX11-TRUE16-NEXT: s_mov_b32 s1, exec_lo93; GFX11-TRUE16-NEXT: .LBB0_1: ; =>This Inner Loop Header: Depth=194; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s4, v095; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s5, v196; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s6, v297; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s7, v398; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)99; GFX11-TRUE16-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]100; GFX11-TRUE16-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[2:3]101; GFX11-TRUE16-NEXT: s_and_b32 s0, vcc_lo, s0102; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)103; GFX11-TRUE16-NEXT: s_and_saveexec_b32 s0, s0104; GFX11-TRUE16-NEXT: buffer_load_d16_format_xyz v[5:6], v4, s[4:7], 0 idxen105; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3106; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr4107; GFX11-TRUE16-NEXT: s_xor_b32 exec_lo, exec_lo, s0108; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB0_1109; GFX11-TRUE16-NEXT: ; %bb.2:110; GFX11-TRUE16-NEXT: s_mov_b32 exec_lo, s1111; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, 0112; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)113; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v5.h114; GFX11-TRUE16-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_and_b32 v1, 0xffff, v6115; GFX11-TRUE16-NEXT: ds_store_2addr_b32 v2, v0, v1 offset0:7 offset1:8116;117; GFX11-FAKE16-LABEL: main:118; GFX11-FAKE16: ; %bb.0: ; %bb119; GFX11-FAKE16-NEXT: s_mov_b32 s1, exec_lo120; GFX11-FAKE16-NEXT: .LBB0_1: ; =>This Inner Loop Header: Depth=1121; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s4, v0122; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s5, v1123; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s6, v2124; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s7, v3125; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)126; GFX11-FAKE16-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]127; GFX11-FAKE16-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[2:3]128; GFX11-FAKE16-NEXT: s_and_b32 s0, vcc_lo, s0129; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)130; GFX11-FAKE16-NEXT: s_and_saveexec_b32 s0, s0131; GFX11-FAKE16-NEXT: buffer_load_d16_format_xyz v[5:6], v4, s[4:7], 0 idxen132; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3133; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr4134; GFX11-FAKE16-NEXT: s_xor_b32 exec_lo, exec_lo, s0135; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB0_1136; GFX11-FAKE16-NEXT: ; %bb.2:137; GFX11-FAKE16-NEXT: s_mov_b32 exec_lo, s1138; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0)139; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, 16, v5140; GFX11-FAKE16-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_and_b32 v1, 0xffff, v6141; GFX11-FAKE16-NEXT: ds_store_2addr_b32 v2, v0, v1 offset0:7 offset1:8142bb:143 %i = call i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 poison)144 %i2 = call nsz arcp <3 x half> @llvm.amdgcn.struct.ptr.buffer.load.format.v3f16(ptr addrspace(8) %arg, i32 %arg1, i32 0, i32 0, i32 0)145 %i3 = bitcast <3 x half> %i2 to <3 x i16>146 %i4 = extractelement <3 x i16> %i3, i32 1147 %i5 = bitcast <3 x half> %i2 to <3 x i16>148 %i6 = extractelement <3 x i16> %i5, i32 2149 %i7 = zext i16 %i4 to i32150 %i8 = zext i16 %i6 to i32151 %i9 = add nuw nsw i32 0, 7152 %i10 = getelementptr [0 x i32], ptr addrspace(3) @esgs_ring, i32 0, i32 %i9153 store i32 %i7, ptr addrspace(3) %i10, align 4154 %i11 = add nuw nsw i32 0, 8155 %i12 = getelementptr [0 x i32], ptr addrspace(3) @esgs_ring, i32 0, i32 %i11156 store i32 %i8, ptr addrspace(3) %i12, align 4157 unreachable158}159; Function Attrs: nounwind readnone willreturn160declare i32 @llvm.amdgcn.mbcnt.hi(i32, i32) #0161; Function Attrs: nounwind readonly willreturn162declare <3 x half> @llvm.amdgcn.struct.ptr.buffer.load.format.v3f16(ptr addrspace(8), i32, i32, i32, i32 immarg) #1163attributes #0 = { nounwind readnone willreturn }164attributes #1 = { nounwind readonly willreturn }165;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:166; GFX11: {{.*}}167