193 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 52; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx900 < %s | FileCheck -check-prefixes=GFX9,GFX900 %s3; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 < %s | FileCheck -check-prefixes=GFX9,GFX942 %s4 5define <3 x float> @extract_subvector_v3f32_v33f32_elt30_0(ptr addrspace(1) %ptr) #0 {6; GFX900-LABEL: extract_subvector_v3f32_v33f32_elt30_0:7; GFX900: ; %bb.0:8; GFX900-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)9; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:96 glc10; GFX900-NEXT: s_waitcnt vmcnt(0)11; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:80 glc12; GFX900-NEXT: s_waitcnt vmcnt(0)13; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:64 glc14; GFX900-NEXT: s_waitcnt vmcnt(0)15; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:48 glc16; GFX900-NEXT: s_waitcnt vmcnt(0)17; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:32 glc18; GFX900-NEXT: s_waitcnt vmcnt(0)19; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:16 glc20; GFX900-NEXT: s_waitcnt vmcnt(0)21; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off glc22; GFX900-NEXT: s_waitcnt vmcnt(0)23; GFX900-NEXT: global_load_dword v2, v[0:1], off offset:128 glc24; GFX900-NEXT: s_waitcnt vmcnt(0)25; GFX900-NEXT: global_load_dwordx4 v[3:6], v[0:1], off offset:112 glc26; GFX900-NEXT: s_waitcnt vmcnt(0)27; GFX900-NEXT: v_mov_b32_e32 v0, v528; GFX900-NEXT: v_mov_b32_e32 v1, v629; GFX900-NEXT: s_setpc_b64 s[30:31]30;31; GFX942-LABEL: extract_subvector_v3f32_v33f32_elt30_0:32; GFX942: ; %bb.0:33; GFX942-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)34; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:96 sc0 sc135; GFX942-NEXT: s_waitcnt vmcnt(0)36; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:80 sc0 sc137; GFX942-NEXT: s_waitcnt vmcnt(0)38; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:64 sc0 sc139; GFX942-NEXT: s_waitcnt vmcnt(0)40; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:48 sc0 sc141; GFX942-NEXT: s_waitcnt vmcnt(0)42; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:32 sc0 sc143; GFX942-NEXT: s_waitcnt vmcnt(0)44; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:16 sc0 sc145; GFX942-NEXT: s_waitcnt vmcnt(0)46; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off sc0 sc147; GFX942-NEXT: s_waitcnt vmcnt(0)48; GFX942-NEXT: global_load_dword v2, v[0:1], off offset:128 sc0 sc149; GFX942-NEXT: s_waitcnt vmcnt(0)50; GFX942-NEXT: global_load_dwordx4 v[4:7], v[0:1], off offset:112 sc0 sc151; GFX942-NEXT: s_waitcnt vmcnt(0)52; GFX942-NEXT: v_mov_b32_e32 v0, v653; GFX942-NEXT: v_mov_b32_e32 v1, v754; GFX942-NEXT: s_setpc_b64 s[30:31]55 %val = load volatile <33 x float>, ptr addrspace(1) %ptr, align 456 %extract.subvector = shufflevector <33 x float> %val, <33 x float> poison, <3 x i32> <i32 30, i32 31, i32 32>57 ret <3 x float> %extract.subvector58}59 60define <3 x float> @extract_subvector_v3f32_v33f32_elt30_1(ptr addrspace(1) %ptr) #0 {61; GFX900-LABEL: extract_subvector_v3f32_v33f32_elt30_1:62; GFX900: ; %bb.0:63; GFX900-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)64; GFX900-NEXT: global_load_dwordx4 v[3:6], v[0:1], off65; GFX900-NEXT: global_load_dwordx4 v[7:10], v[0:1], off offset:11266; GFX900-NEXT: global_load_dword v2, v[0:1], off offset:12867; GFX900-NEXT: s_mov_b32 s4, 068; GFX900-NEXT: s_mov_b32 s5, s469; GFX900-NEXT: s_mov_b32 s6, s470; GFX900-NEXT: s_mov_b32 s7, s471; GFX900-NEXT: s_waitcnt vmcnt(2)72; GFX900-NEXT: buffer_store_dwordx4 v[3:6], off, s[4:7], 073; GFX900-NEXT: s_waitcnt vmcnt(2)74; GFX900-NEXT: v_mov_b32_e32 v0, v975; GFX900-NEXT: v_mov_b32_e32 v1, v1076; GFX900-NEXT: s_waitcnt vmcnt(0)77; GFX900-NEXT: s_setpc_b64 s[30:31]78;79; GFX942-LABEL: extract_subvector_v3f32_v33f32_elt30_1:80; GFX942: ; %bb.0:81; GFX942-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)82; GFX942-NEXT: global_load_dwordx4 v[4:7], v[0:1], off83; GFX942-NEXT: global_load_dwordx4 v[8:11], v[0:1], off offset:11284; GFX942-NEXT: global_load_dword v2, v[0:1], off offset:12885; GFX942-NEXT: s_mov_b32 s0, 086; GFX942-NEXT: s_mov_b32 s1, s087; GFX942-NEXT: s_mov_b32 s2, s088; GFX942-NEXT: s_mov_b32 s3, s089; GFX942-NEXT: s_waitcnt vmcnt(2)90; GFX942-NEXT: buffer_store_dwordx4 v[4:7], off, s[0:3], 091; GFX942-NEXT: s_waitcnt vmcnt(2)92; GFX942-NEXT: v_mov_b32_e32 v0, v1093; GFX942-NEXT: v_mov_b32_e32 v1, v1194; GFX942-NEXT: s_waitcnt vmcnt(0)95; GFX942-NEXT: s_setpc_b64 s[30:31]96 %val = load <33 x float>, ptr addrspace(1) %ptr, align 497 %val.slice.0 = shufflevector <33 x float> %val, <33 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>98 call void @llvm.amdgcn.raw.ptr.buffer.store.v4f32(<4 x float> %val.slice.0, ptr addrspace(8) null, i32 0, i32 0, i32 0)99 %val.slice.48 = shufflevector <33 x float> %val, <33 x float> poison, <3 x i32> <i32 30, i32 31, i32 32>100 ret <3 x float> %val.slice.48101}102 103define <6 x float> @extract_subvector_v6f32_v36f32_elt30(ptr addrspace(1) %ptr) #0 {104; GFX900-LABEL: extract_subvector_v6f32_v36f32_elt30:105; GFX900: ; %bb.0:106; GFX900-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)107; GFX900-NEXT: global_load_dwordx4 v[6:9], v[0:1], off108; GFX900-NEXT: global_load_dwordx4 v[10:13], v[0:1], off offset:112109; GFX900-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:128110; GFX900-NEXT: s_mov_b32 s4, 0111; GFX900-NEXT: s_mov_b32 s5, s4112; GFX900-NEXT: s_mov_b32 s6, s4113; GFX900-NEXT: s_mov_b32 s7, s4114; GFX900-NEXT: s_waitcnt vmcnt(2)115; GFX900-NEXT: buffer_store_dwordx4 v[6:9], off, s[4:7], 0116; GFX900-NEXT: s_waitcnt vmcnt(2)117; GFX900-NEXT: v_mov_b32_e32 v0, v12118; GFX900-NEXT: v_mov_b32_e32 v1, v13119; GFX900-NEXT: s_waitcnt vmcnt(0)120; GFX900-NEXT: s_setpc_b64 s[30:31]121;122; GFX942-LABEL: extract_subvector_v6f32_v36f32_elt30:123; GFX942: ; %bb.0:124; GFX942-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)125; GFX942-NEXT: global_load_dwordx4 v[6:9], v[0:1], off126; GFX942-NEXT: global_load_dwordx4 v[10:13], v[0:1], off offset:112127; GFX942-NEXT: global_load_dwordx4 v[2:5], v[0:1], off offset:128128; GFX942-NEXT: s_mov_b32 s0, 0129; GFX942-NEXT: s_mov_b32 s1, s0130; GFX942-NEXT: s_mov_b32 s2, s0131; GFX942-NEXT: s_mov_b32 s3, s0132; GFX942-NEXT: s_waitcnt vmcnt(2)133; GFX942-NEXT: buffer_store_dwordx4 v[6:9], off, s[0:3], 0134; GFX942-NEXT: s_waitcnt vmcnt(2)135; GFX942-NEXT: v_mov_b32_e32 v0, v12136; GFX942-NEXT: v_mov_b32_e32 v1, v13137; GFX942-NEXT: s_waitcnt vmcnt(0)138; GFX942-NEXT: s_setpc_b64 s[30:31]139 %val = load <36 x float>, ptr addrspace(1) %ptr, align 4140 %val.slice.0 = shufflevector <36 x float> %val, <36 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>141 call void @llvm.amdgcn.raw.ptr.buffer.store.v4f32(<4 x float> %val.slice.0, ptr addrspace(8) null, i32 0, i32 0, i32 0)142 %val.slice.1 = shufflevector <36 x float> %val, <36 x float> poison, <6 x i32> <i32 30, i32 31, i32 32, i32 33, i32 34, i32 35>143 ret <6 x float> %val.slice.1144}145 146define <3 x float> @issue153808_vector_extract_assert(ptr addrspace(1) %ptr) #0 {147; GFX900-LABEL: issue153808_vector_extract_assert:148; GFX900: ; %bb.0:149; GFX900-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)150; GFX900-NEXT: v_mov_b32_e32 v4, v1151; GFX900-NEXT: v_mov_b32_e32 v3, v0152; GFX900-NEXT: global_load_dwordx4 v[5:8], v[3:4], off153; GFX900-NEXT: global_load_dwordx3 v[0:2], v[3:4], off offset:192154; GFX900-NEXT: s_mov_b32 s4, 0155; GFX900-NEXT: s_mov_b32 s5, s4156; GFX900-NEXT: s_mov_b32 s6, s4157; GFX900-NEXT: s_mov_b32 s7, s4158; GFX900-NEXT: s_waitcnt vmcnt(1)159; GFX900-NEXT: buffer_store_dwordx4 v[5:8], off, s[4:7], 0160; GFX900-NEXT: s_waitcnt vmcnt(0)161; GFX900-NEXT: s_setpc_b64 s[30:31]162;163; GFX942-LABEL: issue153808_vector_extract_assert:164; GFX942: ; %bb.0:165; GFX942-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)166; GFX942-NEXT: global_load_dwordx4 v[6:9], v[0:1], off167; GFX942-NEXT: global_load_dwordx3 v[2:4], v[0:1], off offset:192168; GFX942-NEXT: s_mov_b32 s0, 0169; GFX942-NEXT: s_mov_b32 s1, s0170; GFX942-NEXT: s_mov_b32 s2, s0171; GFX942-NEXT: s_mov_b32 s3, s0172; GFX942-NEXT: s_waitcnt vmcnt(1)173; GFX942-NEXT: buffer_store_dwordx4 v[6:9], off, s[0:3], 0174; GFX942-NEXT: s_waitcnt vmcnt(1)175; GFX942-NEXT: v_mov_b32_e32 v0, v2176; GFX942-NEXT: v_mov_b32_e32 v1, v3177; GFX942-NEXT: v_mov_b32_e32 v2, v4178; GFX942-NEXT: s_waitcnt vmcnt(0)179; GFX942-NEXT: s_setpc_b64 s[30:31]180 %val = load <51 x float>, ptr addrspace(1) %ptr, align 4181 %val.slice.0 = shufflevector <51 x float> %val, <51 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>182 call void @llvm.amdgcn.raw.ptr.buffer.store.v4f32(<4 x float> %val.slice.0, ptr addrspace(8) null, i32 0, i32 0, i32 0)183 %val.slice.48 = shufflevector <51 x float> %val, <51 x float> poison, <3 x i32> <i32 48, i32 49, i32 50>184 ret <3 x float> %val.slice.48185}186 187declare void @llvm.amdgcn.raw.ptr.buffer.store.v4f32(<4 x float>, ptr addrspace(8) writeonly captures(none), i32, i32, i32 immarg) #1188 189attributes #0 = { nounwind }190attributes #1 = { nocallback nofree nosync nounwind willreturn memory(argmem: write) }191;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:192; GFX9: {{.*}}193