brintos

brintos / llvm-project-archived public Read only

0
0
Text · 19.8 KiB · 13a96cf Raw
439 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 42; RUN: llc -mtriple=amdgcn -mcpu=gfx950 < %s | FileCheck -enable-var-scope --check-prefix=GCN %s3 4; FIXME: bfloat vector arguments are broken in globalisel.5; https://github.com/llvm/llvm-project/issues/770556 7declare <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat>, <8 x bfloat>, <16 x float>, i32 immarg, i32 immarg, i32 immarg)8 9; --------------------------------------------------------------------10; llvm.amdgcn.mfma.f32.32x32x16.bf1611; --------------------------------------------------------------------12 13define amdgpu_kernel void @test_mfma_f32_32x32x16_bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2) #1 {14; GCN-LABEL: test_mfma_f32_32x32x16_bf16:15; GCN:       ; %bb.0:16; GCN-NEXT:    s_load_dwordx8 s[24:31], s[4:5], 0x2417; GCN-NEXT:    s_load_dwordx16 s[8:23], s[4:5], 0x6418; GCN-NEXT:    v_mov_b64_e32 v[0:1], 4819; GCN-NEXT:    v_mov_b64_e32 v[2:3], 3220; GCN-NEXT:    v_mov_b64_e32 v[4:5], 1621; GCN-NEXT:    s_waitcnt lgkmcnt(0)22; GCN-NEXT:    v_mov_b64_e32 v[8:9], s[24:25]23; GCN-NEXT:    v_mov_b64_e32 v[10:11], s[26:27]24; GCN-NEXT:    v_mov_b64_e32 v[12:13], s[28:29]25; GCN-NEXT:    v_accvgpr_write_b32 a0, s826; GCN-NEXT:    v_mov_b64_e32 v[14:15], s[30:31]27; GCN-NEXT:    v_accvgpr_write_b32 a1, s928; GCN-NEXT:    v_accvgpr_write_b32 a2, s1029; GCN-NEXT:    v_accvgpr_write_b32 a3, s1130; GCN-NEXT:    v_accvgpr_write_b32 a4, s1231; GCN-NEXT:    v_accvgpr_write_b32 a5, s1332; GCN-NEXT:    v_accvgpr_write_b32 a6, s1433; GCN-NEXT:    v_accvgpr_write_b32 a7, s1534; GCN-NEXT:    v_accvgpr_write_b32 a8, s1635; GCN-NEXT:    v_accvgpr_write_b32 a9, s1736; GCN-NEXT:    v_accvgpr_write_b32 a10, s1837; GCN-NEXT:    v_accvgpr_write_b32 a11, s1938; GCN-NEXT:    v_accvgpr_write_b32 a12, s2039; GCN-NEXT:    v_accvgpr_write_b32 a13, s2140; GCN-NEXT:    v_accvgpr_write_b32 a14, s2241; GCN-NEXT:    v_accvgpr_write_b32 a15, s2342; GCN-NEXT:    v_mov_b32_e32 v16, s1643; GCN-NEXT:    v_mov_b32_e32 v17, s1744; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 a[16:31], v[8:11], v[12:15], a[0:15]45; GCN-NEXT:    v_mov_b32_e32 v18, s1846; GCN-NEXT:    v_mov_b32_e32 v19, s1947; GCN-NEXT:    v_mov_b32_e32 v8, s2048; GCN-NEXT:    v_mov_b32_e32 v9, s2149; GCN-NEXT:    v_mov_b32_e32 v10, s2250; GCN-NEXT:    v_mov_b32_e32 v11, s2351; GCN-NEXT:    v_mov_b64_e32 v[6:7], 052; GCN-NEXT:    s_nop 453; GCN-NEXT:    global_store_dwordx4 v[0:1], a[28:31], off sc0 sc154; GCN-NEXT:    s_waitcnt vmcnt(0)55; GCN-NEXT:    global_store_dwordx4 v[2:3], a[24:27], off sc0 sc156; GCN-NEXT:    s_waitcnt vmcnt(0)57; GCN-NEXT:    global_store_dwordx4 v[4:5], a[20:23], off sc0 sc158; GCN-NEXT:    s_waitcnt vmcnt(0)59; GCN-NEXT:    global_store_dwordx4 v[6:7], a[16:19], off sc0 sc160; GCN-NEXT:    s_waitcnt vmcnt(0)61; GCN-NEXT:    global_store_dwordx4 v[2:3], v[16:19], off sc0 sc162; GCN-NEXT:    s_waitcnt vmcnt(0)63; GCN-NEXT:    global_store_dwordx4 v[0:1], v[8:11], off sc0 sc164; GCN-NEXT:    s_waitcnt vmcnt(0)65; GCN-NEXT:    v_mov_b32_e32 v0, s866; GCN-NEXT:    v_mov_b32_e32 v1, s967; GCN-NEXT:    v_mov_b32_e32 v2, s1068; GCN-NEXT:    v_mov_b32_e32 v3, s1169; GCN-NEXT:    global_store_dwordx4 v[6:7], v[0:3], off sc0 sc170; GCN-NEXT:    s_waitcnt vmcnt(0)71; GCN-NEXT:    s_nop 072; GCN-NEXT:    v_mov_b32_e32 v0, s1273; GCN-NEXT:    v_mov_b32_e32 v1, s1374; GCN-NEXT:    v_mov_b32_e32 v2, s1475; GCN-NEXT:    v_mov_b32_e32 v3, s1576; GCN-NEXT:    global_store_dwordx4 v[4:5], v[0:3], off sc0 sc177; GCN-NEXT:    s_waitcnt vmcnt(0)78; GCN-NEXT:    s_endpgm79  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 0, i32 0, i32 0)80  store volatile <16 x float> %result, ptr addrspace(1) null81  store volatile <16 x float> %arg2, ptr addrspace(1) null82  ret void83}84 85define amdgpu_kernel void @test_mfma_f32_32x32x16_bf16__flags(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2) #1 {86; GCN-LABEL: test_mfma_f32_32x32x16_bf16__flags:87; GCN:       ; %bb.0:88; GCN-NEXT:    s_load_dwordx8 s[24:31], s[4:5], 0x2489; GCN-NEXT:    s_load_dwordx16 s[8:23], s[4:5], 0x6490; GCN-NEXT:    v_mov_b64_e32 v[0:1], 4891; GCN-NEXT:    v_mov_b64_e32 v[2:3], 3292; GCN-NEXT:    v_mov_b64_e32 v[4:5], 1693; GCN-NEXT:    s_waitcnt lgkmcnt(0)94; GCN-NEXT:    v_mov_b64_e32 v[8:9], s[24:25]95; GCN-NEXT:    v_mov_b64_e32 v[10:11], s[26:27]96; GCN-NEXT:    v_mov_b64_e32 v[12:13], s[28:29]97; GCN-NEXT:    v_accvgpr_write_b32 a0, s898; GCN-NEXT:    v_mov_b64_e32 v[14:15], s[30:31]99; GCN-NEXT:    v_accvgpr_write_b32 a1, s9100; GCN-NEXT:    v_accvgpr_write_b32 a2, s10101; GCN-NEXT:    v_accvgpr_write_b32 a3, s11102; GCN-NEXT:    v_accvgpr_write_b32 a4, s12103; GCN-NEXT:    v_accvgpr_write_b32 a5, s13104; GCN-NEXT:    v_accvgpr_write_b32 a6, s14105; GCN-NEXT:    v_accvgpr_write_b32 a7, s15106; GCN-NEXT:    v_accvgpr_write_b32 a8, s16107; GCN-NEXT:    v_accvgpr_write_b32 a9, s17108; GCN-NEXT:    v_accvgpr_write_b32 a10, s18109; GCN-NEXT:    v_accvgpr_write_b32 a11, s19110; GCN-NEXT:    v_accvgpr_write_b32 a12, s20111; GCN-NEXT:    v_accvgpr_write_b32 a13, s21112; GCN-NEXT:    v_accvgpr_write_b32 a14, s22113; GCN-NEXT:    v_accvgpr_write_b32 a15, s23114; GCN-NEXT:    v_mov_b32_e32 v16, s16115; GCN-NEXT:    v_mov_b32_e32 v17, s17116; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 a[16:31], v[8:11], v[12:15], a[0:15] cbsz:2 abid:3 blgp:1117; GCN-NEXT:    v_mov_b32_e32 v18, s18118; GCN-NEXT:    v_mov_b32_e32 v19, s19119; GCN-NEXT:    v_mov_b32_e32 v8, s20120; GCN-NEXT:    v_mov_b32_e32 v9, s21121; GCN-NEXT:    v_mov_b32_e32 v10, s22122; GCN-NEXT:    v_mov_b32_e32 v11, s23123; GCN-NEXT:    v_mov_b64_e32 v[6:7], 0124; GCN-NEXT:    s_nop 4125; GCN-NEXT:    global_store_dwordx4 v[0:1], a[28:31], off sc0 sc1126; GCN-NEXT:    s_waitcnt vmcnt(0)127; GCN-NEXT:    global_store_dwordx4 v[2:3], a[24:27], off sc0 sc1128; GCN-NEXT:    s_waitcnt vmcnt(0)129; GCN-NEXT:    global_store_dwordx4 v[4:5], a[20:23], off sc0 sc1130; GCN-NEXT:    s_waitcnt vmcnt(0)131; GCN-NEXT:    global_store_dwordx4 v[6:7], a[16:19], off sc0 sc1132; GCN-NEXT:    s_waitcnt vmcnt(0)133; GCN-NEXT:    global_store_dwordx4 v[2:3], v[16:19], off sc0 sc1134; GCN-NEXT:    s_waitcnt vmcnt(0)135; GCN-NEXT:    global_store_dwordx4 v[0:1], v[8:11], off sc0 sc1136; GCN-NEXT:    s_waitcnt vmcnt(0)137; GCN-NEXT:    v_mov_b32_e32 v0, s8138; GCN-NEXT:    v_mov_b32_e32 v1, s9139; GCN-NEXT:    v_mov_b32_e32 v2, s10140; GCN-NEXT:    v_mov_b32_e32 v3, s11141; GCN-NEXT:    global_store_dwordx4 v[6:7], v[0:3], off sc0 sc1142; GCN-NEXT:    s_waitcnt vmcnt(0)143; GCN-NEXT:    s_nop 0144; GCN-NEXT:    v_mov_b32_e32 v0, s12145; GCN-NEXT:    v_mov_b32_e32 v1, s13146; GCN-NEXT:    v_mov_b32_e32 v2, s14147; GCN-NEXT:    v_mov_b32_e32 v3, s15148; GCN-NEXT:    global_store_dwordx4 v[4:5], v[0:3], off sc0 sc1149; GCN-NEXT:    s_waitcnt vmcnt(0)150; GCN-NEXT:    s_endpgm151  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 2, i32 3, i32 1)152  store volatile <16 x float> %result, ptr addrspace(1) null153  store volatile <16 x float> %arg2, ptr addrspace(1) null154  ret void155}156 157define <16 x float> @test_mfma_f32_32x32x16_bf16__mac(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2) {158; GCN-LABEL: test_mfma_f32_32x32x16_bf16__mac:159; GCN:       ; %bb.0:160; GCN-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)161; GCN-NEXT:    v_accvgpr_write_b32 a0, v8162; GCN-NEXT:    v_accvgpr_write_b32 a1, v9163; GCN-NEXT:    v_accvgpr_write_b32 a2, v10164; GCN-NEXT:    v_accvgpr_write_b32 a3, v11165; GCN-NEXT:    v_accvgpr_write_b32 a4, v12166; GCN-NEXT:    v_accvgpr_write_b32 a5, v13167; GCN-NEXT:    v_accvgpr_write_b32 a6, v14168; GCN-NEXT:    v_accvgpr_write_b32 a7, v15169; GCN-NEXT:    v_accvgpr_write_b32 a8, v16170; GCN-NEXT:    v_accvgpr_write_b32 a9, v17171; GCN-NEXT:    v_accvgpr_write_b32 a10, v18172; GCN-NEXT:    v_accvgpr_write_b32 a11, v19173; GCN-NEXT:    v_accvgpr_write_b32 a12, v20174; GCN-NEXT:    v_accvgpr_write_b32 a13, v21175; GCN-NEXT:    v_accvgpr_write_b32 a14, v22176; GCN-NEXT:    v_accvgpr_write_b32 a15, v23177; GCN-NEXT:    s_nop 1178; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 a[0:15], v[0:3], v[4:7], a[0:15]179; GCN-NEXT:    s_nop 11180; GCN-NEXT:    v_accvgpr_read_b32 v0, a0181; GCN-NEXT:    v_accvgpr_read_b32 v1, a1182; GCN-NEXT:    v_accvgpr_read_b32 v2, a2183; GCN-NEXT:    v_accvgpr_read_b32 v3, a3184; GCN-NEXT:    v_accvgpr_read_b32 v4, a4185; GCN-NEXT:    v_accvgpr_read_b32 v5, a5186; GCN-NEXT:    v_accvgpr_read_b32 v6, a6187; GCN-NEXT:    v_accvgpr_read_b32 v7, a7188; GCN-NEXT:    v_accvgpr_read_b32 v8, a8189; GCN-NEXT:    v_accvgpr_read_b32 v9, a9190; GCN-NEXT:    v_accvgpr_read_b32 v10, a10191; GCN-NEXT:    v_accvgpr_read_b32 v11, a11192; GCN-NEXT:    v_accvgpr_read_b32 v12, a12193; GCN-NEXT:    v_accvgpr_read_b32 v13, a13194; GCN-NEXT:    v_accvgpr_read_b32 v14, a14195; GCN-NEXT:    v_accvgpr_read_b32 v15, a15196; GCN-NEXT:    s_setpc_b64 s[30:31]197  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 0, i32 0, i32 0)198  ret <16 x float> %result199}200 201define <16 x float> @test_mfma_f32_32x32x16_bf16__mac__flags(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2) {202; GCN-LABEL: test_mfma_f32_32x32x16_bf16__mac__flags:203; GCN:       ; %bb.0:204; GCN-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)205; GCN-NEXT:    v_accvgpr_write_b32 a0, v8206; GCN-NEXT:    v_accvgpr_write_b32 a1, v9207; GCN-NEXT:    v_accvgpr_write_b32 a2, v10208; GCN-NEXT:    v_accvgpr_write_b32 a3, v11209; GCN-NEXT:    v_accvgpr_write_b32 a4, v12210; GCN-NEXT:    v_accvgpr_write_b32 a5, v13211; GCN-NEXT:    v_accvgpr_write_b32 a6, v14212; GCN-NEXT:    v_accvgpr_write_b32 a7, v15213; GCN-NEXT:    v_accvgpr_write_b32 a8, v16214; GCN-NEXT:    v_accvgpr_write_b32 a9, v17215; GCN-NEXT:    v_accvgpr_write_b32 a10, v18216; GCN-NEXT:    v_accvgpr_write_b32 a11, v19217; GCN-NEXT:    v_accvgpr_write_b32 a12, v20218; GCN-NEXT:    v_accvgpr_write_b32 a13, v21219; GCN-NEXT:    v_accvgpr_write_b32 a14, v22220; GCN-NEXT:    v_accvgpr_write_b32 a15, v23221; GCN-NEXT:    s_nop 1222; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 a[0:15], v[0:3], v[4:7], a[0:15] cbsz:1 abid:1 blgp:1223; GCN-NEXT:    s_nop 11224; GCN-NEXT:    v_accvgpr_read_b32 v0, a0225; GCN-NEXT:    v_accvgpr_read_b32 v1, a1226; GCN-NEXT:    v_accvgpr_read_b32 v2, a2227; GCN-NEXT:    v_accvgpr_read_b32 v3, a3228; GCN-NEXT:    v_accvgpr_read_b32 v4, a4229; GCN-NEXT:    v_accvgpr_read_b32 v5, a5230; GCN-NEXT:    v_accvgpr_read_b32 v6, a6231; GCN-NEXT:    v_accvgpr_read_b32 v7, a7232; GCN-NEXT:    v_accvgpr_read_b32 v8, a8233; GCN-NEXT:    v_accvgpr_read_b32 v9, a9234; GCN-NEXT:    v_accvgpr_read_b32 v10, a10235; GCN-NEXT:    v_accvgpr_read_b32 v11, a11236; GCN-NEXT:    v_accvgpr_read_b32 v12, a12237; GCN-NEXT:    v_accvgpr_read_b32 v13, a13238; GCN-NEXT:    v_accvgpr_read_b32 v14, a14239; GCN-NEXT:    v_accvgpr_read_b32 v15, a15240; GCN-NEXT:    s_setpc_b64 s[30:31]241  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 1, i32 1, i32 1)242  ret <16 x float> %result243}244 245define amdgpu_kernel void @test_mfma_f32_32x32x16_bf16__vgprcd(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, ptr addrspace(1) %out) #0 {246; GCN-LABEL: test_mfma_f32_32x32x16_bf16__vgprcd:247; GCN:       ; %bb.0:248; GCN-NEXT:    s_load_dwordx8 s[24:31], s[4:5], 0x24249; GCN-NEXT:    s_load_dwordx16 s[8:23], s[4:5], 0x64250; GCN-NEXT:    s_load_dwordx2 s[0:1], s[4:5], 0xa4251; GCN-NEXT:    v_mov_b32_e32 v36, 0252; GCN-NEXT:    s_waitcnt lgkmcnt(0)253; GCN-NEXT:    v_mov_b64_e32 v[40:41], s[26:27]254; GCN-NEXT:    v_mov_b64_e32 v[38:39], s[24:25]255; GCN-NEXT:    v_mov_b64_e32 v[44:45], s[30:31]256; GCN-NEXT:    v_mov_b64_e32 v[30:31], s[22:23]257; GCN-NEXT:    v_mov_b64_e32 v[42:43], s[28:29]258; GCN-NEXT:    v_mov_b64_e32 v[28:29], s[20:21]259; GCN-NEXT:    v_mov_b64_e32 v[26:27], s[18:19]260; GCN-NEXT:    v_mov_b64_e32 v[24:25], s[16:17]261; GCN-NEXT:    v_mov_b64_e32 v[22:23], s[14:15]262; GCN-NEXT:    v_mov_b64_e32 v[20:21], s[12:13]263; GCN-NEXT:    v_mov_b64_e32 v[18:19], s[10:11]264; GCN-NEXT:    v_mov_b64_e32 v[16:17], s[8:9]265; GCN-NEXT:    v_mov_b32_e32 v32, s20266; GCN-NEXT:    v_mov_b32_e32 v33, s21267; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 v[0:15], v[38:41], v[42:45], v[16:31]268; GCN-NEXT:    v_mov_b32_e32 v34, s22269; GCN-NEXT:    v_mov_b32_e32 v35, s23270; GCN-NEXT:    global_store_dwordx4 v36, v[32:35], s[0:1] offset:48 sc0 sc1271; GCN-NEXT:    s_waitcnt vmcnt(0)272; GCN-NEXT:    s_nop 2273; GCN-NEXT:    v_mov_b32_e32 v16, s16274; GCN-NEXT:    v_mov_b32_e32 v17, s17275; GCN-NEXT:    v_mov_b32_e32 v18, s18276; GCN-NEXT:    v_mov_b32_e32 v19, s19277; GCN-NEXT:    global_store_dwordx4 v36, v[16:19], s[0:1] offset:32 sc0 sc1278; GCN-NEXT:    s_waitcnt vmcnt(0)279; GCN-NEXT:    s_nop 0280; GCN-NEXT:    v_mov_b32_e32 v16, s12281; GCN-NEXT:    v_mov_b32_e32 v17, s13282; GCN-NEXT:    v_mov_b32_e32 v18, s14283; GCN-NEXT:    v_mov_b32_e32 v19, s15284; GCN-NEXT:    global_store_dwordx4 v36, v[16:19], s[0:1] offset:16 sc0 sc1285; GCN-NEXT:    s_waitcnt vmcnt(0)286; GCN-NEXT:    s_nop 0287; GCN-NEXT:    v_mov_b32_e32 v16, s8288; GCN-NEXT:    v_mov_b32_e32 v17, s9289; GCN-NEXT:    v_mov_b32_e32 v18, s10290; GCN-NEXT:    v_mov_b32_e32 v19, s11291; GCN-NEXT:    global_store_dwordx4 v36, v[16:19], s[0:1] sc0 sc1292; GCN-NEXT:    s_waitcnt vmcnt(0)293; GCN-NEXT:    global_store_dwordx4 v36, v[8:11], s[0:1] offset:32 sc0 sc1294; GCN-NEXT:    s_waitcnt vmcnt(0)295; GCN-NEXT:    global_store_dwordx4 v36, v[12:15], s[0:1] offset:48 sc0 sc1296; GCN-NEXT:    s_waitcnt vmcnt(0)297; GCN-NEXT:    global_store_dwordx4 v36, v[0:3], s[0:1] sc0 sc1298; GCN-NEXT:    s_waitcnt vmcnt(0)299; GCN-NEXT:    global_store_dwordx4 v36, v[4:7], s[0:1] offset:16 sc0 sc1300; GCN-NEXT:    s_waitcnt vmcnt(0)301; GCN-NEXT:    s_endpgm302  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 0, i32 0, i32 0)303  store volatile <16 x float> %arg2, ptr addrspace(1) %out304  store volatile <16 x float> %result, ptr addrspace(1) %out305  ret void306}307 308define amdgpu_kernel void @test_mfma_f32_32x32x16_bf16__vgprcd__flags(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, ptr addrspace(1) %out) #0 {309; GCN-LABEL: test_mfma_f32_32x32x16_bf16__vgprcd__flags:310; GCN:       ; %bb.0:311; GCN-NEXT:    s_load_dwordx8 s[24:31], s[4:5], 0x24312; GCN-NEXT:    s_load_dwordx16 s[8:23], s[4:5], 0x64313; GCN-NEXT:    s_load_dwordx2 s[0:1], s[4:5], 0xa4314; GCN-NEXT:    v_mov_b32_e32 v36, 0315; GCN-NEXT:    s_waitcnt lgkmcnt(0)316; GCN-NEXT:    v_mov_b64_e32 v[40:41], s[26:27]317; GCN-NEXT:    v_mov_b64_e32 v[38:39], s[24:25]318; GCN-NEXT:    v_mov_b64_e32 v[44:45], s[30:31]319; GCN-NEXT:    v_mov_b64_e32 v[30:31], s[22:23]320; GCN-NEXT:    v_mov_b64_e32 v[42:43], s[28:29]321; GCN-NEXT:    v_mov_b64_e32 v[28:29], s[20:21]322; GCN-NEXT:    v_mov_b64_e32 v[26:27], s[18:19]323; GCN-NEXT:    v_mov_b64_e32 v[24:25], s[16:17]324; GCN-NEXT:    v_mov_b64_e32 v[22:23], s[14:15]325; GCN-NEXT:    v_mov_b64_e32 v[20:21], s[12:13]326; GCN-NEXT:    v_mov_b64_e32 v[18:19], s[10:11]327; GCN-NEXT:    v_mov_b64_e32 v[16:17], s[8:9]328; GCN-NEXT:    v_mov_b32_e32 v32, s20329; GCN-NEXT:    v_mov_b32_e32 v33, s21330; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 v[0:15], v[38:41], v[42:45], v[16:31] cbsz:1 abid:2 blgp:3331; GCN-NEXT:    v_mov_b32_e32 v34, s22332; GCN-NEXT:    v_mov_b32_e32 v35, s23333; GCN-NEXT:    global_store_dwordx4 v36, v[32:35], s[0:1] offset:48 sc0 sc1334; GCN-NEXT:    s_waitcnt vmcnt(0)335; GCN-NEXT:    s_nop 2336; GCN-NEXT:    v_mov_b32_e32 v16, s16337; GCN-NEXT:    v_mov_b32_e32 v17, s17338; GCN-NEXT:    v_mov_b32_e32 v18, s18339; GCN-NEXT:    v_mov_b32_e32 v19, s19340; GCN-NEXT:    global_store_dwordx4 v36, v[16:19], s[0:1] offset:32 sc0 sc1341; GCN-NEXT:    s_waitcnt vmcnt(0)342; GCN-NEXT:    s_nop 0343; GCN-NEXT:    v_mov_b32_e32 v16, s12344; GCN-NEXT:    v_mov_b32_e32 v17, s13345; GCN-NEXT:    v_mov_b32_e32 v18, s14346; GCN-NEXT:    v_mov_b32_e32 v19, s15347; GCN-NEXT:    global_store_dwordx4 v36, v[16:19], s[0:1] offset:16 sc0 sc1348; GCN-NEXT:    s_waitcnt vmcnt(0)349; GCN-NEXT:    s_nop 0350; GCN-NEXT:    v_mov_b32_e32 v16, s8351; GCN-NEXT:    v_mov_b32_e32 v17, s9352; GCN-NEXT:    v_mov_b32_e32 v18, s10353; GCN-NEXT:    v_mov_b32_e32 v19, s11354; GCN-NEXT:    global_store_dwordx4 v36, v[16:19], s[0:1] sc0 sc1355; GCN-NEXT:    s_waitcnt vmcnt(0)356; GCN-NEXT:    global_store_dwordx4 v36, v[8:11], s[0:1] offset:32 sc0 sc1357; GCN-NEXT:    s_waitcnt vmcnt(0)358; GCN-NEXT:    global_store_dwordx4 v36, v[12:15], s[0:1] offset:48 sc0 sc1359; GCN-NEXT:    s_waitcnt vmcnt(0)360; GCN-NEXT:    global_store_dwordx4 v36, v[0:3], s[0:1] sc0 sc1361; GCN-NEXT:    s_waitcnt vmcnt(0)362; GCN-NEXT:    global_store_dwordx4 v36, v[4:7], s[0:1] offset:16 sc0 sc1363; GCN-NEXT:    s_waitcnt vmcnt(0)364; GCN-NEXT:    s_endpgm365  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 1, i32 2, i32 3)366  store volatile <16 x float> %arg2, ptr addrspace(1) %out367  store volatile <16 x float> %result, ptr addrspace(1) %out368  ret void369}370 371define amdgpu_kernel void @test_mfma_f32_32x32x16_bf16__vgprcd_mac(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, ptr addrspace(1) %out) #0 {372; GCN-LABEL: test_mfma_f32_32x32x16_bf16__vgprcd_mac:373; GCN:       ; %bb.0:374; GCN-NEXT:    s_load_dwordx8 s[24:31], s[4:5], 0x24375; GCN-NEXT:    s_load_dwordx16 s[8:23], s[4:5], 0x64376; GCN-NEXT:    s_load_dwordx2 s[0:1], s[4:5], 0xa4377; GCN-NEXT:    s_waitcnt lgkmcnt(0)378; GCN-NEXT:    v_mov_b64_e32 v[16:17], s[24:25]379; GCN-NEXT:    v_mov_b64_e32 v[18:19], s[26:27]380; GCN-NEXT:    v_mov_b64_e32 v[20:21], s[28:29]381; GCN-NEXT:    v_mov_b64_e32 v[0:1], s[8:9]382; GCN-NEXT:    v_mov_b64_e32 v[22:23], s[30:31]383; GCN-NEXT:    v_mov_b64_e32 v[2:3], s[10:11]384; GCN-NEXT:    v_mov_b64_e32 v[4:5], s[12:13]385; GCN-NEXT:    v_mov_b64_e32 v[6:7], s[14:15]386; GCN-NEXT:    v_mov_b64_e32 v[8:9], s[16:17]387; GCN-NEXT:    v_mov_b64_e32 v[10:11], s[18:19]388; GCN-NEXT:    v_mov_b64_e32 v[12:13], s[20:21]389; GCN-NEXT:    v_mov_b64_e32 v[14:15], s[22:23]390; GCN-NEXT:    s_nop 1391; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 v[0:15], v[16:19], v[20:23], v[0:15]392; GCN-NEXT:    v_mov_b32_e32 v16, 0393; GCN-NEXT:    s_nop 10394; GCN-NEXT:    global_store_dwordx4 v16, v[12:15], s[0:1] offset:48395; GCN-NEXT:    global_store_dwordx4 v16, v[8:11], s[0:1] offset:32396; GCN-NEXT:    global_store_dwordx4 v16, v[4:7], s[0:1] offset:16397; GCN-NEXT:    global_store_dwordx4 v16, v[0:3], s[0:1]398; GCN-NEXT:    s_endpgm399  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 0, i32 0, i32 0)400  store <16 x float> %result, ptr addrspace(1) %out401  ret void402}403 404define amdgpu_kernel void @test_mfma_f32_32x32x16_bf16__vgprcd_mac_flags(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, ptr addrspace(1) %out) #0 {405; GCN-LABEL: test_mfma_f32_32x32x16_bf16__vgprcd_mac_flags:406; GCN:       ; %bb.0:407; GCN-NEXT:    s_load_dwordx8 s[24:31], s[4:5], 0x24408; GCN-NEXT:    s_load_dwordx16 s[8:23], s[4:5], 0x64409; GCN-NEXT:    s_load_dwordx2 s[0:1], s[4:5], 0xa4410; GCN-NEXT:    s_waitcnt lgkmcnt(0)411; GCN-NEXT:    v_mov_b64_e32 v[16:17], s[24:25]412; GCN-NEXT:    v_mov_b64_e32 v[18:19], s[26:27]413; GCN-NEXT:    v_mov_b64_e32 v[20:21], s[28:29]414; GCN-NEXT:    v_mov_b64_e32 v[0:1], s[8:9]415; GCN-NEXT:    v_mov_b64_e32 v[22:23], s[30:31]416; GCN-NEXT:    v_mov_b64_e32 v[2:3], s[10:11]417; GCN-NEXT:    v_mov_b64_e32 v[4:5], s[12:13]418; GCN-NEXT:    v_mov_b64_e32 v[6:7], s[14:15]419; GCN-NEXT:    v_mov_b64_e32 v[8:9], s[16:17]420; GCN-NEXT:    v_mov_b64_e32 v[10:11], s[18:19]421; GCN-NEXT:    v_mov_b64_e32 v[12:13], s[20:21]422; GCN-NEXT:    v_mov_b64_e32 v[14:15], s[22:23]423; GCN-NEXT:    s_nop 1424; GCN-NEXT:    v_mfma_f32_32x32x16_bf16 v[0:15], v[16:19], v[20:23], v[0:15] cbsz:3 abid:2 blgp:1425; GCN-NEXT:    v_mov_b32_e32 v16, 0426; GCN-NEXT:    s_nop 10427; GCN-NEXT:    global_store_dwordx4 v16, v[12:15], s[0:1] offset:48428; GCN-NEXT:    global_store_dwordx4 v16, v[8:11], s[0:1] offset:32429; GCN-NEXT:    global_store_dwordx4 v16, v[4:7], s[0:1] offset:16430; GCN-NEXT:    global_store_dwordx4 v16, v[0:3], s[0:1]431; GCN-NEXT:    s_endpgm432  %result = call <16 x float> @llvm.amdgcn.mfma.f32.32x32x16.bf16(<8 x bfloat> %arg0, <8 x bfloat> %arg1, <16 x float> %arg2, i32 3, i32 2, i32 1)433  store <16 x float> %result, ptr addrspace(1) %out434  ret void435}436 437attributes #0 = { "amdgpu-flat-work-group-size"="512,512" "amdgpu-agpr-alloc"="0,0" }438attributes #1 = { "amdgpu-flat-work-group-size"="1,64" }439