446 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py2; RUN: llc -mtriple=amdgcn -mcpu=gfx1010 < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX10 %s3; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX11 %s4 5define amdgpu_ps float @_amdgpu_ps_main() #0 {6; GFX10-LABEL: _amdgpu_ps_main:7; GFX10: ; %bb.0: ; %.entry8; GFX10-NEXT: s_mov_b32 s0, exec_lo9; GFX10-NEXT: s_wqm_b32 exec_lo, exec_lo10; GFX10-NEXT: image_sample v[0:1], v[0:1], s[0:7], s[0:3] dmask:0x3 dim:SQ_RSRC_IMG_2D11; GFX10-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)12; GFX10-NEXT: s_and_b32 exec_lo, exec_lo, s013; GFX10-NEXT: s_waitcnt vmcnt(0)14; GFX10-NEXT: s_clause 0x115; GFX10-NEXT: image_sample v2, v[0:1], s[0:7], s[0:3] dmask:0x4 dim:SQ_RSRC_IMG_2D16; GFX10-NEXT: image_sample v3, v[0:1], s[0:7], s[0:3] dmask:0x1 dim:SQ_RSRC_IMG_2D17; GFX10-NEXT: v_mov_b32_e32 v4, 018; GFX10-NEXT: s_waitcnt vmcnt(0)19; GFX10-NEXT: image_load_mip v4, v[2:4], s[0:7] dmask:0x4 dim:SQ_RSRC_IMG_2D unorm20; GFX10-NEXT: s_clause 0x321; GFX10-NEXT: s_buffer_load_dword s24, s[0:3], 0x5c22; GFX10-NEXT: s_buffer_load_dword s28, s[0:3], 0x7c23; GFX10-NEXT: s_buffer_load_dword s29, s[0:3], 0xc024; GFX10-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)25; GFX10-NEXT: s_nop 026; GFX10-NEXT: s_buffer_load_dwordx4 s[0:3], s[0:3], 0x4027; GFX10-NEXT: s_waitcnt lgkmcnt(0)28; GFX10-NEXT: s_clause 0x129; GFX10-NEXT: s_buffer_load_dwordx4 s[4:7], s[0:3], 0x5030; GFX10-NEXT: s_nop 031; GFX10-NEXT: s_buffer_load_dword s0, s[0:3], 0x2c32; GFX10-NEXT: v_sub_f32_e64 v5, s24, s2833; GFX10-NEXT: s_waitcnt lgkmcnt(0)34; GFX10-NEXT: s_clause 0x435; GFX10-NEXT: s_buffer_load_dwordx4 s[8:11], s[0:3], 0x6036; GFX10-NEXT: s_buffer_load_dwordx4 s[12:15], s[0:3], 0x2037; GFX10-NEXT: s_buffer_load_dwordx4 s[16:19], s[0:3], 0x038; GFX10-NEXT: s_buffer_load_dwordx4 s[20:23], s[0:3], 0x7039; GFX10-NEXT: s_buffer_load_dwordx4 s[24:27], s[0:3], 0x1040; GFX10-NEXT: v_fma_f32 v1, v1, v5, s2841; GFX10-NEXT: v_max_f32_e64 v6, s0, s0 clamp42; GFX10-NEXT: v_add_f32_e64 v5, s29, -1.043; GFX10-NEXT: v_sub_f32_e32 v8, s0, v144; GFX10-NEXT: v_fma_f32 v7, -s2, v6, s645; GFX10-NEXT: v_fma_f32 v5, v6, v5, 1.046; GFX10-NEXT: v_mad_f32 v10, s2, v6, v247; GFX10-NEXT: s_mov_b32 s0, 0x3c23d70a48; GFX10-NEXT: v_fmac_f32_e32 v1, v6, v849; GFX10-NEXT: v_fmac_f32_e32 v10, v7, v650; GFX10-NEXT: s_waitcnt lgkmcnt(0)51; GFX10-NEXT: v_mul_f32_e32 v9, s10, v052; GFX10-NEXT: v_fma_f32 v0, -v0, s10, s1453; GFX10-NEXT: v_mul_f32_e32 v8, s18, v254; GFX10-NEXT: v_mul_f32_e32 v3, s22, v355; GFX10-NEXT: v_fmac_f32_e32 v9, v0, v656; GFX10-NEXT: v_sub_f32_e32 v0, v1, v557; GFX10-NEXT: v_mul_f32_e32 v1, v8, v658; GFX10-NEXT: v_mul_f32_e32 v7, v6, v359; GFX10-NEXT: v_fma_f32 v3, -v6, v3, v960; GFX10-NEXT: v_fmac_f32_e32 v5, v0, v661; GFX10-NEXT: v_fma_f32 v0, v2, s26, -v162; GFX10-NEXT: v_fmac_f32_e32 v7, v3, v663; GFX10-NEXT: v_fmac_f32_e32 v1, v0, v664; GFX10-NEXT: v_mul_f32_e32 v0, v2, v665; GFX10-NEXT: s_waitcnt vmcnt(0)66; GFX10-NEXT: v_add_f32_e32 v4, v4, v1067; GFX10-NEXT: v_mul_f32_e32 v3, v4, v668; GFX10-NEXT: v_fmaak_f32 v4, s0, v5, 0x3ca3d70a69; GFX10-NEXT: v_mul_f32_e32 v1, v3, v170; GFX10-NEXT: v_mul_f32_e32 v2, v7, v471; GFX10-NEXT: v_fmac_f32_e32 v1, v2, v072; GFX10-NEXT: v_max_f32_e32 v0, 0, v173; GFX10-NEXT: ; return to shader part epilog74;75; GFX11-LABEL: _amdgpu_ps_main:76; GFX11: ; %bb.0: ; %.entry77; GFX11-NEXT: s_mov_b32 s0, exec_lo78; GFX11-NEXT: s_wqm_b32 exec_lo, exec_lo79; GFX11-NEXT: image_sample v[0:1], v[0:1], s[0:7], s[0:3] dmask:0x3 dim:SQ_RSRC_IMG_2D80; GFX11-NEXT: s_and_b32 exec_lo, exec_lo, s081; GFX11-NEXT: s_waitcnt vmcnt(0)82; GFX11-NEXT: s_clause 0x183; GFX11-NEXT: image_sample v2, v[0:1], s[0:7], s[0:3] dmask:0x4 dim:SQ_RSRC_IMG_2D84; GFX11-NEXT: image_sample v3, v[0:1], s[0:7], s[0:3] dmask:0x1 dim:SQ_RSRC_IMG_2D85; GFX11-NEXT: v_mov_b32_e32 v4, 086; GFX11-NEXT: s_waitcnt vmcnt(0)87; GFX11-NEXT: image_load_mip v4, v[2:4], s[0:7] dmask:0x4 dim:SQ_RSRC_IMG_2D unorm88; GFX11-NEXT: s_clause 0x389; GFX11-NEXT: s_buffer_load_b32 s24, s[0:3], 0x5c90; GFX11-NEXT: s_buffer_load_b32 s28, s[0:3], 0x7c91; GFX11-NEXT: s_buffer_load_b32 s29, s[0:3], 0xc092; GFX11-NEXT: s_buffer_load_b128 s[0:3], s[0:3], 0x4093; GFX11-NEXT: s_waitcnt lgkmcnt(0)94; GFX11-NEXT: s_clause 0x195; GFX11-NEXT: s_buffer_load_b128 s[4:7], s[0:3], 0x5096; GFX11-NEXT: s_buffer_load_b32 s0, s[0:3], 0x2c97; GFX11-NEXT: v_sub_f32_e64 v5, s24, s2898; GFX11-NEXT: s_waitcnt lgkmcnt(0)99; GFX11-NEXT: s_clause 0x3100; GFX11-NEXT: s_buffer_load_b128 s[8:11], s[0:3], 0x60101; GFX11-NEXT: s_buffer_load_b128 s[12:15], s[0:3], 0x20102; GFX11-NEXT: s_buffer_load_b128 s[16:19], s[0:3], 0x0103; GFX11-NEXT: s_buffer_load_b128 s[20:23], s[0:3], 0x70104; GFX11-NEXT: v_fma_f32 v1, v1, v5, s28105; GFX11-NEXT: v_max_f32_e64 v6, s0, s0 clamp106; GFX11-NEXT: s_buffer_load_b128 s[24:27], s[0:3], 0x10107; GFX11-NEXT: v_add_f32_e64 v5, s29, -1.0108; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)109; GFX11-NEXT: v_sub_f32_e32 v8, s0, v1110; GFX11-NEXT: v_fma_f32 v7, -s2, v6, s6111; GFX11-NEXT: v_fma_f32 v10, s2, v6, v2112; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)113; GFX11-NEXT: v_fma_f32 v5, v6, v5, 1.0114; GFX11-NEXT: s_mov_b32 s0, 0x3c23d70a115; GFX11-NEXT: s_waitcnt lgkmcnt(0)116; GFX11-NEXT: v_mul_f32_e32 v9, s10, v0117; GFX11-NEXT: v_fma_f32 v0, -v0, s10, s14118; GFX11-NEXT: v_mul_f32_e32 v3, s22, v3119; GFX11-NEXT: v_dual_fmac_f32 v1, v6, v8 :: v_dual_mul_f32 v8, s18, v2120; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)121; GFX11-NEXT: v_fmac_f32_e32 v9, v0, v6122; GFX11-NEXT: v_dual_fmac_f32 v10, v7, v6 :: v_dual_mul_f32 v7, v6, v3123; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)124; GFX11-NEXT: v_sub_f32_e32 v0, v1, v5125; GFX11-NEXT: v_fma_f32 v3, -v6, v3, v9126; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)127; GFX11-NEXT: v_fmac_f32_e32 v7, v3, v6128; GFX11-NEXT: v_fmac_f32_e32 v5, v0, v6129; GFX11-NEXT: v_mul_f32_e32 v1, v8, v6130; GFX11-NEXT: s_waitcnt vmcnt(0)131; GFX11-NEXT: v_add_f32_e32 v4, v4, v10132; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)133; GFX11-NEXT: v_dual_mul_f32 v3, v4, v6 :: v_dual_fmaak_f32 v4, s0, v5, 0x3ca3d70a134; GFX11-NEXT: v_fma_f32 v0, v2, s26, -v1135; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)136; GFX11-NEXT: v_fmac_f32_e32 v1, v0, v6137; GFX11-NEXT: v_mul_f32_e32 v0, v2, v6138; GFX11-NEXT: v_mul_f32_e32 v2, v7, v4139; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)140; GFX11-NEXT: v_mul_f32_e32 v1, v3, v1141; GFX11-NEXT: v_fmac_f32_e32 v1, v2, v0142; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)143; GFX11-NEXT: v_max_f32_e32 v0, 0, v1144; GFX11-NEXT: ; return to shader part epilog145.entry:146 %0 = call <3 x float> @llvm.amdgcn.image.sample.2d.v3f32.f32(i32 7, float poison, float poison, <8 x i32> poison, <4 x i32> poison, i1 false, i32 0, i32 0)147 %.i2243 = extractelement <3 x float> %0, i32 2148 %1 = call <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32> poison, i32 0, i32 0)149 %2 = shufflevector <3 x i32> %1, <3 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>150 %3 = bitcast <4 x i32> %2 to <4 x float>151 %.i2248 = extractelement <4 x float> %3, i32 2152 %.i2249 = fmul reassoc nnan nsz arcp contract afn float %.i2243, %.i2248153 %4 = call reassoc nnan nsz arcp contract afn float @llvm.amdgcn.fmed3.f32(float poison, float 0.000000e+00, float 1.000000e+00)154 %5 = call <3 x float> @llvm.amdgcn.image.sample.2d.v3f32.f32(i32 7, float poison, float poison, <8 x i32> poison, <4 x i32> poison, i1 false, i32 0, i32 0)155 %.i2333 = extractelement <3 x float> %5, i32 2156 %6 = call reassoc nnan nsz arcp contract afn float @llvm.amdgcn.fmed3.f32(float poison, float 0.000000e+00, float 1.000000e+00)157 %7 = call <2 x float> @llvm.amdgcn.image.sample.2d.v2f32.f32(i32 3, float poison, float poison, <8 x i32> poison, <4 x i32> poison, i1 false, i32 0, i32 0)158 %.i1408 = extractelement <2 x float> %7, i32 1159 %.i0364 = extractelement <2 x float> %7, i32 0160 %8 = call float @llvm.amdgcn.image.sample.2d.f32.f32(i32 1, float poison, float poison, <8 x i32> poison, <4 x i32> poison, i1 false, i32 0, i32 0)161 %9 = call <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32> poison, i32 112, i32 0)162 %10 = shufflevector <3 x i32> %9, <3 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>163 %11 = bitcast <4 x i32> %10 to <4 x float>164 %.i2360 = extractelement <4 x float> %11, i32 2165 %.i2363 = fmul reassoc nnan nsz arcp contract afn float %.i2360, %8166 %12 = call <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32> poison, i32 96, i32 0)167 %13 = shufflevector <3 x i32> %12, <3 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>168 %14 = bitcast <4 x i32> %13 to <4 x float>169 %.i2367 = extractelement <4 x float> %14, i32 2170 %.i2370 = fmul reassoc nnan nsz arcp contract afn float %.i0364, %.i2367171 %15 = call <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32> poison, i32 32, i32 0)172 %16 = shufflevector <3 x i32> %15, <3 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>173 %17 = bitcast <4 x i32> %16 to <4 x float>174 %.i2373 = extractelement <4 x float> %17, i32 2175 %.i2376 = fsub reassoc nnan nsz arcp contract afn float %.i2373, %.i2370176 %.i2383 = fmul reassoc nnan nsz arcp contract afn float %.i2376, %6177 %.i2386 = fadd reassoc nnan nsz arcp contract afn float %.i2370, %.i2383178 %18 = call reassoc nnan nsz arcp contract afn float @llvm.amdgcn.fmed3.f32(float poison, float 0.000000e+00, float 1.000000e+00)179 %19 = fmul reassoc nnan nsz arcp contract afn float %18, %.i2363180 %.i2394 = fsub reassoc nnan nsz arcp contract afn float %.i2386, %19181 %.i2397 = fmul reassoc nnan nsz arcp contract afn float %.i2363, %18182 %.i2404 = fmul reassoc nnan nsz arcp contract afn float %.i2394, %4183 %.i2407 = fadd reassoc nnan nsz arcp contract afn float %.i2397, %.i2404184 %20 = call i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32> poison, i32 92, i32 0)185 %21 = bitcast i32 %20 to float186 %22 = call i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32> poison, i32 124, i32 0)187 %23 = bitcast i32 %22 to float188 %24 = fsub reassoc nnan nsz arcp contract afn float %21, %23189 %25 = fmul reassoc nnan nsz arcp contract afn float %.i1408, %24190 %26 = fadd reassoc nnan nsz arcp contract afn float %25, %23191 %27 = call i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32> poison, i32 44, i32 0)192 %28 = bitcast i32 %27 to float193 %29 = fsub reassoc nnan nsz arcp contract afn float %28, %26194 %30 = fmul reassoc nnan nsz arcp contract afn float %6, %29195 %31 = fadd reassoc nnan nsz arcp contract afn float %26, %30196 %32 = call i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32> poison, i32 192, i32 0)197 %33 = bitcast i32 %32 to float198 %34 = fadd reassoc nnan nsz arcp contract afn float %33, -1.000000e+00199 %35 = fmul reassoc nnan nsz arcp contract afn float %18, %34200 %36 = fadd reassoc nnan nsz arcp contract afn float %35, 1.000000e+00201 %37 = fsub reassoc nnan nsz arcp contract afn float %31, %36202 %38 = fmul reassoc nnan nsz arcp contract afn float %37, %4203 %39 = fadd reassoc nnan nsz arcp contract afn float %36, %38204 %40 = fmul reassoc nnan nsz arcp contract afn float %39, 0x3F847AE140000000205 %41 = fadd reassoc nnan nsz arcp contract afn float %40, 0x3F947AE140000000206 %.i2415 = fmul reassoc nnan nsz arcp contract afn float %.i2407, %41207 %42 = call <3 x float> @llvm.amdgcn.image.load.mip.2d.v3f32.i32(i32 7, i32 poison, i32 poison, i32 0, <8 x i32> poison, i32 0, i32 0)208 %.i2521 = extractelement <3 x float> %42, i32 2209 %43 = call reassoc nnan nsz arcp contract afn float @llvm.amdgcn.fmed3.f32(float poison, float 0.000000e+00, float 1.000000e+00)210 %44 = call <3 x float> @llvm.amdgcn.image.sample.2d.v3f32.f32(i32 7, float poison, float poison, <8 x i32> poison, <4 x i32> poison, i1 false, i32 0, i32 0)211 %.i2465 = extractelement <3 x float> %44, i32 2212 %.i2466 = fmul reassoc nnan nsz arcp contract afn float %.i2465, %43213 %.i2469 = fmul reassoc nnan nsz arcp contract afn float %.i2415, %.i2466214 %45 = call <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32> poison, i32 64, i32 0)215 %46 = shufflevector <3 x i32> %45, <3 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>216 %47 = bitcast <4 x i32> %46 to <4 x float>217 %.i2476 = extractelement <4 x float> %47, i32 2218 %.i2479 = fmul reassoc nnan nsz arcp contract afn float %.i2476, %18219 %48 = call <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32> poison, i32 80, i32 0)220 %49 = shufflevector <3 x i32> %48, <3 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>221 %50 = bitcast <4 x i32> %49 to <4 x float>222 %.i2482 = extractelement <4 x float> %50, i32 2223 %.i2485 = fsub reassoc nnan nsz arcp contract afn float %.i2482, %.i2479224 %.i2488 = fmul reassoc nnan nsz arcp contract afn float %.i2249, %18225 %.i2491 = fmul reassoc nnan nsz arcp contract afn float %.i2485, %4226 %.i2494 = fadd reassoc nnan nsz arcp contract afn float %.i2479, %.i2491227 %51 = call <3 x float> @llvm.amdgcn.image.sample.2d.v3f32.f32(i32 7, float poison, float poison, <8 x i32> poison, <4 x i32> poison, i1 false, i32 0, i32 0)228 %.i2515 = extractelement <3 x float> %51, i32 2229 %.i2516 = fadd reassoc nnan nsz arcp contract afn float %.i2515, %.i2494230 %.i2522 = fadd reassoc nnan nsz arcp contract afn float %.i2521, %.i2516231 %.i2525 = fmul reassoc nnan nsz arcp contract afn float %.i2522, %43232 %52 = call <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32> poison, i32 16, i32 0)233 %53 = shufflevector <3 x i32> %52, <3 x i32> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>234 %54 = bitcast <4 x i32> %53 to <4 x float>235 %.i2530 = extractelement <4 x float> %54, i32 2236 %.i2531 = fmul reassoc nnan nsz arcp contract afn float %.i2333, %.i2530237 %.i2536 = fsub reassoc nnan nsz arcp contract afn float %.i2531, %.i2488238 %.i2539 = fmul reassoc nnan nsz arcp contract afn float %.i2536, %4239 %.i2542 = fadd reassoc nnan nsz arcp contract afn float %.i2488, %.i2539240 %.i2545 = fmul reassoc nnan nsz arcp contract afn float %.i2525, %.i2542241 %.i2548 = fadd reassoc nnan nsz arcp contract afn float %.i2469, %.i2545242 %.i2551 = call reassoc nnan nsz arcp contract afn float @llvm.maxnum.f32(float %.i2548, float 0.000000e+00)243 ret float %.i2551244}245 246define float @fmac_sequence_simple(float %a, float %b, float %c, float %d, float %e) #0 {247; GFX10-LABEL: fmac_sequence_simple:248; GFX10: ; %bb.0:249; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)250; GFX10-NEXT: v_fma_f32 v2, v2, v3, v4251; GFX10-NEXT: v_fmac_f32_e32 v2, v0, v1252; GFX10-NEXT: v_mov_b32_e32 v0, v2253; GFX10-NEXT: s_setpc_b64 s[30:31]254;255; GFX11-LABEL: fmac_sequence_simple:256; GFX11: ; %bb.0:257; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)258; GFX11-NEXT: v_fma_f32 v2, v2, v3, v4259; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)260; GFX11-NEXT: v_fmac_f32_e32 v2, v0, v1261; GFX11-NEXT: v_mov_b32_e32 v0, v2262; GFX11-NEXT: s_setpc_b64 s[30:31]263 %t0 = fmul fast float %a, %b264 %t1 = fmul fast float %c, %d265 %t2 = fadd fast float %t0, %t1266 %t5 = fadd fast float %t2, %e267 ret float %t5268}269 270define float @fmac_sequence_innermost_fmul(float %a, float %b, float %c, float %d, float %e, float %f, float %g) #0 {271; GFX10-LABEL: fmac_sequence_innermost_fmul:272; GFX10: ; %bb.0:273; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)274; GFX10-NEXT: v_mad_f32 v2, v2, v3, v6275; GFX10-NEXT: v_fmac_f32_e32 v2, v0, v1276; GFX10-NEXT: v_fmac_f32_e32 v2, v4, v5277; GFX10-NEXT: v_mov_b32_e32 v0, v2278; GFX10-NEXT: s_setpc_b64 s[30:31]279;280; GFX11-LABEL: fmac_sequence_innermost_fmul:281; GFX11: ; %bb.0:282; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)283; GFX11-NEXT: v_fma_f32 v2, v2, v3, v6284; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)285; GFX11-NEXT: v_fmac_f32_e32 v2, v0, v1286; GFX11-NEXT: v_fmac_f32_e32 v2, v4, v5287; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)288; GFX11-NEXT: v_mov_b32_e32 v0, v2289; GFX11-NEXT: s_setpc_b64 s[30:31]290 %t0 = fmul fast float %a, %b291 %t1 = fmul fast float %c, %d292 %t2 = fadd fast float %t0, %t1293 %t3 = fmul fast float %e, %f294 %t4 = fadd fast float %t2, %t3295 %t5 = fadd fast float %t4, %g296 ret float %t5297}298 299define float @fmac_sequence_innermost_fmul_swapped_operands(float %a, float %b, float %c, float %d, float %e, float %f, float %g) #0 {300; GFX10-LABEL: fmac_sequence_innermost_fmul_swapped_operands:301; GFX10: ; %bb.0:302; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)303; GFX10-NEXT: v_mad_f32 v2, v2, v3, v6304; GFX10-NEXT: v_fmac_f32_e32 v2, v0, v1305; GFX10-NEXT: v_fmac_f32_e32 v2, v4, v5306; GFX10-NEXT: v_mov_b32_e32 v0, v2307; GFX10-NEXT: s_setpc_b64 s[30:31]308;309; GFX11-LABEL: fmac_sequence_innermost_fmul_swapped_operands:310; GFX11: ; %bb.0:311; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)312; GFX11-NEXT: v_fma_f32 v2, v2, v3, v6313; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)314; GFX11-NEXT: v_fmac_f32_e32 v2, v0, v1315; GFX11-NEXT: v_fmac_f32_e32 v2, v4, v5316; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)317; GFX11-NEXT: v_mov_b32_e32 v0, v2318; GFX11-NEXT: s_setpc_b64 s[30:31]319 %t0 = fmul fast float %a, %b320 %t1 = fmul fast float %c, %d321 %t2 = fadd fast float %t0, %t1322 %t3 = fmul fast float %e, %f323 %t4 = fadd fast float %t2, %t3324 %t5 = fadd fast float %g, %t4325 ret float %t5326}327 328define amdgpu_ps float @fmac_sequence_innermost_fmul_sgpr(float inreg %a, float inreg %b, float inreg %c, float inreg %d, float inreg %e, float inreg %f, float %g) #0 {329; GFX10-LABEL: fmac_sequence_innermost_fmul_sgpr:330; GFX10: ; %bb.0:331; GFX10-NEXT: v_mac_f32_e64 v0, s2, s3332; GFX10-NEXT: v_fmac_f32_e64 v0, s0, s1333; GFX10-NEXT: v_fmac_f32_e64 v0, s4, s5334; GFX10-NEXT: ; return to shader part epilog335;336; GFX11-LABEL: fmac_sequence_innermost_fmul_sgpr:337; GFX11: ; %bb.0:338; GFX11-NEXT: v_fmac_f32_e64 v0, s2, s3339; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)340; GFX11-NEXT: v_fmac_f32_e64 v0, s0, s1341; GFX11-NEXT: v_fmac_f32_e64 v0, s4, s5342; GFX11-NEXT: ; return to shader part epilog343 %t0 = fmul fast float %a, %b344 %t1 = fmul fast float %c, %d345 %t2 = fadd fast float %t0, %t1346 %t3 = fmul fast float %e, %f347 %t4 = fadd fast float %t2, %t3348 %t5 = fadd fast float %t4, %g349 ret float %t5350}351 352define amdgpu_ps float @fmac_sequence_innermost_fmul_multiple_use(float inreg %a, float inreg %b, float inreg %c, float inreg %d, float inreg %e, float inreg %f, float %g) #0 {353; GFX10-LABEL: fmac_sequence_innermost_fmul_multiple_use:354; GFX10: ; %bb.0:355; GFX10-NEXT: v_mul_f32_e64 v1, s2, s3356; GFX10-NEXT: v_fmac_f32_e64 v1, s0, s1357; GFX10-NEXT: v_fma_f32 v2, s5, s4, v1358; GFX10-NEXT: v_fmac_f32_e32 v1, s5, v2359; GFX10-NEXT: v_add_f32_e32 v0, v1, v0360; GFX10-NEXT: ; return to shader part epilog361;362; GFX11-LABEL: fmac_sequence_innermost_fmul_multiple_use:363; GFX11: ; %bb.0:364; GFX11-NEXT: v_mul_f32_e64 v1, s2, s3365; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)366; GFX11-NEXT: v_fmac_f32_e64 v1, s0, s1367; GFX11-NEXT: v_fma_f32 v2, s5, s4, v1368; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)369; GFX11-NEXT: v_fmac_f32_e32 v1, s5, v2370; GFX11-NEXT: v_add_f32_e32 v0, v1, v0371; GFX11-NEXT: ; return to shader part epilog372 %t0 = fmul fast float %a, %b373 %t1 = fmul fast float %c, %d374 %t2 = fadd fast float %t0, %t1375 %t3 = fmul fast float %e, %f376 %t4 = fadd fast float %t2, %t3377 %t5 = fmul fast float %f, %t4378 %t6 = fadd fast float %t5, %t2379 %t7 = fadd fast float %t6, %g380 ret float %t7381}382 383; "fmul %m, 2.0" could select to an FMA instruction, but it is no better than384; selecting it as a multiply. In some cases the multiply is better because385; SIFoldOperands can fold it into a previous instruction as an output modifier.386define amdgpu_ps float @fma_vs_output_modifier(float %x, i32 %n) #0 {387; GFX10-LABEL: fma_vs_output_modifier:388; GFX10: ; %bb.0:389; GFX10-NEXT: v_cvt_f32_i32_e64 v1, v1 mul:2390; GFX10-NEXT: v_mul_f32_e32 v0, v0, v0391; GFX10-NEXT: v_mul_f32_e32 v0, v0, v1392; GFX10-NEXT: ; return to shader part epilog393;394; GFX11-LABEL: fma_vs_output_modifier:395; GFX11: ; %bb.0:396; GFX11-NEXT: v_cvt_f32_i32_e64 v1, v1 mul:2397; GFX11-NEXT: v_mul_f32_e32 v0, v0, v0398; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)399; GFX11-NEXT: v_mul_f32_e32 v0, v0, v1400; GFX11-NEXT: ; return to shader part epilog401 %s = sitofp i32 %n to float402 %m = fmul contract float %x, %x403 %a = fmul contract float %m, 2.0404 %r = fmul reassoc nsz float %a, %s405 ret float %r406}407 408define amdgpu_ps float @fma_vs_output_modifier_2(float %x) #0 {409; GCN-LABEL: fma_vs_output_modifier_2:410; GCN: ; %bb.0:411; GCN-NEXT: v_mul_f32_e64 v0, v0, v0 mul:2412; GCN-NEXT: ; return to shader part epilog413 %m = fmul contract float %x, %x414 %a = fadd nsz contract float %m, %m415 ret float %a416}417 418; Function Attrs: nofree nosync nounwind readnone speculatable willreturn419declare float @llvm.maxnum.f32(float, float) #1420 421; Function Attrs: nounwind readnone speculatable willreturn422declare float @llvm.amdgcn.fmed3.f32(float, float, float) #2423 424; Function Attrs: nounwind readonly willreturn425declare <2 x float> @llvm.amdgcn.image.sample.2d.v2f32.f32(i32 immarg, float, float, <8 x i32>, <4 x i32>, i1 immarg, i32 immarg, i32 immarg) #3426 427; Function Attrs: nounwind readonly willreturn428declare float @llvm.amdgcn.image.sample.2d.f32.f32(i32 immarg, float, float, <8 x i32>, <4 x i32>, i1 immarg, i32 immarg, i32 immarg) #3429 430; Function Attrs: nounwind readonly willreturn431declare <3 x float> @llvm.amdgcn.image.sample.2d.v3f32.f32(i32 immarg, float, float, <8 x i32>, <4 x i32>, i1 immarg, i32 immarg, i32 immarg) #3432 433; Function Attrs: nounwind readonly willreturn434declare <3 x float> @llvm.amdgcn.image.load.mip.2d.v3f32.i32(i32 immarg, i32, i32, i32, <8 x i32>, i32 immarg, i32 immarg) #3435 436; Function Attrs: nounwind readnone willreturn437declare i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32>, i32, i32 immarg) #3438 439; Function Attrs: nounwind readnone willreturn440declare <3 x i32> @llvm.amdgcn.s.buffer.load.v3i32(<4 x i32>, i32, i32 immarg) #3441 442attributes #0 = { "denormal-fp-math-f32"="preserve-sign" }443attributes #1 = { nofree nosync nounwind readnone speculatable willreturn }444attributes #2 = { nounwind readnone speculatable willreturn }445attributes #3 = { nounwind readonly willreturn }446