brintos

brintos / llvm-project-archived public Read only

0
0
Text · 50.3 KiB · c98feeb Raw
1236 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py2; RUN: llc -mtriple=amdgcn -mcpu=tonga < %s | FileCheck -check-prefix=SI %s3; RUN: llc -mtriple=amdgcn -mcpu=gfx900 < %s | FileCheck -check-prefix=GFX9 %s4; RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -mattr=+wavefrontsize32 < %s | FileCheck -check-prefixes=GFX10-32 %s5; RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -mattr=+wavefrontsize64 < %s | FileCheck -check-prefixes=GFX10-64 %s6 7define amdgpu_ps void @static_exact(float %arg0, float %arg1) {8; SI-LABEL: static_exact:9; SI:       ; %bb.0: ; %.entry10; SI-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v011; SI-NEXT:    s_andn2_b64 exec, exec, exec12; SI-NEXT:    s_cbranch_scc0 .LBB0_213; SI-NEXT:  ; %bb.1: ; %.entry14; SI-NEXT:    s_mov_b64 exec, 015; SI-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc16; SI-NEXT:    exp mrt1 v0, v0, v0, v0 done vm17; SI-NEXT:    s_endpgm18; SI-NEXT:  .LBB0_2:19; SI-NEXT:    s_mov_b64 exec, 020; SI-NEXT:    exp null off, off, off, off done vm21; SI-NEXT:    s_endpgm22;23; GFX9-LABEL: static_exact:24; GFX9:       ; %bb.0: ; %.entry25; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v026; GFX9-NEXT:    s_andn2_b64 exec, exec, exec27; GFX9-NEXT:    s_cbranch_scc0 .LBB0_228; GFX9-NEXT:  ; %bb.1: ; %.entry29; GFX9-NEXT:    s_mov_b64 exec, 030; GFX9-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc31; GFX9-NEXT:    exp mrt1 v0, v0, v0, v0 done vm32; GFX9-NEXT:    s_endpgm33; GFX9-NEXT:  .LBB0_2:34; GFX9-NEXT:    s_mov_b64 exec, 035; GFX9-NEXT:    exp null off, off, off, off done vm36; GFX9-NEXT:    s_endpgm37;38; GFX10-32-LABEL: static_exact:39; GFX10-32:       ; %bb.0: ; %.entry40; GFX10-32-NEXT:    v_cmp_gt_f32_e32 vcc_lo, 0, v041; GFX10-32-NEXT:    s_andn2_b32 exec_lo, exec_lo, exec_lo42; GFX10-32-NEXT:    s_cbranch_scc0 .LBB0_243; GFX10-32-NEXT:  ; %bb.1: ; %.entry44; GFX10-32-NEXT:    s_mov_b32 exec_lo, 045; GFX10-32-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc_lo46; GFX10-32-NEXT:    exp mrt1 v0, v0, v0, v0 done vm47; GFX10-32-NEXT:    s_endpgm48; GFX10-32-NEXT:  .LBB0_2:49; GFX10-32-NEXT:    s_mov_b32 exec_lo, 050; GFX10-32-NEXT:    exp null off, off, off, off done vm51; GFX10-32-NEXT:    s_endpgm52;53; GFX10-64-LABEL: static_exact:54; GFX10-64:       ; %bb.0: ; %.entry55; GFX10-64-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v056; GFX10-64-NEXT:    s_andn2_b64 exec, exec, exec57; GFX10-64-NEXT:    s_cbranch_scc0 .LBB0_258; GFX10-64-NEXT:  ; %bb.1: ; %.entry59; GFX10-64-NEXT:    s_mov_b64 exec, 060; GFX10-64-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc61; GFX10-64-NEXT:    exp mrt1 v0, v0, v0, v0 done vm62; GFX10-64-NEXT:    s_endpgm63; GFX10-64-NEXT:  .LBB0_2:64; GFX10-64-NEXT:    s_mov_b64 exec, 065; GFX10-64-NEXT:    exp null off, off, off, off done vm66; GFX10-64-NEXT:    s_endpgm67.entry:68  %c0 = fcmp olt float %arg0, 0.000000e+0069  %c1 = fcmp oge float %arg1, 0.070  call void @llvm.amdgcn.wqm.demote(i1 false)71  %tmp1 = select i1 %c0, float 1.000000e+00, float 0.000000e+0072  call void @llvm.amdgcn.exp.f32(i32 1, i32 15, float %tmp1, float %tmp1, float %tmp1, float %tmp1, i1 true, i1 true) #073  ret void74}75 76define amdgpu_ps void @dynamic_exact(float %arg0, float %arg1) {77; SI-LABEL: dynamic_exact:78; SI:       ; %bb.0: ; %.entry79; SI-NEXT:    v_cmp_le_f32_e64 s[0:1], 0, v180; SI-NEXT:    s_mov_b64 s[2:3], exec81; SI-NEXT:    s_andn2_b64 s[0:1], exec, s[0:1]82; SI-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[0:1]83; SI-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v084; SI-NEXT:    s_cbranch_scc0 .LBB1_285; SI-NEXT:  ; %bb.1: ; %.entry86; SI-NEXT:    s_and_b64 exec, exec, s[2:3]87; SI-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc88; SI-NEXT:    exp mrt1 v0, v0, v0, v0 done vm89; SI-NEXT:    s_endpgm90; SI-NEXT:  .LBB1_2:91; SI-NEXT:    s_mov_b64 exec, 092; SI-NEXT:    exp null off, off, off, off done vm93; SI-NEXT:    s_endpgm94;95; GFX9-LABEL: dynamic_exact:96; GFX9:       ; %bb.0: ; %.entry97; GFX9-NEXT:    v_cmp_le_f32_e64 s[0:1], 0, v198; GFX9-NEXT:    s_mov_b64 s[2:3], exec99; GFX9-NEXT:    s_andn2_b64 s[0:1], exec, s[0:1]100; GFX9-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[0:1]101; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v0102; GFX9-NEXT:    s_cbranch_scc0 .LBB1_2103; GFX9-NEXT:  ; %bb.1: ; %.entry104; GFX9-NEXT:    s_and_b64 exec, exec, s[2:3]105; GFX9-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc106; GFX9-NEXT:    exp mrt1 v0, v0, v0, v0 done vm107; GFX9-NEXT:    s_endpgm108; GFX9-NEXT:  .LBB1_2:109; GFX9-NEXT:    s_mov_b64 exec, 0110; GFX9-NEXT:    exp null off, off, off, off done vm111; GFX9-NEXT:    s_endpgm112;113; GFX10-32-LABEL: dynamic_exact:114; GFX10-32:       ; %bb.0: ; %.entry115; GFX10-32-NEXT:    v_cmp_le_f32_e64 s0, 0, v1116; GFX10-32-NEXT:    s_mov_b32 s1, exec_lo117; GFX10-32-NEXT:    v_cmp_gt_f32_e32 vcc_lo, 0, v0118; GFX10-32-NEXT:    s_andn2_b32 s0, exec_lo, s0119; GFX10-32-NEXT:    s_andn2_b32 s1, s1, s0120; GFX10-32-NEXT:    s_cbranch_scc0 .LBB1_2121; GFX10-32-NEXT:  ; %bb.1: ; %.entry122; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s1123; GFX10-32-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc_lo124; GFX10-32-NEXT:    exp mrt1 v0, v0, v0, v0 done vm125; GFX10-32-NEXT:    s_endpgm126; GFX10-32-NEXT:  .LBB1_2:127; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0128; GFX10-32-NEXT:    exp null off, off, off, off done vm129; GFX10-32-NEXT:    s_endpgm130;131; GFX10-64-LABEL: dynamic_exact:132; GFX10-64:       ; %bb.0: ; %.entry133; GFX10-64-NEXT:    v_cmp_le_f32_e64 s[0:1], 0, v1134; GFX10-64-NEXT:    s_mov_b64 s[2:3], exec135; GFX10-64-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v0136; GFX10-64-NEXT:    s_andn2_b64 s[0:1], exec, s[0:1]137; GFX10-64-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[0:1]138; GFX10-64-NEXT:    s_cbranch_scc0 .LBB1_2139; GFX10-64-NEXT:  ; %bb.1: ; %.entry140; GFX10-64-NEXT:    s_and_b64 exec, exec, s[2:3]141; GFX10-64-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc142; GFX10-64-NEXT:    exp mrt1 v0, v0, v0, v0 done vm143; GFX10-64-NEXT:    s_endpgm144; GFX10-64-NEXT:  .LBB1_2:145; GFX10-64-NEXT:    s_mov_b64 exec, 0146; GFX10-64-NEXT:    exp null off, off, off, off done vm147; GFX10-64-NEXT:    s_endpgm148.entry:149  %c0 = fcmp olt float %arg0, 0.000000e+00150  %c1 = fcmp oge float %arg1, 0.0151  call void @llvm.amdgcn.wqm.demote(i1 %c1)152  %tmp1 = select i1 %c0, float 1.000000e+00, float 0.000000e+00153  call void @llvm.amdgcn.exp.f32(i32 1, i32 15, float %tmp1, float %tmp1, float %tmp1, float %tmp1, i1 true, i1 true) #0154  ret void155}156 157define amdgpu_ps void @branch(float %arg0, float %arg1) {158; SI-LABEL: branch:159; SI:       ; %bb.0: ; %.entry160; SI-NEXT:    v_cvt_i32_f32_e32 v0, v0161; SI-NEXT:    v_cvt_i32_f32_e32 v1, v1162; SI-NEXT:    s_mov_b64 s[2:3], exec163; SI-NEXT:    v_or_b32_e32 v0, v0, v1164; SI-NEXT:    v_and_b32_e32 v0, 1, v0165; SI-NEXT:    v_cmp_eq_u32_e32 vcc, 0, v0166; SI-NEXT:    v_cmp_eq_u32_e64 s[0:1], 1, v0167; SI-NEXT:    s_and_saveexec_b64 s[4:5], s[0:1]168; SI-NEXT:    s_xor_b64 s[0:1], exec, s[4:5]169; SI-NEXT:    s_cbranch_execz .LBB2_3170; SI-NEXT:  ; %bb.1: ; %.demote171; SI-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec172; SI-NEXT:    s_cbranch_scc0 .LBB2_4173; SI-NEXT:  ; %bb.2: ; %.demote174; SI-NEXT:    s_mov_b64 exec, 0175; SI-NEXT:  .LBB2_3: ; %.continue176; SI-NEXT:    s_or_b64 exec, exec, s[0:1]177; SI-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc178; SI-NEXT:    exp mrt1 v0, v0, v0, v0 done vm179; SI-NEXT:    s_endpgm180; SI-NEXT:  .LBB2_4:181; SI-NEXT:    s_mov_b64 exec, 0182; SI-NEXT:    exp null off, off, off, off done vm183; SI-NEXT:    s_endpgm184;185; GFX9-LABEL: branch:186; GFX9:       ; %bb.0: ; %.entry187; GFX9-NEXT:    v_cvt_i32_f32_e32 v0, v0188; GFX9-NEXT:    v_cvt_i32_f32_e32 v1, v1189; GFX9-NEXT:    s_mov_b64 s[2:3], exec190; GFX9-NEXT:    v_or_b32_e32 v0, v0, v1191; GFX9-NEXT:    v_and_b32_e32 v0, 1, v0192; GFX9-NEXT:    v_cmp_eq_u32_e32 vcc, 0, v0193; GFX9-NEXT:    v_cmp_eq_u32_e64 s[0:1], 1, v0194; GFX9-NEXT:    s_and_saveexec_b64 s[4:5], s[0:1]195; GFX9-NEXT:    s_xor_b64 s[0:1], exec, s[4:5]196; GFX9-NEXT:    s_cbranch_execz .LBB2_3197; GFX9-NEXT:  ; %bb.1: ; %.demote198; GFX9-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec199; GFX9-NEXT:    s_cbranch_scc0 .LBB2_4200; GFX9-NEXT:  ; %bb.2: ; %.demote201; GFX9-NEXT:    s_mov_b64 exec, 0202; GFX9-NEXT:  .LBB2_3: ; %.continue203; GFX9-NEXT:    s_or_b64 exec, exec, s[0:1]204; GFX9-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc205; GFX9-NEXT:    exp mrt1 v0, v0, v0, v0 done vm206; GFX9-NEXT:    s_endpgm207; GFX9-NEXT:  .LBB2_4:208; GFX9-NEXT:    s_mov_b64 exec, 0209; GFX9-NEXT:    exp null off, off, off, off done vm210; GFX9-NEXT:    s_endpgm211;212; GFX10-32-LABEL: branch:213; GFX10-32:       ; %bb.0: ; %.entry214; GFX10-32-NEXT:    v_cvt_i32_f32_e32 v0, v0215; GFX10-32-NEXT:    v_cvt_i32_f32_e32 v1, v1216; GFX10-32-NEXT:    s_mov_b32 s1, exec_lo217; GFX10-32-NEXT:    v_or_b32_e32 v0, v0, v1218; GFX10-32-NEXT:    v_and_b32_e32 v0, 1, v0219; GFX10-32-NEXT:    v_cmp_eq_u32_e64 s0, 1, v0220; GFX10-32-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v0221; GFX10-32-NEXT:    s_and_saveexec_b32 s2, s0222; GFX10-32-NEXT:    s_xor_b32 s0, exec_lo, s2223; GFX10-32-NEXT:    s_cbranch_execz .LBB2_3224; GFX10-32-NEXT:  ; %bb.1: ; %.demote225; GFX10-32-NEXT:    s_andn2_b32 s1, s1, exec_lo226; GFX10-32-NEXT:    s_cbranch_scc0 .LBB2_4227; GFX10-32-NEXT:  ; %bb.2: ; %.demote228; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0229; GFX10-32-NEXT:  .LBB2_3: ; %.continue230; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s0231; GFX10-32-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc_lo232; GFX10-32-NEXT:    exp mrt1 v0, v0, v0, v0 done vm233; GFX10-32-NEXT:    s_endpgm234; GFX10-32-NEXT:  .LBB2_4:235; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0236; GFX10-32-NEXT:    exp null off, off, off, off done vm237; GFX10-32-NEXT:    s_endpgm238;239; GFX10-64-LABEL: branch:240; GFX10-64:       ; %bb.0: ; %.entry241; GFX10-64-NEXT:    v_cvt_i32_f32_e32 v0, v0242; GFX10-64-NEXT:    v_cvt_i32_f32_e32 v1, v1243; GFX10-64-NEXT:    s_mov_b64 s[2:3], exec244; GFX10-64-NEXT:    v_or_b32_e32 v0, v0, v1245; GFX10-64-NEXT:    v_and_b32_e32 v0, 1, v0246; GFX10-64-NEXT:    v_cmp_eq_u32_e32 vcc, 0, v0247; GFX10-64-NEXT:    v_cmp_eq_u32_e64 s[0:1], 1, v0248; GFX10-64-NEXT:    s_and_saveexec_b64 s[4:5], s[0:1]249; GFX10-64-NEXT:    s_xor_b64 s[0:1], exec, s[4:5]250; GFX10-64-NEXT:    s_cbranch_execz .LBB2_3251; GFX10-64-NEXT:  ; %bb.1: ; %.demote252; GFX10-64-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec253; GFX10-64-NEXT:    s_cbranch_scc0 .LBB2_4254; GFX10-64-NEXT:  ; %bb.2: ; %.demote255; GFX10-64-NEXT:    s_mov_b64 exec, 0256; GFX10-64-NEXT:  .LBB2_3: ; %.continue257; GFX10-64-NEXT:    s_or_b64 exec, exec, s[0:1]258; GFX10-64-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc259; GFX10-64-NEXT:    exp mrt1 v0, v0, v0, v0 done vm260; GFX10-64-NEXT:    s_endpgm261; GFX10-64-NEXT:  .LBB2_4:262; GFX10-64-NEXT:    s_mov_b64 exec, 0263; GFX10-64-NEXT:    exp null off, off, off, off done vm264; GFX10-64-NEXT:    s_endpgm265.entry:266  %i0 = fptosi float %arg0 to i32267  %i1 = fptosi float %arg1 to i32268  %c0 = or i32 %i0, %i1269  %c1 = and i32 %c0, 1270  %c2 = icmp eq i32 %c1, 0271  br i1 %c2, label %.continue, label %.demote272 273.demote:274  call void @llvm.amdgcn.wqm.demote(i1 false)275  br label %.continue276 277.continue:278  %tmp1 = select i1 %c2, float 1.000000e+00, float 0.000000e+00279  call void @llvm.amdgcn.exp.f32(i32 1, i32 15, float %tmp1, float %tmp1, float %tmp1, float %tmp1, i1 true, i1 true) #0280  ret void281}282 283 284define amdgpu_ps <4 x float> @wqm_demote_1(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %idx, float %data, float %coord, float %coord2, float %z) {285; SI-LABEL: wqm_demote_1:286; SI:       ; %bb.0: ; %.entry287; SI-NEXT:    s_mov_b64 s[12:13], exec288; SI-NEXT:    s_wqm_b64 exec, exec289; SI-NEXT:    v_cmp_ngt_f32_e32 vcc, 0, v1290; SI-NEXT:    s_and_saveexec_b64 s[14:15], vcc291; SI-NEXT:    s_xor_b64 s[14:15], exec, s[14:15]292; SI-NEXT:    s_cbranch_execz .LBB3_3293; SI-NEXT:  ; %bb.1: ; %.demote294; SI-NEXT:    s_andn2_b64 s[12:13], s[12:13], exec295; SI-NEXT:    s_cbranch_scc0 .LBB3_4296; SI-NEXT:  ; %bb.2: ; %.demote297; SI-NEXT:    s_wqm_b64 s[16:17], s[12:13]298; SI-NEXT:    s_and_b64 exec, exec, s[16:17]299; SI-NEXT:  .LBB3_3: ; %.continue300; SI-NEXT:    s_or_b64 exec, exec, s[14:15]301; SI-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1302; SI-NEXT:    s_waitcnt vmcnt(0)303; SI-NEXT:    v_add_f32_e32 v0, v0, v0304; SI-NEXT:    s_and_b64 exec, exec, s[12:13]305; SI-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf306; SI-NEXT:    s_waitcnt vmcnt(0)307; SI-NEXT:    s_branch .LBB3_5308; SI-NEXT:  .LBB3_4:309; SI-NEXT:    s_mov_b64 exec, 0310; SI-NEXT:    exp null off, off, off, off done vm311; SI-NEXT:    s_endpgm312; SI-NEXT:  .LBB3_5:313;314; GFX9-LABEL: wqm_demote_1:315; GFX9:       ; %bb.0: ; %.entry316; GFX9-NEXT:    s_mov_b64 s[12:13], exec317; GFX9-NEXT:    s_wqm_b64 exec, exec318; GFX9-NEXT:    v_cmp_ngt_f32_e32 vcc, 0, v1319; GFX9-NEXT:    s_and_saveexec_b64 s[14:15], vcc320; GFX9-NEXT:    s_xor_b64 s[14:15], exec, s[14:15]321; GFX9-NEXT:    s_cbranch_execz .LBB3_3322; GFX9-NEXT:  ; %bb.1: ; %.demote323; GFX9-NEXT:    s_andn2_b64 s[12:13], s[12:13], exec324; GFX9-NEXT:    s_cbranch_scc0 .LBB3_4325; GFX9-NEXT:  ; %bb.2: ; %.demote326; GFX9-NEXT:    s_wqm_b64 s[16:17], s[12:13]327; GFX9-NEXT:    s_and_b64 exec, exec, s[16:17]328; GFX9-NEXT:  .LBB3_3: ; %.continue329; GFX9-NEXT:    s_or_b64 exec, exec, s[14:15]330; GFX9-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1331; GFX9-NEXT:    s_waitcnt vmcnt(0)332; GFX9-NEXT:    v_add_f32_e32 v0, v0, v0333; GFX9-NEXT:    s_and_b64 exec, exec, s[12:13]334; GFX9-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf335; GFX9-NEXT:    s_waitcnt vmcnt(0)336; GFX9-NEXT:    s_branch .LBB3_5337; GFX9-NEXT:  .LBB3_4:338; GFX9-NEXT:    s_mov_b64 exec, 0339; GFX9-NEXT:    exp null off, off, off, off done vm340; GFX9-NEXT:    s_endpgm341; GFX9-NEXT:  .LBB3_5:342;343; GFX10-32-LABEL: wqm_demote_1:344; GFX10-32:       ; %bb.0: ; %.entry345; GFX10-32-NEXT:    s_mov_b32 s12, exec_lo346; GFX10-32-NEXT:    s_wqm_b32 exec_lo, exec_lo347; GFX10-32-NEXT:    v_cmp_ngt_f32_e32 vcc_lo, 0, v1348; GFX10-32-NEXT:    s_and_saveexec_b32 s13, vcc_lo349; GFX10-32-NEXT:    s_xor_b32 s13, exec_lo, s13350; GFX10-32-NEXT:    s_cbranch_execz .LBB3_3351; GFX10-32-NEXT:  ; %bb.1: ; %.demote352; GFX10-32-NEXT:    s_andn2_b32 s12, s12, exec_lo353; GFX10-32-NEXT:    s_cbranch_scc0 .LBB3_4354; GFX10-32-NEXT:  ; %bb.2: ; %.demote355; GFX10-32-NEXT:    s_wqm_b32 s14, s12356; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s14357; GFX10-32-NEXT:  .LBB3_3: ; %.continue358; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s13359; GFX10-32-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1 dim:SQ_RSRC_IMG_1D360; GFX10-32-NEXT:    s_waitcnt vmcnt(0)361; GFX10-32-NEXT:    v_add_f32_e32 v0, v0, v0362; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s12363; GFX10-32-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D364; GFX10-32-NEXT:    s_waitcnt vmcnt(0)365; GFX10-32-NEXT:    s_branch .LBB3_5366; GFX10-32-NEXT:  .LBB3_4:367; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0368; GFX10-32-NEXT:    exp null off, off, off, off done vm369; GFX10-32-NEXT:    s_endpgm370; GFX10-32-NEXT:  .LBB3_5:371;372; GFX10-64-LABEL: wqm_demote_1:373; GFX10-64:       ; %bb.0: ; %.entry374; GFX10-64-NEXT:    s_mov_b64 s[12:13], exec375; GFX10-64-NEXT:    s_wqm_b64 exec, exec376; GFX10-64-NEXT:    v_cmp_ngt_f32_e32 vcc, 0, v1377; GFX10-64-NEXT:    s_and_saveexec_b64 s[14:15], vcc378; GFX10-64-NEXT:    s_xor_b64 s[14:15], exec, s[14:15]379; GFX10-64-NEXT:    s_cbranch_execz .LBB3_3380; GFX10-64-NEXT:  ; %bb.1: ; %.demote381; GFX10-64-NEXT:    s_andn2_b64 s[12:13], s[12:13], exec382; GFX10-64-NEXT:    s_cbranch_scc0 .LBB3_4383; GFX10-64-NEXT:  ; %bb.2: ; %.demote384; GFX10-64-NEXT:    s_wqm_b64 s[16:17], s[12:13]385; GFX10-64-NEXT:    s_and_b64 exec, exec, s[16:17]386; GFX10-64-NEXT:  .LBB3_3: ; %.continue387; GFX10-64-NEXT:    s_or_b64 exec, exec, s[14:15]388; GFX10-64-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1 dim:SQ_RSRC_IMG_1D389; GFX10-64-NEXT:    s_waitcnt vmcnt(0)390; GFX10-64-NEXT:    v_add_f32_e32 v0, v0, v0391; GFX10-64-NEXT:    s_and_b64 exec, exec, s[12:13]392; GFX10-64-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D393; GFX10-64-NEXT:    s_waitcnt vmcnt(0)394; GFX10-64-NEXT:    s_branch .LBB3_5395; GFX10-64-NEXT:  .LBB3_4:396; GFX10-64-NEXT:    s_mov_b64 exec, 0397; GFX10-64-NEXT:    exp null off, off, off, off done vm398; GFX10-64-NEXT:    s_endpgm399; GFX10-64-NEXT:  .LBB3_5:400.entry:401  %z.cmp = fcmp olt float %z, 0.0402  br i1 %z.cmp, label %.continue, label %.demote403 404.demote:405  call void @llvm.amdgcn.wqm.demote(i1 false)406  br label %.continue407 408.continue:409  %tex = call <4 x float> @llvm.amdgcn.image.sample.1d.v4f32.f32(i32 15, float %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i1 0, i32 0, i32 0) #0410  %tex0 = extractelement <4 x float> %tex, i32 0411  %tex1 = extractelement <4 x float> %tex, i32 0412  %coord1 = fadd float %tex0, %tex1413  %rtex = call <4 x float> @llvm.amdgcn.image.sample.1d.v4f32.f32(i32 15, float %coord1, <8 x i32> %rsrc, <4 x i32> %sampler, i1 0, i32 0, i32 0) #0414 415  ret <4 x float> %rtex416}417 418define amdgpu_ps <4 x float> @wqm_demote_2(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %idx, float %data, float %coord, float %coord2, float %z) {419; SI-LABEL: wqm_demote_2:420; SI:       ; %bb.0: ; %.entry421; SI-NEXT:    s_mov_b64 s[12:13], exec422; SI-NEXT:    s_wqm_b64 exec, exec423; SI-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1424; SI-NEXT:    s_waitcnt vmcnt(0)425; SI-NEXT:    v_cmp_ngt_f32_e32 vcc, 0, v0426; SI-NEXT:    s_and_saveexec_b64 s[14:15], vcc427; SI-NEXT:    s_xor_b64 s[14:15], exec, s[14:15]428; SI-NEXT:    s_cbranch_execz .LBB4_3429; SI-NEXT:  ; %bb.1: ; %.demote430; SI-NEXT:    s_andn2_b64 s[12:13], s[12:13], exec431; SI-NEXT:    s_cbranch_scc0 .LBB4_4432; SI-NEXT:  ; %bb.2: ; %.demote433; SI-NEXT:    s_wqm_b64 s[16:17], s[12:13]434; SI-NEXT:    s_and_b64 exec, exec, s[16:17]435; SI-NEXT:  .LBB4_3: ; %.continue436; SI-NEXT:    s_or_b64 exec, exec, s[14:15]437; SI-NEXT:    v_add_f32_e32 v0, v0, v0438; SI-NEXT:    s_and_b64 exec, exec, s[12:13]439; SI-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf440; SI-NEXT:    s_waitcnt vmcnt(0)441; SI-NEXT:    s_branch .LBB4_5442; SI-NEXT:  .LBB4_4:443; SI-NEXT:    s_mov_b64 exec, 0444; SI-NEXT:    exp null off, off, off, off done vm445; SI-NEXT:    s_endpgm446; SI-NEXT:  .LBB4_5:447;448; GFX9-LABEL: wqm_demote_2:449; GFX9:       ; %bb.0: ; %.entry450; GFX9-NEXT:    s_mov_b64 s[12:13], exec451; GFX9-NEXT:    s_wqm_b64 exec, exec452; GFX9-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1453; GFX9-NEXT:    s_waitcnt vmcnt(0)454; GFX9-NEXT:    v_cmp_ngt_f32_e32 vcc, 0, v0455; GFX9-NEXT:    s_and_saveexec_b64 s[14:15], vcc456; GFX9-NEXT:    s_xor_b64 s[14:15], exec, s[14:15]457; GFX9-NEXT:    s_cbranch_execz .LBB4_3458; GFX9-NEXT:  ; %bb.1: ; %.demote459; GFX9-NEXT:    s_andn2_b64 s[12:13], s[12:13], exec460; GFX9-NEXT:    s_cbranch_scc0 .LBB4_4461; GFX9-NEXT:  ; %bb.2: ; %.demote462; GFX9-NEXT:    s_wqm_b64 s[16:17], s[12:13]463; GFX9-NEXT:    s_and_b64 exec, exec, s[16:17]464; GFX9-NEXT:  .LBB4_3: ; %.continue465; GFX9-NEXT:    s_or_b64 exec, exec, s[14:15]466; GFX9-NEXT:    v_add_f32_e32 v0, v0, v0467; GFX9-NEXT:    s_and_b64 exec, exec, s[12:13]468; GFX9-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf469; GFX9-NEXT:    s_waitcnt vmcnt(0)470; GFX9-NEXT:    s_branch .LBB4_5471; GFX9-NEXT:  .LBB4_4:472; GFX9-NEXT:    s_mov_b64 exec, 0473; GFX9-NEXT:    exp null off, off, off, off done vm474; GFX9-NEXT:    s_endpgm475; GFX9-NEXT:  .LBB4_5:476;477; GFX10-32-LABEL: wqm_demote_2:478; GFX10-32:       ; %bb.0: ; %.entry479; GFX10-32-NEXT:    s_mov_b32 s12, exec_lo480; GFX10-32-NEXT:    s_wqm_b32 exec_lo, exec_lo481; GFX10-32-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1 dim:SQ_RSRC_IMG_1D482; GFX10-32-NEXT:    s_waitcnt vmcnt(0)483; GFX10-32-NEXT:    v_cmp_ngt_f32_e32 vcc_lo, 0, v0484; GFX10-32-NEXT:    s_and_saveexec_b32 s13, vcc_lo485; GFX10-32-NEXT:    s_xor_b32 s13, exec_lo, s13486; GFX10-32-NEXT:    s_cbranch_execz .LBB4_3487; GFX10-32-NEXT:  ; %bb.1: ; %.demote488; GFX10-32-NEXT:    s_andn2_b32 s12, s12, exec_lo489; GFX10-32-NEXT:    s_cbranch_scc0 .LBB4_4490; GFX10-32-NEXT:  ; %bb.2: ; %.demote491; GFX10-32-NEXT:    s_wqm_b32 s14, s12492; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s14493; GFX10-32-NEXT:  .LBB4_3: ; %.continue494; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s13495; GFX10-32-NEXT:    v_add_f32_e32 v0, v0, v0496; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s12497; GFX10-32-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D498; GFX10-32-NEXT:    s_waitcnt vmcnt(0)499; GFX10-32-NEXT:    s_branch .LBB4_5500; GFX10-32-NEXT:  .LBB4_4:501; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0502; GFX10-32-NEXT:    exp null off, off, off, off done vm503; GFX10-32-NEXT:    s_endpgm504; GFX10-32-NEXT:  .LBB4_5:505;506; GFX10-64-LABEL: wqm_demote_2:507; GFX10-64:       ; %bb.0: ; %.entry508; GFX10-64-NEXT:    s_mov_b64 s[12:13], exec509; GFX10-64-NEXT:    s_wqm_b64 exec, exec510; GFX10-64-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1 dim:SQ_RSRC_IMG_1D511; GFX10-64-NEXT:    s_waitcnt vmcnt(0)512; GFX10-64-NEXT:    v_cmp_ngt_f32_e32 vcc, 0, v0513; GFX10-64-NEXT:    s_and_saveexec_b64 s[14:15], vcc514; GFX10-64-NEXT:    s_xor_b64 s[14:15], exec, s[14:15]515; GFX10-64-NEXT:    s_cbranch_execz .LBB4_3516; GFX10-64-NEXT:  ; %bb.1: ; %.demote517; GFX10-64-NEXT:    s_andn2_b64 s[12:13], s[12:13], exec518; GFX10-64-NEXT:    s_cbranch_scc0 .LBB4_4519; GFX10-64-NEXT:  ; %bb.2: ; %.demote520; GFX10-64-NEXT:    s_wqm_b64 s[16:17], s[12:13]521; GFX10-64-NEXT:    s_and_b64 exec, exec, s[16:17]522; GFX10-64-NEXT:  .LBB4_3: ; %.continue523; GFX10-64-NEXT:    s_or_b64 exec, exec, s[14:15]524; GFX10-64-NEXT:    v_add_f32_e32 v0, v0, v0525; GFX10-64-NEXT:    s_and_b64 exec, exec, s[12:13]526; GFX10-64-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D527; GFX10-64-NEXT:    s_waitcnt vmcnt(0)528; GFX10-64-NEXT:    s_branch .LBB4_5529; GFX10-64-NEXT:  .LBB4_4:530; GFX10-64-NEXT:    s_mov_b64 exec, 0531; GFX10-64-NEXT:    exp null off, off, off, off done vm532; GFX10-64-NEXT:    s_endpgm533; GFX10-64-NEXT:  .LBB4_5:534.entry:535  %tex = call <4 x float> @llvm.amdgcn.image.sample.1d.v4f32.f32(i32 15, float %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i1 0, i32 0, i32 0) #0536  %tex0 = extractelement <4 x float> %tex, i32 0537  %tex1 = extractelement <4 x float> %tex, i32 0538  %z.cmp = fcmp olt float %tex0, 0.0539  br i1 %z.cmp, label %.continue, label %.demote540 541.demote:542  call void @llvm.amdgcn.wqm.demote(i1 false)543  br label %.continue544 545.continue:546  %coord1 = fadd float %tex0, %tex1547  %rtex = call <4 x float> @llvm.amdgcn.image.sample.1d.v4f32.f32(i32 15, float %coord1, <8 x i32> %rsrc, <4 x i32> %sampler, i1 0, i32 0, i32 0) #0548 549  ret <4 x float> %rtex550}551 552define amdgpu_ps <4 x float> @wqm_demote_dynamic(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %idx, float %data, float %coord, float %coord2, float %z) {553; SI-LABEL: wqm_demote_dynamic:554; SI:       ; %bb.0: ; %.entry555; SI-NEXT:    s_mov_b64 s[12:13], exec556; SI-NEXT:    s_wqm_b64 exec, exec557; SI-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1558; SI-NEXT:    s_waitcnt vmcnt(0)559; SI-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v0560; SI-NEXT:    s_andn2_b64 s[14:15], exec, vcc561; SI-NEXT:    s_andn2_b64 s[12:13], s[12:13], s[14:15]562; SI-NEXT:    s_cbranch_scc0 .LBB5_2563; SI-NEXT:  ; %bb.1: ; %.entry564; SI-NEXT:    s_wqm_b64 s[14:15], s[12:13]565; SI-NEXT:    s_and_b64 exec, exec, s[14:15]566; SI-NEXT:    v_add_f32_e32 v0, v0, v0567; SI-NEXT:    s_and_b64 exec, exec, s[12:13]568; SI-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf569; SI-NEXT:    s_waitcnt vmcnt(0)570; SI-NEXT:    s_branch .LBB5_3571; SI-NEXT:  .LBB5_2:572; SI-NEXT:    s_mov_b64 exec, 0573; SI-NEXT:    exp null off, off, off, off done vm574; SI-NEXT:    s_endpgm575; SI-NEXT:  .LBB5_3:576;577; GFX9-LABEL: wqm_demote_dynamic:578; GFX9:       ; %bb.0: ; %.entry579; GFX9-NEXT:    s_mov_b64 s[12:13], exec580; GFX9-NEXT:    s_wqm_b64 exec, exec581; GFX9-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1582; GFX9-NEXT:    s_waitcnt vmcnt(0)583; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v0584; GFX9-NEXT:    s_andn2_b64 s[14:15], exec, vcc585; GFX9-NEXT:    s_andn2_b64 s[12:13], s[12:13], s[14:15]586; GFX9-NEXT:    s_cbranch_scc0 .LBB5_2587; GFX9-NEXT:  ; %bb.1: ; %.entry588; GFX9-NEXT:    s_wqm_b64 s[14:15], s[12:13]589; GFX9-NEXT:    s_and_b64 exec, exec, s[14:15]590; GFX9-NEXT:    v_add_f32_e32 v0, v0, v0591; GFX9-NEXT:    s_and_b64 exec, exec, s[12:13]592; GFX9-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf593; GFX9-NEXT:    s_waitcnt vmcnt(0)594; GFX9-NEXT:    s_branch .LBB5_3595; GFX9-NEXT:  .LBB5_2:596; GFX9-NEXT:    s_mov_b64 exec, 0597; GFX9-NEXT:    exp null off, off, off, off done vm598; GFX9-NEXT:    s_endpgm599; GFX9-NEXT:  .LBB5_3:600;601; GFX10-32-LABEL: wqm_demote_dynamic:602; GFX10-32:       ; %bb.0: ; %.entry603; GFX10-32-NEXT:    s_mov_b32 s12, exec_lo604; GFX10-32-NEXT:    s_wqm_b32 exec_lo, exec_lo605; GFX10-32-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1 dim:SQ_RSRC_IMG_1D606; GFX10-32-NEXT:    s_waitcnt vmcnt(0)607; GFX10-32-NEXT:    v_cmp_gt_f32_e32 vcc_lo, 0, v0608; GFX10-32-NEXT:    s_andn2_b32 s13, exec_lo, vcc_lo609; GFX10-32-NEXT:    s_andn2_b32 s12, s12, s13610; GFX10-32-NEXT:    s_cbranch_scc0 .LBB5_2611; GFX10-32-NEXT:  ; %bb.1: ; %.entry612; GFX10-32-NEXT:    s_wqm_b32 s13, s12613; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s13614; GFX10-32-NEXT:    v_add_f32_e32 v0, v0, v0615; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s12616; GFX10-32-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D617; GFX10-32-NEXT:    s_waitcnt vmcnt(0)618; GFX10-32-NEXT:    s_branch .LBB5_3619; GFX10-32-NEXT:  .LBB5_2:620; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0621; GFX10-32-NEXT:    exp null off, off, off, off done vm622; GFX10-32-NEXT:    s_endpgm623; GFX10-32-NEXT:  .LBB5_3:624;625; GFX10-64-LABEL: wqm_demote_dynamic:626; GFX10-64:       ; %bb.0: ; %.entry627; GFX10-64-NEXT:    s_mov_b64 s[12:13], exec628; GFX10-64-NEXT:    s_wqm_b64 exec, exec629; GFX10-64-NEXT:    image_sample v0, v0, s[0:7], s[8:11] dmask:0x1 dim:SQ_RSRC_IMG_1D630; GFX10-64-NEXT:    s_waitcnt vmcnt(0)631; GFX10-64-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v0632; GFX10-64-NEXT:    s_andn2_b64 s[14:15], exec, vcc633; GFX10-64-NEXT:    s_andn2_b64 s[12:13], s[12:13], s[14:15]634; GFX10-64-NEXT:    s_cbranch_scc0 .LBB5_2635; GFX10-64-NEXT:  ; %bb.1: ; %.entry636; GFX10-64-NEXT:    s_wqm_b64 s[14:15], s[12:13]637; GFX10-64-NEXT:    s_and_b64 exec, exec, s[14:15]638; GFX10-64-NEXT:    v_add_f32_e32 v0, v0, v0639; GFX10-64-NEXT:    s_and_b64 exec, exec, s[12:13]640; GFX10-64-NEXT:    image_sample v[0:3], v0, s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D641; GFX10-64-NEXT:    s_waitcnt vmcnt(0)642; GFX10-64-NEXT:    s_branch .LBB5_3643; GFX10-64-NEXT:  .LBB5_2:644; GFX10-64-NEXT:    s_mov_b64 exec, 0645; GFX10-64-NEXT:    exp null off, off, off, off done vm646; GFX10-64-NEXT:    s_endpgm647; GFX10-64-NEXT:  .LBB5_3:648.entry:649  %tex = call <4 x float> @llvm.amdgcn.image.sample.1d.v4f32.f32(i32 15, float %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i1 0, i32 0, i32 0) #0650  %tex0 = extractelement <4 x float> %tex, i32 0651  %tex1 = extractelement <4 x float> %tex, i32 0652  %z.cmp = fcmp olt float %tex0, 0.0653  call void @llvm.amdgcn.wqm.demote(i1 %z.cmp)654  %coord1 = fadd float %tex0, %tex1655  %rtex = call <4 x float> @llvm.amdgcn.image.sample.1d.v4f32.f32(i32 15, float %coord1, <8 x i32> %rsrc, <4 x i32> %sampler, i1 0, i32 0, i32 0) #0656 657  ret <4 x float> %rtex658}659 660 661define amdgpu_ps void @wqm_deriv(<2 x float> %input, float %arg, i32 %index) {662; SI-LABEL: wqm_deriv:663; SI:       ; %bb.0: ; %.entry664; SI-NEXT:    s_mov_b64 s[0:1], exec665; SI-NEXT:    s_wqm_b64 exec, exec666; SI-NEXT:    v_cvt_i32_f32_e32 v0, v0667; SI-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v0668; SI-NEXT:    s_and_saveexec_b64 s[2:3], vcc669; SI-NEXT:    s_xor_b64 s[2:3], exec, s[2:3]670; SI-NEXT:    s_cbranch_execz .LBB6_3671; SI-NEXT:  ; %bb.1: ; %.demote0672; SI-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec673; SI-NEXT:    s_cbranch_scc0 .LBB6_7674; SI-NEXT:  ; %bb.2: ; %.demote0675; SI-NEXT:    s_wqm_b64 s[4:5], s[0:1]676; SI-NEXT:    s_and_b64 exec, exec, s[4:5]677; SI-NEXT:  .LBB6_3: ; %.continue0678; SI-NEXT:    s_or_b64 exec, exec, s[2:3]679; SI-NEXT:    s_mov_b64 s[2:3], s[0:1]680; SI-NEXT:    v_cndmask_b32_e64 v0, 1.0, 0, s[2:3]681; SI-NEXT:    v_mov_b32_e32 v1, v0682; SI-NEXT:    s_xor_b64 s[2:3], s[0:1], -1683; SI-NEXT:    s_nop 0684; SI-NEXT:    v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1685; SI-NEXT:    s_nop 1686; SI-NEXT:    v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1687; SI-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec688; SI-NEXT:    s_and_b64 exec, exec, s[0:1]689; SI-NEXT:    v_cmp_neq_f32_e32 vcc, 0, v0690; SI-NEXT:    s_or_b64 s[2:3], s[2:3], vcc691; SI-NEXT:    s_and_saveexec_b64 s[4:5], s[2:3]692; SI-NEXT:    s_xor_b64 s[2:3], exec, s[4:5]693; SI-NEXT:    s_cbranch_execz .LBB6_6694; SI-NEXT:  ; %bb.4: ; %.demote1695; SI-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec696; SI-NEXT:    s_cbranch_scc0 .LBB6_7697; SI-NEXT:  ; %bb.5: ; %.demote1698; SI-NEXT:    s_mov_b64 exec, 0699; SI-NEXT:  .LBB6_6: ; %.continue1700; SI-NEXT:    s_or_b64 exec, exec, s[2:3]701; SI-NEXT:    v_bfrev_b32_e32 v0, 60702; SI-NEXT:    v_mov_b32_e32 v1, 0x3c00703; SI-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm704; SI-NEXT:    s_endpgm705; SI-NEXT:  .LBB6_7:706; SI-NEXT:    s_mov_b64 exec, 0707; SI-NEXT:    exp null off, off, off, off done vm708; SI-NEXT:    s_endpgm709;710; GFX9-LABEL: wqm_deriv:711; GFX9:       ; %bb.0: ; %.entry712; GFX9-NEXT:    s_mov_b64 s[0:1], exec713; GFX9-NEXT:    s_wqm_b64 exec, exec714; GFX9-NEXT:    v_cvt_i32_f32_e32 v0, v0715; GFX9-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v0716; GFX9-NEXT:    s_and_saveexec_b64 s[2:3], vcc717; GFX9-NEXT:    s_xor_b64 s[2:3], exec, s[2:3]718; GFX9-NEXT:    s_cbranch_execz .LBB6_3719; GFX9-NEXT:  ; %bb.1: ; %.demote0720; GFX9-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec721; GFX9-NEXT:    s_cbranch_scc0 .LBB6_7722; GFX9-NEXT:  ; %bb.2: ; %.demote0723; GFX9-NEXT:    s_wqm_b64 s[4:5], s[0:1]724; GFX9-NEXT:    s_and_b64 exec, exec, s[4:5]725; GFX9-NEXT:  .LBB6_3: ; %.continue0726; GFX9-NEXT:    s_or_b64 exec, exec, s[2:3]727; GFX9-NEXT:    s_mov_b64 s[2:3], s[0:1]728; GFX9-NEXT:    v_cndmask_b32_e64 v0, 1.0, 0, s[2:3]729; GFX9-NEXT:    v_mov_b32_e32 v1, v0730; GFX9-NEXT:    s_xor_b64 s[2:3], s[0:1], -1731; GFX9-NEXT:    s_nop 0732; GFX9-NEXT:    v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1733; GFX9-NEXT:    s_nop 1734; GFX9-NEXT:    v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1735; GFX9-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec736; GFX9-NEXT:    s_and_b64 exec, exec, s[0:1]737; GFX9-NEXT:    v_cmp_neq_f32_e32 vcc, 0, v0738; GFX9-NEXT:    s_or_b64 s[2:3], s[2:3], vcc739; GFX9-NEXT:    s_and_saveexec_b64 s[4:5], s[2:3]740; GFX9-NEXT:    s_xor_b64 s[2:3], exec, s[4:5]741; GFX9-NEXT:    s_cbranch_execz .LBB6_6742; GFX9-NEXT:  ; %bb.4: ; %.demote1743; GFX9-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec744; GFX9-NEXT:    s_cbranch_scc0 .LBB6_7745; GFX9-NEXT:  ; %bb.5: ; %.demote1746; GFX9-NEXT:    s_mov_b64 exec, 0747; GFX9-NEXT:  .LBB6_6: ; %.continue1748; GFX9-NEXT:    s_or_b64 exec, exec, s[2:3]749; GFX9-NEXT:    v_bfrev_b32_e32 v0, 60750; GFX9-NEXT:    v_mov_b32_e32 v1, 0x3c00751; GFX9-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm752; GFX9-NEXT:    s_endpgm753; GFX9-NEXT:  .LBB6_7:754; GFX9-NEXT:    s_mov_b64 exec, 0755; GFX9-NEXT:    exp null off, off, off, off done vm756; GFX9-NEXT:    s_endpgm757;758; GFX10-32-LABEL: wqm_deriv:759; GFX10-32:       ; %bb.0: ; %.entry760; GFX10-32-NEXT:    s_mov_b32 s0, exec_lo761; GFX10-32-NEXT:    s_wqm_b32 exec_lo, exec_lo762; GFX10-32-NEXT:    v_cvt_i32_f32_e32 v0, v0763; GFX10-32-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v0764; GFX10-32-NEXT:    s_and_saveexec_b32 s1, vcc_lo765; GFX10-32-NEXT:    s_xor_b32 s1, exec_lo, s1766; GFX10-32-NEXT:    s_cbranch_execz .LBB6_3767; GFX10-32-NEXT:  ; %bb.1: ; %.demote0768; GFX10-32-NEXT:    s_andn2_b32 s0, s0, exec_lo769; GFX10-32-NEXT:    s_cbranch_scc0 .LBB6_7770; GFX10-32-NEXT:  ; %bb.2: ; %.demote0771; GFX10-32-NEXT:    s_wqm_b32 s2, s0772; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s2773; GFX10-32-NEXT:  .LBB6_3: ; %.continue0774; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s1775; GFX10-32-NEXT:    s_mov_b32 s1, s0776; GFX10-32-NEXT:    v_cndmask_b32_e64 v0, 1.0, 0, s1777; GFX10-32-NEXT:    v_mov_b32_e32 v1, v0778; GFX10-32-NEXT:    v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1779; GFX10-32-NEXT:    v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1780; GFX10-32-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec781; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s0782; GFX10-32-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v0783; GFX10-32-NEXT:    s_xor_b32 s1, s0, -1784; GFX10-32-NEXT:    s_or_b32 s1, s1, vcc_lo785; GFX10-32-NEXT:    s_and_saveexec_b32 s2, s1786; GFX10-32-NEXT:    s_xor_b32 s1, exec_lo, s2787; GFX10-32-NEXT:    s_cbranch_execz .LBB6_6788; GFX10-32-NEXT:  ; %bb.4: ; %.demote1789; GFX10-32-NEXT:    s_andn2_b32 s0, s0, exec_lo790; GFX10-32-NEXT:    s_cbranch_scc0 .LBB6_7791; GFX10-32-NEXT:  ; %bb.5: ; %.demote1792; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0793; GFX10-32-NEXT:  .LBB6_6: ; %.continue1794; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s1795; GFX10-32-NEXT:    v_bfrev_b32_e32 v0, 60796; GFX10-32-NEXT:    v_mov_b32_e32 v1, 0x3c00797; GFX10-32-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm798; GFX10-32-NEXT:    s_endpgm799; GFX10-32-NEXT:  .LBB6_7:800; GFX10-32-NEXT:    s_mov_b32 exec_lo, 0801; GFX10-32-NEXT:    exp null off, off, off, off done vm802; GFX10-32-NEXT:    s_endpgm803;804; GFX10-64-LABEL: wqm_deriv:805; GFX10-64:       ; %bb.0: ; %.entry806; GFX10-64-NEXT:    s_mov_b64 s[0:1], exec807; GFX10-64-NEXT:    s_wqm_b64 exec, exec808; GFX10-64-NEXT:    v_cvt_i32_f32_e32 v0, v0809; GFX10-64-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v0810; GFX10-64-NEXT:    s_and_saveexec_b64 s[2:3], vcc811; GFX10-64-NEXT:    s_xor_b64 s[2:3], exec, s[2:3]812; GFX10-64-NEXT:    s_cbranch_execz .LBB6_3813; GFX10-64-NEXT:  ; %bb.1: ; %.demote0814; GFX10-64-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec815; GFX10-64-NEXT:    s_cbranch_scc0 .LBB6_7816; GFX10-64-NEXT:  ; %bb.2: ; %.demote0817; GFX10-64-NEXT:    s_wqm_b64 s[4:5], s[0:1]818; GFX10-64-NEXT:    s_and_b64 exec, exec, s[4:5]819; GFX10-64-NEXT:  .LBB6_3: ; %.continue0820; GFX10-64-NEXT:    s_or_b64 exec, exec, s[2:3]821; GFX10-64-NEXT:    s_mov_b64 s[2:3], s[0:1]822; GFX10-64-NEXT:    v_cndmask_b32_e64 v0, 1.0, 0, s[2:3]823; GFX10-64-NEXT:    v_mov_b32_e32 v1, v0824; GFX10-64-NEXT:    v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1825; GFX10-64-NEXT:    v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1826; GFX10-64-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec827; GFX10-64-NEXT:    s_and_b64 exec, exec, s[0:1]828; GFX10-64-NEXT:    v_cmp_neq_f32_e32 vcc, 0, v0829; GFX10-64-NEXT:    s_xor_b64 s[2:3], s[0:1], -1830; GFX10-64-NEXT:    s_or_b64 s[2:3], s[2:3], vcc831; GFX10-64-NEXT:    s_and_saveexec_b64 s[4:5], s[2:3]832; GFX10-64-NEXT:    s_xor_b64 s[2:3], exec, s[4:5]833; GFX10-64-NEXT:    s_cbranch_execz .LBB6_6834; GFX10-64-NEXT:  ; %bb.4: ; %.demote1835; GFX10-64-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec836; GFX10-64-NEXT:    s_cbranch_scc0 .LBB6_7837; GFX10-64-NEXT:  ; %bb.5: ; %.demote1838; GFX10-64-NEXT:    s_mov_b64 exec, 0839; GFX10-64-NEXT:  .LBB6_6: ; %.continue1840; GFX10-64-NEXT:    s_or_b64 exec, exec, s[2:3]841; GFX10-64-NEXT:    v_bfrev_b32_e32 v0, 60842; GFX10-64-NEXT:    v_mov_b32_e32 v1, 0x3c00843; GFX10-64-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm844; GFX10-64-NEXT:    s_endpgm845; GFX10-64-NEXT:  .LBB6_7:846; GFX10-64-NEXT:    s_mov_b64 exec, 0847; GFX10-64-NEXT:    exp null off, off, off, off done vm848; GFX10-64-NEXT:    s_endpgm849.entry:850  %p0 = extractelement <2 x float> %input, i32 0851  %p1 = extractelement <2 x float> %input, i32 1852  %x0 = call float @llvm.amdgcn.interp.p1(float %p0, i32 0, i32 0, i32 %index) #2853  %x1 = call float @llvm.amdgcn.interp.p2(float %x0, float %p1, i32 0, i32 0, i32 %index) #2854  %argi = fptosi float %arg to i32855  %cond0 = icmp eq i32 %argi, 0856  br i1 %cond0, label %.continue0, label %.demote0857 858.demote0:859  call void @llvm.amdgcn.wqm.demote(i1 false)860  br label %.continue0861 862.continue0:863  %live = call i1 @llvm.amdgcn.live.mask()864  %live.cond = select i1 %live, i32 0, i32 1065353216865  %live.v0 = call i32 @llvm.amdgcn.mov.dpp.i32(i32 %live.cond, i32 85, i32 15, i32 15, i1 true)866  %live.v0f = bitcast i32 %live.v0 to float867  %live.v1 = call i32 @llvm.amdgcn.mov.dpp.i32(i32 %live.cond, i32 0, i32 15, i32 15, i1 true)868  %live.v1f = bitcast i32 %live.v1 to float869  %v0 = fsub float %live.v0f, %live.v1f870  %v0.wqm = call float @llvm.amdgcn.wqm.f32(float %v0)871  %cond1 = fcmp oeq float %v0.wqm, 0.000000e+00872  %cond2 = and i1 %live, %cond1873  br i1 %cond2, label %.continue1, label %.demote1874 875.demote1:876  call void @llvm.amdgcn.wqm.demote(i1 false)877  br label %.continue1878 879.continue1:880  call void @llvm.amdgcn.exp.compr.v2f16(i32 0, i32 15, <2 x half> <half 0xH3C00, half 0xH0000>, <2 x half> <half 0xH0000, half 0xH3C00>, i1 true, i1 true) #3881  ret void882}883 884define amdgpu_ps void @wqm_deriv_loop(<2 x float> %input, float %arg, i32 %index, i32 %limit) {885; SI-LABEL: wqm_deriv_loop:886; SI:       ; %bb.0: ; %.entry887; SI-NEXT:    s_mov_b64 s[0:1], exec888; SI-NEXT:    s_wqm_b64 exec, exec889; SI-NEXT:    v_cvt_i32_f32_e32 v0, v0890; SI-NEXT:    s_mov_b32 s6, 0891; SI-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v0892; SI-NEXT:    s_and_saveexec_b64 s[2:3], vcc893; SI-NEXT:    s_xor_b64 s[2:3], exec, s[2:3]894; SI-NEXT:    s_cbranch_execz .LBB7_3895; SI-NEXT:  ; %bb.1: ; %.demote0896; SI-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec897; SI-NEXT:    s_cbranch_scc0 .LBB7_9898; SI-NEXT:  ; %bb.2: ; %.demote0899; SI-NEXT:    s_wqm_b64 s[4:5], s[0:1]900; SI-NEXT:    s_and_b64 exec, exec, s[4:5]901; SI-NEXT:  .LBB7_3: ; %.continue0.preheader902; SI-NEXT:    s_or_b64 exec, exec, s[2:3]903; SI-NEXT:    s_mov_b64 s[2:3], 0904; SI-NEXT:    s_branch .LBB7_5905; SI-NEXT:  .LBB7_4: ; %.continue1906; SI-NEXT:    ; in Loop: Header=BB7_5 Depth=1907; SI-NEXT:    s_or_b64 exec, exec, s[4:5]908; SI-NEXT:    s_add_i32 s6, s6, 1909; SI-NEXT:    v_cmp_ge_i32_e32 vcc, s6, v1910; SI-NEXT:    s_or_b64 s[2:3], vcc, s[2:3]911; SI-NEXT:    s_andn2_b64 exec, exec, s[2:3]912; SI-NEXT:    s_cbranch_execz .LBB7_8913; SI-NEXT:  .LBB7_5: ; %.continue0914; SI-NEXT:    ; =>This Inner Loop Header: Depth=1915; SI-NEXT:    v_mov_b32_e32 v0, s6916; SI-NEXT:    s_mov_b64 s[4:5], s[0:1]917; SI-NEXT:    v_cndmask_b32_e64 v0, v0, 0, s[4:5]918; SI-NEXT:    v_mov_b32_e32 v2, v0919; SI-NEXT:    s_xor_b64 s[4:5], s[0:1], -1920; SI-NEXT:    s_nop 0921; SI-NEXT:    v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1922; SI-NEXT:    s_nop 1923; SI-NEXT:    v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1924; SI-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec925; SI-NEXT:    v_cmp_neq_f32_e32 vcc, 0, v0926; SI-NEXT:    s_or_b64 s[4:5], s[4:5], vcc927; SI-NEXT:    s_and_saveexec_b64 s[8:9], s[4:5]928; SI-NEXT:    s_xor_b64 s[4:5], exec, s[8:9]929; SI-NEXT:    s_cbranch_execz .LBB7_4930; SI-NEXT:  ; %bb.6: ; %.demote1931; SI-NEXT:    ; in Loop: Header=BB7_5 Depth=1932; SI-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec933; SI-NEXT:    s_cbranch_scc0 .LBB7_9934; SI-NEXT:  ; %bb.7: ; %.demote1935; SI-NEXT:    ; in Loop: Header=BB7_5 Depth=1936; SI-NEXT:    s_wqm_b64 s[8:9], s[0:1]937; SI-NEXT:    s_and_b64 exec, exec, s[8:9]938; SI-NEXT:    s_branch .LBB7_4939; SI-NEXT:  .LBB7_8: ; %.return940; SI-NEXT:    s_or_b64 exec, exec, s[2:3]941; SI-NEXT:    s_and_b64 exec, exec, s[0:1]942; SI-NEXT:    v_bfrev_b32_e32 v0, 60943; SI-NEXT:    v_mov_b32_e32 v1, 0x3c00944; SI-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm945; SI-NEXT:    s_endpgm946; SI-NEXT:  .LBB7_9:947; SI-NEXT:    s_mov_b64 exec, 0948; SI-NEXT:    exp null off, off, off, off done vm949; SI-NEXT:    s_endpgm950;951; GFX9-LABEL: wqm_deriv_loop:952; GFX9:       ; %bb.0: ; %.entry953; GFX9-NEXT:    s_mov_b64 s[0:1], exec954; GFX9-NEXT:    s_wqm_b64 exec, exec955; GFX9-NEXT:    v_cvt_i32_f32_e32 v0, v0956; GFX9-NEXT:    s_mov_b32 s6, 0957; GFX9-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v0958; GFX9-NEXT:    s_and_saveexec_b64 s[2:3], vcc959; GFX9-NEXT:    s_xor_b64 s[2:3], exec, s[2:3]960; GFX9-NEXT:    s_cbranch_execz .LBB7_3961; GFX9-NEXT:  ; %bb.1: ; %.demote0962; GFX9-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec963; GFX9-NEXT:    s_cbranch_scc0 .LBB7_9964; GFX9-NEXT:  ; %bb.2: ; %.demote0965; GFX9-NEXT:    s_wqm_b64 s[4:5], s[0:1]966; GFX9-NEXT:    s_and_b64 exec, exec, s[4:5]967; GFX9-NEXT:  .LBB7_3: ; %.continue0.preheader968; GFX9-NEXT:    s_or_b64 exec, exec, s[2:3]969; GFX9-NEXT:    s_mov_b64 s[2:3], 0970; GFX9-NEXT:    s_branch .LBB7_5971; GFX9-NEXT:  .LBB7_4: ; %.continue1972; GFX9-NEXT:    ; in Loop: Header=BB7_5 Depth=1973; GFX9-NEXT:    s_or_b64 exec, exec, s[4:5]974; GFX9-NEXT:    s_add_i32 s6, s6, 1975; GFX9-NEXT:    v_cmp_ge_i32_e32 vcc, s6, v1976; GFX9-NEXT:    s_or_b64 s[2:3], vcc, s[2:3]977; GFX9-NEXT:    s_andn2_b64 exec, exec, s[2:3]978; GFX9-NEXT:    s_cbranch_execz .LBB7_8979; GFX9-NEXT:  .LBB7_5: ; %.continue0980; GFX9-NEXT:    ; =>This Inner Loop Header: Depth=1981; GFX9-NEXT:    v_mov_b32_e32 v0, s6982; GFX9-NEXT:    s_mov_b64 s[4:5], s[0:1]983; GFX9-NEXT:    v_cndmask_b32_e64 v0, v0, 0, s[4:5]984; GFX9-NEXT:    v_mov_b32_e32 v2, v0985; GFX9-NEXT:    s_xor_b64 s[4:5], s[0:1], -1986; GFX9-NEXT:    s_nop 0987; GFX9-NEXT:    v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1988; GFX9-NEXT:    s_nop 1989; GFX9-NEXT:    v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1990; GFX9-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec991; GFX9-NEXT:    v_cmp_neq_f32_e32 vcc, 0, v0992; GFX9-NEXT:    s_or_b64 s[4:5], s[4:5], vcc993; GFX9-NEXT:    s_and_saveexec_b64 s[8:9], s[4:5]994; GFX9-NEXT:    s_xor_b64 s[4:5], exec, s[8:9]995; GFX9-NEXT:    s_cbranch_execz .LBB7_4996; GFX9-NEXT:  ; %bb.6: ; %.demote1997; GFX9-NEXT:    ; in Loop: Header=BB7_5 Depth=1998; GFX9-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec999; GFX9-NEXT:    s_cbranch_scc0 .LBB7_91000; GFX9-NEXT:  ; %bb.7: ; %.demote11001; GFX9-NEXT:    ; in Loop: Header=BB7_5 Depth=11002; GFX9-NEXT:    s_wqm_b64 s[8:9], s[0:1]1003; GFX9-NEXT:    s_and_b64 exec, exec, s[8:9]1004; GFX9-NEXT:    s_branch .LBB7_41005; GFX9-NEXT:  .LBB7_8: ; %.return1006; GFX9-NEXT:    s_or_b64 exec, exec, s[2:3]1007; GFX9-NEXT:    s_and_b64 exec, exec, s[0:1]1008; GFX9-NEXT:    v_bfrev_b32_e32 v0, 601009; GFX9-NEXT:    v_mov_b32_e32 v1, 0x3c001010; GFX9-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm1011; GFX9-NEXT:    s_endpgm1012; GFX9-NEXT:  .LBB7_9:1013; GFX9-NEXT:    s_mov_b64 exec, 01014; GFX9-NEXT:    exp null off, off, off, off done vm1015; GFX9-NEXT:    s_endpgm1016;1017; GFX10-32-LABEL: wqm_deriv_loop:1018; GFX10-32:       ; %bb.0: ; %.entry1019; GFX10-32-NEXT:    s_mov_b32 s0, exec_lo1020; GFX10-32-NEXT:    s_wqm_b32 exec_lo, exec_lo1021; GFX10-32-NEXT:    v_cvt_i32_f32_e32 v0, v01022; GFX10-32-NEXT:    s_mov_b32 s1, 01023; GFX10-32-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v01024; GFX10-32-NEXT:    s_and_saveexec_b32 s2, vcc_lo1025; GFX10-32-NEXT:    s_xor_b32 s2, exec_lo, s21026; GFX10-32-NEXT:    s_cbranch_execz .LBB7_31027; GFX10-32-NEXT:  ; %bb.1: ; %.demote01028; GFX10-32-NEXT:    s_andn2_b32 s0, s0, exec_lo1029; GFX10-32-NEXT:    s_cbranch_scc0 .LBB7_91030; GFX10-32-NEXT:  ; %bb.2: ; %.demote01031; GFX10-32-NEXT:    s_wqm_b32 s3, s01032; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s31033; GFX10-32-NEXT:  .LBB7_3: ; %.continue0.preheader1034; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s21035; GFX10-32-NEXT:    s_mov_b32 s2, 01036; GFX10-32-NEXT:    s_branch .LBB7_51037; GFX10-32-NEXT:  .LBB7_4: ; %.continue11038; GFX10-32-NEXT:    ; in Loop: Header=BB7_5 Depth=11039; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s31040; GFX10-32-NEXT:    s_add_i32 s2, s2, 11041; GFX10-32-NEXT:    v_cmp_ge_i32_e32 vcc_lo, s2, v11042; GFX10-32-NEXT:    s_or_b32 s1, vcc_lo, s11043; GFX10-32-NEXT:    s_andn2_b32 exec_lo, exec_lo, s11044; GFX10-32-NEXT:    s_cbranch_execz .LBB7_81045; GFX10-32-NEXT:  .LBB7_5: ; %.continue01046; GFX10-32-NEXT:    ; =>This Inner Loop Header: Depth=11047; GFX10-32-NEXT:    s_mov_b32 s3, s01048; GFX10-32-NEXT:    v_cndmask_b32_e64 v0, s2, 0, s31049; GFX10-32-NEXT:    s_xor_b32 s3, s0, -11050; GFX10-32-NEXT:    v_mov_b32_e32 v2, v01051; GFX10-32-NEXT:    v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:11052; GFX10-32-NEXT:    v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:11053; GFX10-32-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec1054; GFX10-32-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v01055; GFX10-32-NEXT:    s_or_b32 s3, s3, vcc_lo1056; GFX10-32-NEXT:    s_and_saveexec_b32 s4, s31057; GFX10-32-NEXT:    s_xor_b32 s3, exec_lo, s41058; GFX10-32-NEXT:    s_cbranch_execz .LBB7_41059; GFX10-32-NEXT:  ; %bb.6: ; %.demote11060; GFX10-32-NEXT:    ; in Loop: Header=BB7_5 Depth=11061; GFX10-32-NEXT:    s_andn2_b32 s0, s0, exec_lo1062; GFX10-32-NEXT:    s_cbranch_scc0 .LBB7_91063; GFX10-32-NEXT:  ; %bb.7: ; %.demote11064; GFX10-32-NEXT:    ; in Loop: Header=BB7_5 Depth=11065; GFX10-32-NEXT:    s_wqm_b32 s4, s01066; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s41067; GFX10-32-NEXT:    s_branch .LBB7_41068; GFX10-32-NEXT:  .LBB7_8: ; %.return1069; GFX10-32-NEXT:    s_or_b32 exec_lo, exec_lo, s11070; GFX10-32-NEXT:    s_and_b32 exec_lo, exec_lo, s01071; GFX10-32-NEXT:    v_bfrev_b32_e32 v0, 601072; GFX10-32-NEXT:    v_mov_b32_e32 v1, 0x3c001073; GFX10-32-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm1074; GFX10-32-NEXT:    s_endpgm1075; GFX10-32-NEXT:  .LBB7_9:1076; GFX10-32-NEXT:    s_mov_b32 exec_lo, 01077; GFX10-32-NEXT:    exp null off, off, off, off done vm1078; GFX10-32-NEXT:    s_endpgm1079;1080; GFX10-64-LABEL: wqm_deriv_loop:1081; GFX10-64:       ; %bb.0: ; %.entry1082; GFX10-64-NEXT:    s_mov_b64 s[0:1], exec1083; GFX10-64-NEXT:    s_wqm_b64 exec, exec1084; GFX10-64-NEXT:    v_cvt_i32_f32_e32 v0, v01085; GFX10-64-NEXT:    s_mov_b32 s6, 01086; GFX10-64-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v01087; GFX10-64-NEXT:    s_and_saveexec_b64 s[2:3], vcc1088; GFX10-64-NEXT:    s_xor_b64 s[2:3], exec, s[2:3]1089; GFX10-64-NEXT:    s_cbranch_execz .LBB7_31090; GFX10-64-NEXT:  ; %bb.1: ; %.demote01091; GFX10-64-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec1092; GFX10-64-NEXT:    s_cbranch_scc0 .LBB7_91093; GFX10-64-NEXT:  ; %bb.2: ; %.demote01094; GFX10-64-NEXT:    s_wqm_b64 s[4:5], s[0:1]1095; GFX10-64-NEXT:    s_and_b64 exec, exec, s[4:5]1096; GFX10-64-NEXT:  .LBB7_3: ; %.continue0.preheader1097; GFX10-64-NEXT:    s_or_b64 exec, exec, s[2:3]1098; GFX10-64-NEXT:    s_mov_b64 s[2:3], 01099; GFX10-64-NEXT:    s_branch .LBB7_51100; GFX10-64-NEXT:  .LBB7_4: ; %.continue11101; GFX10-64-NEXT:    ; in Loop: Header=BB7_5 Depth=11102; GFX10-64-NEXT:    s_or_b64 exec, exec, s[4:5]1103; GFX10-64-NEXT:    s_add_i32 s6, s6, 11104; GFX10-64-NEXT:    v_cmp_ge_i32_e32 vcc, s6, v11105; GFX10-64-NEXT:    s_or_b64 s[2:3], vcc, s[2:3]1106; GFX10-64-NEXT:    s_andn2_b64 exec, exec, s[2:3]1107; GFX10-64-NEXT:    s_cbranch_execz .LBB7_81108; GFX10-64-NEXT:  .LBB7_5: ; %.continue01109; GFX10-64-NEXT:    ; =>This Inner Loop Header: Depth=11110; GFX10-64-NEXT:    s_mov_b64 s[4:5], s[0:1]1111; GFX10-64-NEXT:    v_cndmask_b32_e64 v0, s6, 0, s[4:5]1112; GFX10-64-NEXT:    s_xor_b64 s[4:5], s[0:1], -11113; GFX10-64-NEXT:    v_mov_b32_e32 v2, v01114; GFX10-64-NEXT:    v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:11115; GFX10-64-NEXT:    v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:11116; GFX10-64-NEXT:    ; kill: def $vgpr0 killed $vgpr0 killed $exec1117; GFX10-64-NEXT:    v_cmp_neq_f32_e32 vcc, 0, v01118; GFX10-64-NEXT:    s_or_b64 s[4:5], s[4:5], vcc1119; GFX10-64-NEXT:    s_and_saveexec_b64 s[8:9], s[4:5]1120; GFX10-64-NEXT:    s_xor_b64 s[4:5], exec, s[8:9]1121; GFX10-64-NEXT:    s_cbranch_execz .LBB7_41122; GFX10-64-NEXT:  ; %bb.6: ; %.demote11123; GFX10-64-NEXT:    ; in Loop: Header=BB7_5 Depth=11124; GFX10-64-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec1125; GFX10-64-NEXT:    s_cbranch_scc0 .LBB7_91126; GFX10-64-NEXT:  ; %bb.7: ; %.demote11127; GFX10-64-NEXT:    ; in Loop: Header=BB7_5 Depth=11128; GFX10-64-NEXT:    s_wqm_b64 s[8:9], s[0:1]1129; GFX10-64-NEXT:    s_and_b64 exec, exec, s[8:9]1130; GFX10-64-NEXT:    s_branch .LBB7_41131; GFX10-64-NEXT:  .LBB7_8: ; %.return1132; GFX10-64-NEXT:    s_or_b64 exec, exec, s[2:3]1133; GFX10-64-NEXT:    s_and_b64 exec, exec, s[0:1]1134; GFX10-64-NEXT:    v_bfrev_b32_e32 v0, 601135; GFX10-64-NEXT:    v_mov_b32_e32 v1, 0x3c001136; GFX10-64-NEXT:    exp mrt0 v1, v1, v0, v0 done compr vm1137; GFX10-64-NEXT:    s_endpgm1138; GFX10-64-NEXT:  .LBB7_9:1139; GFX10-64-NEXT:    s_mov_b64 exec, 01140; GFX10-64-NEXT:    exp null off, off, off, off done vm1141; GFX10-64-NEXT:    s_endpgm1142.entry:1143  %p0 = extractelement <2 x float> %input, i32 01144  %p1 = extractelement <2 x float> %input, i32 11145  %x0 = call float @llvm.amdgcn.interp.p1(float %p0, i32 0, i32 0, i32 %index) #21146  %x1 = call float @llvm.amdgcn.interp.p2(float %x0, float %p1, i32 0, i32 0, i32 %index) #21147  %argi = fptosi float %arg to i321148  %cond0 = icmp eq i32 %argi, 01149  br i1 %cond0, label %.continue0, label %.demote01150 1151.demote0:1152  call void @llvm.amdgcn.wqm.demote(i1 false)1153  br label %.continue01154 1155.continue0:1156  %count = phi i32 [ 0, %.entry ], [ 0, %.demote0 ], [ %next, %.continue1 ]1157  %live = call i1 @llvm.amdgcn.live.mask()1158  %live.cond = select i1 %live, i32 0, i32 %count1159  %live.v0 = call i32 @llvm.amdgcn.mov.dpp.i32(i32 %live.cond, i32 85, i32 15, i32 15, i1 true)1160  %live.v0f = bitcast i32 %live.v0 to float1161  %live.v1 = call i32 @llvm.amdgcn.mov.dpp.i32(i32 %live.cond, i32 0, i32 15, i32 15, i1 true)1162  %live.v1f = bitcast i32 %live.v1 to float1163  %v0 = fsub float %live.v0f, %live.v1f1164  %v0.wqm = call float @llvm.amdgcn.wqm.f32(float %v0)1165  %cond1 = fcmp oeq float %v0.wqm, 0.000000e+001166  %cond2 = and i1 %live, %cond11167  br i1 %cond2, label %.continue1, label %.demote11168 1169.demote1:1170  call void @llvm.amdgcn.wqm.demote(i1 false)1171  br label %.continue11172 1173.continue1:1174  %next = add i32 %count, 11175  %loop.cond = icmp slt i32 %next, %limit1176  br i1 %loop.cond, label %.continue0, label %.return1177 1178.return:1179  call void @llvm.amdgcn.exp.compr.v2f16(i32 0, i32 15, <2 x half> <half 0xH3C00, half 0xH0000>, <2 x half> <half 0xH0000, half 0xH3C00>, i1 true, i1 true) #31180  ret void1181}1182 1183define amdgpu_ps void @static_exact_nop(float %arg0, float %arg1) {1184; SI-LABEL: static_exact_nop:1185; SI:       ; %bb.0: ; %.entry1186; SI-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v01187; SI-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc1188; SI-NEXT:    exp mrt1 v0, v0, v0, v0 done vm1189; SI-NEXT:    s_endpgm1190;1191; GFX9-LABEL: static_exact_nop:1192; GFX9:       ; %bb.0: ; %.entry1193; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v01194; GFX9-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc1195; GFX9-NEXT:    exp mrt1 v0, v0, v0, v0 done vm1196; GFX9-NEXT:    s_endpgm1197;1198; GFX10-32-LABEL: static_exact_nop:1199; GFX10-32:       ; %bb.0: ; %.entry1200; GFX10-32-NEXT:    v_cmp_gt_f32_e32 vcc_lo, 0, v01201; GFX10-32-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc_lo1202; GFX10-32-NEXT:    exp mrt1 v0, v0, v0, v0 done vm1203; GFX10-32-NEXT:    s_endpgm1204;1205; GFX10-64-LABEL: static_exact_nop:1206; GFX10-64:       ; %bb.0: ; %.entry1207; GFX10-64-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v01208; GFX10-64-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, vcc1209; GFX10-64-NEXT:    exp mrt1 v0, v0, v0, v0 done vm1210; GFX10-64-NEXT:    s_endpgm1211.entry:1212  %c0 = fcmp olt float %arg0, 0.000000e+001213  %c1 = fcmp oge float %arg1, 0.01214  call void @llvm.amdgcn.wqm.demote(i1 true)1215  %tmp1 = select i1 %c0, float 1.000000e+00, float 0.000000e+001216  call void @llvm.amdgcn.exp.f32(i32 1, i32 15, float %tmp1, float %tmp1, float %tmp1, float %tmp1, i1 true, i1 true) #01217  ret void1218}1219 1220 1221declare void @llvm.amdgcn.wqm.demote(i1) #01222declare i1 @llvm.amdgcn.live.mask() #01223declare void @llvm.amdgcn.exp.f32(i32, i32, float, float, float, float, i1, i1) #01224declare <4 x float> @llvm.amdgcn.image.sample.1d.v4f32.f32(i32, float, <8 x i32>, <4 x i32>, i1, i32, i32) #11225declare float @llvm.amdgcn.wqm.f32(float) #11226declare float @llvm.amdgcn.interp.p1(float, i32 immarg, i32 immarg, i32) #21227declare float @llvm.amdgcn.interp.p2(float, float, i32 immarg, i32 immarg, i32) #21228declare void @llvm.amdgcn.exp.compr.v2f16(i32 immarg, i32 immarg, <2 x half>, <2 x half>, i1 immarg, i1 immarg) #31229declare i32 @llvm.amdgcn.mov.dpp.i32(i32, i32 immarg, i32 immarg, i32 immarg, i1 immarg) #41230 1231attributes #0 = { nounwind }1232attributes #1 = { nounwind readnone }1233attributes #2 = { nounwind readnone speculatable }1234attributes #3 = { inaccessiblememonly nounwind }1235attributes #4 = { convergent nounwind readnone }1236