59 lines · plain
1; RUN: opt -passes=loop-unroll -S -mtriple=amdgcn-- -mcpu=tahiti %s | FileCheck %s2 3; This IR comes from this OpenCL C code:4;5; if (b + 4 > a) {6; for (int i = 0; i < 4; i++, b++) {7; if (b + 1 <= a)8; *(dst + c + b) = 0;9; else10; break;11; }12; }13;14; This test is meant to check that this loop isn't unrolled into more than15; four iterations. The loop unrolling preferences we currently use cause this16; loop to not be unrolled at all, but that may change in the future.17 18; CHECK-LABEL: @test19; CHECK: store i8 0, ptr addrspace(1)20; CHECK-NOT: store i8 0, ptr addrspace(1)21; CHECK: ret void22define amdgpu_kernel void @test(ptr addrspace(1) nocapture %dst, i32 %a, i32 %b, i32 %c) {23entry:24 %add = add nsw i32 %b, 425 %cmp = icmp sgt i32 %add, %a26 br i1 %cmp, label %for.cond.preheader, label %if.end727 28for.cond.preheader: ; preds = %entry29 %cmp313 = icmp slt i32 %b, %a30 br i1 %cmp313, label %if.then4.lr.ph, label %if.end7.loopexit31 32if.then4.lr.ph: ; preds = %for.cond.preheader33 %0 = sext i32 %c to i6434 br label %if.then435 36if.then4: ; preds = %if.then4.lr.ph, %if.then437 %i.015 = phi i32 [ 0, %if.then4.lr.ph ], [ %inc, %if.then4 ]38 %b.addr.014 = phi i32 [ %b, %if.then4.lr.ph ], [ %add2, %if.then4 ]39 %add2 = add nsw i32 %b.addr.014, 140 %1 = sext i32 %b.addr.014 to i6441 %add.ptr.sum = add nsw i64 %1, %042 %add.ptr5 = getelementptr inbounds i8, ptr addrspace(1) %dst, i64 %add.ptr.sum43 store i8 0, ptr addrspace(1) %add.ptr5, align 144 %inc = add nsw i32 %i.015, 145 %cmp1 = icmp slt i32 %inc, 446 %cmp3 = icmp slt i32 %add2, %a47 %or.cond = and i1 %cmp3, %cmp148 br i1 %or.cond, label %if.then4, label %for.cond.if.end7.loopexit_crit_edge49 50for.cond.if.end7.loopexit_crit_edge: ; preds = %if.then451 br label %if.end7.loopexit52 53if.end7.loopexit: ; preds = %for.cond.if.end7.loopexit_crit_edge, %for.cond.preheader54 br label %if.end755 56if.end7: ; preds = %if.end7.loopexit, %entry57 ret void58}59