brintos

brintos / llvm-project-archived public Read only

0
0
Text · 13.8 KiB · d94fa64 Raw
238 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 52; RUN: llc -mtriple aarch64 -mattr=+sve2 -o - %s | FileCheck %s3 4define void @test(ptr nocapture noundef readonly %kernel, i32 noundef %kw, float noundef nofpclass(nan inf) %kernel_factor, ptr %call5.i.i.i119) vscale_range(1, 16) {5; CHECK-LABEL: test:6; CHECK:       // %bb.0: // %entry7; CHECK-NEXT:    cmp w1, #18; CHECK-NEXT:    b.lt .LBB0_69; CHECK-NEXT:  // %bb.1: // %for.body.lr.ph10; CHECK-NEXT:    rdvl x8, #-211; CHECK-NEXT:    mov w9, #608 // =0x26012; CHECK-NEXT:    ands x11, x8, x913; CHECK-NEXT:    b.eq .LBB0_614; CHECK-NEXT:  // %bb.2: // %for.body.us.preheader15; CHECK-NEXT:    ptrue p0.h16; CHECK-NEXT:    add x11, x2, x11, lsl #117; CHECK-NEXT:    mov w8, wzr18; CHECK-NEXT:    ptrue p1.b19; CHECK-NEXT:    mov x9, xzr20; CHECK-NEXT:    mov w10, wzr21; CHECK-NEXT:    mov x12, #4 // =0x422; CHECK-NEXT:    mov x13, #8 // =0x823; CHECK-NEXT:  .LBB0_3: // %for.body.us24; CHECK-NEXT:    // =>This Loop Header: Depth=125; CHECK-NEXT:    // Child Loop BB0_4 Depth 226; CHECK-NEXT:    add x14, x0, x9, lsl #227; CHECK-NEXT:    sbfiz x15, x8, #1, #3228; CHECK-NEXT:    mov x16, x229; CHECK-NEXT:    ldp s0, s1, [x14]30; CHECK-NEXT:    add x15, x15, #831; CHECK-NEXT:    ldp s2, s3, [x14, #8]32; CHECK-NEXT:    ubfiz x14, x8, #1, #3233; CHECK-NEXT:    fcvt h0, s034; CHECK-NEXT:    fcvt h1, s135; CHECK-NEXT:    fcvt h2, s236; CHECK-NEXT:    fcvt h3, s337; CHECK-NEXT:    mov z0.h, h038; CHECK-NEXT:    mov z1.h, h139; CHECK-NEXT:    mov z2.h, h240; CHECK-NEXT:    mov z3.h, h341; CHECK-NEXT:  .LBB0_4: // %for.cond.i.preheader.us42; CHECK-NEXT:    // Parent Loop BB0_3 Depth=143; CHECK-NEXT:    // => This Inner Loop Header: Depth=244; CHECK-NEXT:    ld1b { z4.b }, p1/z, [x16, x14]45; CHECK-NEXT:    ldr z5, [x16]46; CHECK-NEXT:    add x17, x16, x1547; CHECK-NEXT:    add x18, x16, x1448; CHECK-NEXT:    add x3, x17, #849; CHECK-NEXT:    add x4, x17, #1650; CHECK-NEXT:    fmad z4.h, p0/m, z0.h, z5.h51; CHECK-NEXT:    ld1b { z5.b }, p1/z, [x16, x15]52; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z1.h53; CHECK-NEXT:    ld1h { z5.h }, p0/z, [x17, x12, lsl #1]54; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z2.h55; CHECK-NEXT:    ld1h { z5.h }, p0/z, [x17, x13, lsl #1]56; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z3.h57; CHECK-NEXT:    ldr z5, [x16, #1, mul vl]58; CHECK-NEXT:    str z4, [x16]59; CHECK-NEXT:    ldr z4, [x18, #1, mul vl]60; CHECK-NEXT:    fmad z4.h, p0/m, z0.h, z5.h61; CHECK-NEXT:    ldr z5, [x17, #1, mul vl]62; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z1.h63; CHECK-NEXT:    ldr z5, [x3, #1, mul vl]64; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z2.h65; CHECK-NEXT:    ldr z5, [x4, #1, mul vl]66; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z3.h67; CHECK-NEXT:    ldr z5, [x16, #2, mul vl]68; CHECK-NEXT:    str z4, [x16, #1, mul vl]69; CHECK-NEXT:    ldr z4, [x18, #2, mul vl]70; CHECK-NEXT:    fmad z4.h, p0/m, z0.h, z5.h71; CHECK-NEXT:    ldr z5, [x17, #2, mul vl]72; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z1.h73; CHECK-NEXT:    ldr z5, [x3, #2, mul vl]74; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z2.h75; CHECK-NEXT:    ldr z5, [x4, #2, mul vl]76; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z3.h77; CHECK-NEXT:    ldr z5, [x16, #3, mul vl]78; CHECK-NEXT:    str z4, [x16, #2, mul vl]79; CHECK-NEXT:    ldr z4, [x18, #3, mul vl]80; CHECK-NEXT:    fmad z4.h, p0/m, z0.h, z5.h81; CHECK-NEXT:    ldr z5, [x17, #3, mul vl]82; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z1.h83; CHECK-NEXT:    ldr z5, [x3, #3, mul vl]84; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z2.h85; CHECK-NEXT:    ldr z5, [x4, #3, mul vl]86; CHECK-NEXT:    fmla z4.h, p0/m, z5.h, z3.h87; CHECK-NEXT:    str z4, [x16, #3, mul vl]88; CHECK-NEXT:    incb x16, all, mul #489; CHECK-NEXT:    cmp x16, x1190; CHECK-NEXT:    b.lo .LBB0_491; CHECK-NEXT:  // %bb.5: // %while.cond.i..exit_crit_edge.us92; CHECK-NEXT:    // in Loop: Header=BB0_3 Depth=193; CHECK-NEXT:    add w10, w10, #194; CHECK-NEXT:    add x9, x9, #495; CHECK-NEXT:    add w8, w8, #1696; CHECK-NEXT:    cmp w10, w197; CHECK-NEXT:    b.ne .LBB0_398; CHECK-NEXT:  .LBB0_6: // %exit7899; CHECK-NEXT:    ret100entry:101  ;%call5.i.i.i119 = tail call noalias noundef nonnull dereferenceable(1248) ptr @_Znwm(i64 noundef 1248) #7102  %cmp139 = icmp sgt i32 %kw, 0103  ;tail call void @llvm.memset.p0.i64(ptr noundef nonnull align 2 dereferenceable(1248) %call5.i.i.i119, i8 0, i64 1248, i1 false)104  br i1 %cmp139, label %for.body.lr.ph, label %exit78105 106for.body.lr.ph:                                   ; preds = %entry107  %0 = tail call <vscale x 8 x i1> @llvm.aarch64.sve.ptrue.nxv8i1(i32 31)108  %vscale = tail call i64 @llvm.vscale.i64()109  %mul5.i = shl nuw nsw i64 %vscale, 5110  %sub.not.i = sub nsw i64 0, %mul5.i111  %sub6.i = and i64 %sub.not.i, 608112  %add.ptr.i = getelementptr inbounds half, ptr %call5.i.i.i119, i64 %sub6.i113  %cmp.i133.not = icmp eq i64 %sub6.i, 0114  %vs2 = shl nuw nsw i64 %vscale, 4115  br i1 %cmp.i133.not, label %exit78, label %for.body.us.preheader116 117for.body.us.preheader:                            ; preds = %for.body.lr.ph118  %.idx.i.us.2 = shl nuw nsw i64 %vscale, 5119  %.idx.i.us.3 = mul nuw nsw i64 %vscale, 48120  br label %for.body.us121 122for.body.us:                                      ; preds = %for.body.us.preheader, %while.cond.i..exit_crit_edge.us123  %indvars.iv = phi i64 [ 0, %for.body.us.preheader ], [ %indvars.iv.next, %while.cond.i..exit_crit_edge.us ]124  %i4.0140.us = phi i32 [ 0, %for.body.us.preheader ], [ %inc.us, %while.cond.i..exit_crit_edge.us ]125  %3 = trunc nuw nsw i64 %indvars.iv to i32126  %mul6.us = shl i32 %3, 2127  %idx.ext.us = zext nneg i32 %mul6.us to i64128  %add.ptr.us = getelementptr inbounds half, ptr %call5.i.i.i119, i64 %idx.ext.us129  %mul11.us = or disjoint i32 %mul6.us, 4130  %idx.ext12.us = sext i32 %mul11.us to i64131  %add.ptr13.us = getelementptr inbounds half, ptr %call5.i.i.i119, i64 %idx.ext12.us132  %mul18.us = or disjoint i32 %mul6.us, 8133  %idx.ext19.us = sext i32 %mul18.us to i64134  %add.ptr20.us = getelementptr inbounds half, ptr %call5.i.i.i119, i64 %idx.ext19.us135  %mul25.us = or disjoint i32 %mul6.us, 12136  %idx.ext26.us = sext i32 %mul25.us to i64137  %add.ptr27.us = getelementptr inbounds half, ptr %call5.i.i.i119, i64 %idx.ext26.us138  %add.ptr29.us = getelementptr inbounds float, ptr %kernel, i64 %indvars.iv139  %4 = load float, ptr %add.ptr29.us, align 4140  %5 = fptrunc float %4 to half141  %.splatinsert.i.us = insertelement <vscale x 8 x half> poison, half %5, i64 0142  %6 = shufflevector <vscale x 8 x half> %.splatinsert.i.us, <vscale x 8 x half> poison, <vscale x 8 x i32> zeroinitializer143  %arrayidx2.i.us = getelementptr inbounds i8, ptr %add.ptr29.us, i64 4144  %7 = load float, ptr %arrayidx2.i.us, align 4145  %8 = fptrunc float %7 to half146  %.splatinsert57.i.us = insertelement <vscale x 8 x half> poison, half %8, i64 0147  %9 = shufflevector <vscale x 8 x half> %.splatinsert57.i.us, <vscale x 8 x half> poison, <vscale x 8 x i32> zeroinitializer148  %arrayidx3.i.us = getelementptr inbounds i8, ptr %add.ptr29.us, i64 8149  %10 = load float, ptr %arrayidx3.i.us, align 4150  %11 = fptrunc float %10 to half151  %.splatinsert58.i.us = insertelement <vscale x 8 x half> poison, half %11, i64 0152  %12 = shufflevector <vscale x 8 x half> %.splatinsert58.i.us, <vscale x 8 x half> poison, <vscale x 8 x i32> zeroinitializer153  %arrayidx4.i.us = getelementptr inbounds i8, ptr %add.ptr29.us, i64 12154  %13 = load float, ptr %arrayidx4.i.us, align 4155  %14 = fptrunc float %13 to half156  %.splatinsert59.i.us = insertelement <vscale x 8 x half> poison, half %14, i64 0157  %15 = shufflevector <vscale x 8 x half> %.splatinsert59.i.us, <vscale x 8 x half> poison, <vscale x 8 x i32> zeroinitializer158  br label %for.cond.i.preheader.us159 160for.cond.i.preheader.us:                          ; preds = %for.body.us, %for.cond.i.preheader.us161  %vdst.0.i138.us = phi ptr [ %call5.i.i.i119, %for.body.us ], [ %add.ptr15.i.us, %for.cond.i.preheader.us ]162  %s1.0.i137.us = phi ptr [ %add.ptr.us, %for.body.us ], [ %add.ptr16.i.us, %for.cond.i.preheader.us ]163  %s2.0.i136.us = phi ptr [ %add.ptr13.us, %for.body.us ], [ %add.ptr17.i.us, %for.cond.i.preheader.us ]164  %s3.0.i135.us = phi ptr [ %add.ptr20.us, %for.body.us ], [ %add.ptr18.i.us, %for.cond.i.preheader.us ]165  %s4.0.i134.us = phi ptr [ %add.ptr27.us, %for.body.us ], [ %add.ptr19.i.us, %for.cond.i.preheader.us ]166  %16 = load <vscale x 8 x half>, ptr %s1.0.i137.us, align 16167  %17 = load <vscale x 8 x half>, ptr %vdst.0.i138.us, align 16168  %18 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %17, <vscale x 8 x half> %16, <vscale x 8 x half> %6)169  %19 = load <vscale x 8 x half>, ptr %s2.0.i136.us, align 16170  %20 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %18, <vscale x 8 x half> %19, <vscale x 8 x half> %9)171  %21 = load <vscale x 8 x half>, ptr %s3.0.i135.us, align 16172  %22 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %20, <vscale x 8 x half> %21, <vscale x 8 x half> %12)173  %23 = load <vscale x 8 x half>, ptr %s4.0.i134.us, align 16174  %24 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %22, <vscale x 8 x half> %23, <vscale x 8 x half> %15)175  store <vscale x 8 x half> %24, ptr %vdst.0.i138.us, align 16176  %25 = getelementptr i8, ptr %s1.0.i137.us, i64 %vs2177  %26 = load <vscale x 8 x half>, ptr %25, align 16178  %27 = getelementptr i8, ptr %vdst.0.i138.us, i64 %vs2179  %28 = load <vscale x 8 x half>, ptr %27, align 16180  %29 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %28, <vscale x 8 x half> %26, <vscale x 8 x half> %6)181  %30 = getelementptr i8, ptr %s2.0.i136.us, i64 %vs2182  %31 = load <vscale x 8 x half>, ptr %30, align 16183  %32 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %29, <vscale x 8 x half> %31, <vscale x 8 x half> %9)184  %33 = getelementptr i8, ptr %s3.0.i135.us, i64 %vs2185  %34 = load <vscale x 8 x half>, ptr %33, align 16186  %35 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %32, <vscale x 8 x half> %34, <vscale x 8 x half> %12)187  %36 = getelementptr i8, ptr %s4.0.i134.us, i64 %vs2188  %37 = load <vscale x 8 x half>, ptr %36, align 16189  %38 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %35, <vscale x 8 x half> %37, <vscale x 8 x half> %15)190  store <vscale x 8 x half> %38, ptr %27, align 16191  %39 = getelementptr i8, ptr %s1.0.i137.us, i64 %.idx.i.us.2192  %40 = load <vscale x 8 x half>, ptr %39, align 16193  %41 = getelementptr i8, ptr %vdst.0.i138.us, i64 %.idx.i.us.2194  %42 = load <vscale x 8 x half>, ptr %41, align 16195  %43 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %42, <vscale x 8 x half> %40, <vscale x 8 x half> %6)196  %44 = getelementptr i8, ptr %s2.0.i136.us, i64 %.idx.i.us.2197  %45 = load <vscale x 8 x half>, ptr %44, align 16198  %46 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %43, <vscale x 8 x half> %45, <vscale x 8 x half> %9)199  %47 = getelementptr i8, ptr %s3.0.i135.us, i64 %.idx.i.us.2200  %48 = load <vscale x 8 x half>, ptr %47, align 16201  %49 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %46, <vscale x 8 x half> %48, <vscale x 8 x half> %12)202  %50 = getelementptr i8, ptr %s4.0.i134.us, i64 %.idx.i.us.2203  %51 = load <vscale x 8 x half>, ptr %50, align 16204  %52 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %49, <vscale x 8 x half> %51, <vscale x 8 x half> %15)205  store <vscale x 8 x half> %52, ptr %41, align 16206  %53 = getelementptr i8, ptr %s1.0.i137.us, i64 %.idx.i.us.3207  %54 = load <vscale x 8 x half>, ptr %53, align 16208  %55 = getelementptr i8, ptr %vdst.0.i138.us, i64 %.idx.i.us.3209  %56 = load <vscale x 8 x half>, ptr %55, align 16210  %57 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %56, <vscale x 8 x half> %54, <vscale x 8 x half> %6)211  %58 = getelementptr i8, ptr %s2.0.i136.us, i64 %.idx.i.us.3212  %59 = load <vscale x 8 x half>, ptr %58, align 16213  %60 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %57, <vscale x 8 x half> %59, <vscale x 8 x half> %9)214  %61 = getelementptr i8, ptr %s3.0.i135.us, i64 %.idx.i.us.3215  %62 = load <vscale x 8 x half>, ptr %61, align 16216  %63 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %60, <vscale x 8 x half> %62, <vscale x 8 x half> %12)217  %64 = getelementptr i8, ptr %s4.0.i134.us, i64 %.idx.i.us.3218  %65 = load <vscale x 8 x half>, ptr %64, align 16219  %66 = tail call fast <vscale x 8 x half> @llvm.aarch64.sve.fmla.u.nxv8f16(<vscale x 8 x i1> %0, <vscale x 8 x half> %63, <vscale x 8 x half> %65, <vscale x 8 x half> %15)220  store <vscale x 8 x half> %66, ptr %55, align 16221  %add.ptr15.i.us = getelementptr inbounds half, ptr %vdst.0.i138.us, i64 %mul5.i222  %add.ptr16.i.us = getelementptr inbounds half, ptr %s1.0.i137.us, i64 %mul5.i223  %add.ptr17.i.us = getelementptr inbounds half, ptr %s2.0.i136.us, i64 %mul5.i224  %add.ptr18.i.us = getelementptr inbounds half, ptr %s3.0.i135.us, i64 %mul5.i225  %add.ptr19.i.us = getelementptr inbounds half, ptr %s4.0.i134.us, i64 %mul5.i226  %cmp.i.us = icmp ult ptr %add.ptr15.i.us, %add.ptr.i227  br i1 %cmp.i.us, label %for.cond.i.preheader.us, label %while.cond.i..exit_crit_edge.us228 229while.cond.i..exit_crit_edge.us: ; preds = %for.cond.i.preheader.us230  %indvars.iv.next = add nuw nsw i64 %indvars.iv, 4231  %inc.us = add nuw nsw i32 %i4.0140.us, 1232  %exitcond.not = icmp eq i32 %inc.us, %kw233  br i1 %exitcond.not, label %exit78, label %for.body.us234 235exit78:                      ; preds = %while.cond.i..exit_crit_edge.us, %for.body.lr.ph, %entry236  ret void237}238