221 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py2; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2 -force-streaming -enable-subreg-liveness -verify-machineinstrs < %s | FileCheck %s3 4;5; SQCVT6;7 8; x29define <vscale x 8 x i16 > @multi_vector_qcvt_x2_s16_s32(<vscale x 4 x i32> %unused, <vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2) {10; CHECK-LABEL: multi_vector_qcvt_x2_s16_s32:11; CHECK: // %bb.0:12; CHECK-NEXT: mov z3.d, z2.d13; CHECK-NEXT: mov z2.d, z1.d14; CHECK-NEXT: sqcvt z0.h, { z2.s, z3.s }15; CHECK-NEXT: ret16 %res = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x2.nxv4i32(<vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2)17 ret <vscale x 8 x i16> %res18}19 20; x421define <vscale x 16 x i8 > @multi_vector_qcvt_x4_s8_s32(<vscale x 4 x i32> %unused, <vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2, <vscale x 4 x i32> %zn3, <vscale x 4 x i32> %zn4) {22; CHECK-LABEL: multi_vector_qcvt_x4_s8_s32:23; CHECK: // %bb.0:24; CHECK-NEXT: mov z7.d, z4.d25; CHECK-NEXT: mov z6.d, z3.d26; CHECK-NEXT: mov z5.d, z2.d27; CHECK-NEXT: mov z4.d, z1.d28; CHECK-NEXT: sqcvt z0.b, { z4.s - z7.s }29; CHECK-NEXT: ret30 %res = call <vscale x 16 x i8> @llvm.aarch64.sve.sqcvt.x4.nxv4i32(<vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2, <vscale x 4 x i32> %zn3, <vscale x 4 x i32> %zn4)31 ret <vscale x 16 x i8> %res32}33 34define <vscale x 8 x i16> @multi_vector_qcvt_x4_s16_s64(<vscale x 2 x i64> %unused, <vscale x 2 x i64> %zn1, <vscale x 2 x i64> %zn2, <vscale x 2 x i64> %zn3, <vscale x 2 x i64> %zn4) {35; CHECK-LABEL: multi_vector_qcvt_x4_s16_s64:36; CHECK: // %bb.0:37; CHECK-NEXT: mov z7.d, z4.d38; CHECK-NEXT: mov z6.d, z3.d39; CHECK-NEXT: mov z5.d, z2.d40; CHECK-NEXT: mov z4.d, z1.d41; CHECK-NEXT: sqcvt z0.h, { z4.d - z7.d }42; CHECK-NEXT: ret43 %res = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x4.nxv2i64(<vscale x 2 x i64> %zn1, <vscale x 2 x i64> %zn2, <vscale x 2 x i64> %zn3, <vscale x 2 x i64> %zn4)44 ret <vscale x 8 x i16> %res45}46 47define { <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16> } @multi_vector_qcvt_x4_s16_s64_tuple(i64 %stride, ptr %ptr) {48; CHECK-LABEL: multi_vector_qcvt_x4_s16_s64_tuple:49; CHECK: // %bb.0: // %entry50; CHECK-NEXT: str x29, [sp, #-16]! // 8-byte Folded Spill51; CHECK-NEXT: addvl sp, sp, #-952; CHECK-NEXT: str p8, [sp, #7, mul vl] // 2-byte Spill53; CHECK-NEXT: str z23, [sp, #1, mul vl] // 16-byte Folded Spill54; CHECK-NEXT: str z22, [sp, #2, mul vl] // 16-byte Folded Spill55; CHECK-NEXT: str z21, [sp, #3, mul vl] // 16-byte Folded Spill56; CHECK-NEXT: str z20, [sp, #4, mul vl] // 16-byte Folded Spill57; CHECK-NEXT: str z19, [sp, #5, mul vl] // 16-byte Folded Spill58; CHECK-NEXT: str z18, [sp, #6, mul vl] // 16-byte Folded Spill59; CHECK-NEXT: str z17, [sp, #7, mul vl] // 16-byte Folded Spill60; CHECK-NEXT: str z16, [sp, #8, mul vl] // 16-byte Folded Spill61; CHECK-NEXT: .cfi_escape 0x0f, 0x0a, 0x8f, 0x10, 0x92, 0x2e, 0x00, 0x11, 0xc8, 0x00, 0x1e, 0x22 // sp + 16 + 72 * VG62; CHECK-NEXT: .cfi_offset w29, -1663; CHECK-NEXT: lsl x8, x0, #164; CHECK-NEXT: add x9, x1, x065; CHECK-NEXT: ptrue pn8.b66; CHECK-NEXT: ld1d { z16.d, z20.d, z24.d, z28.d }, pn8/z, [x1]67; CHECK-NEXT: ld1d { z17.d, z21.d, z25.d, z29.d }, pn8/z, [x9]68; CHECK-NEXT: add x10, x1, x869; CHECK-NEXT: add x8, x9, x870; CHECK-NEXT: ld1d { z18.d, z22.d, z26.d, z30.d }, pn8/z, [x10]71; CHECK-NEXT: ld1d { z19.d, z23.d, z27.d, z31.d }, pn8/z, [x8]72; CHECK-NEXT: sqcvt z0.h, { z16.d - z19.d }73; CHECK-NEXT: sqcvt z1.h, { z20.d - z23.d }74; CHECK-NEXT: sqcvt z2.h, { z24.d - z27.d }75; CHECK-NEXT: sqcvt z3.h, { z28.d - z31.d }76; CHECK-NEXT: ldr z23, [sp, #1, mul vl] // 16-byte Folded Reload77; CHECK-NEXT: ldr z22, [sp, #2, mul vl] // 16-byte Folded Reload78; CHECK-NEXT: ldr z21, [sp, #3, mul vl] // 16-byte Folded Reload79; CHECK-NEXT: ldr z20, [sp, #4, mul vl] // 16-byte Folded Reload80; CHECK-NEXT: ldr z19, [sp, #5, mul vl] // 16-byte Folded Reload81; CHECK-NEXT: ldr z18, [sp, #6, mul vl] // 16-byte Folded Reload82; CHECK-NEXT: ldr z17, [sp, #7, mul vl] // 16-byte Folded Reload83; CHECK-NEXT: ldr z16, [sp, #8, mul vl] // 16-byte Folded Reload84; CHECK-NEXT: ldr p8, [sp, #7, mul vl] // 2-byte Reload85; CHECK-NEXT: addvl sp, sp, #986; CHECK-NEXT: ldr x29, [sp], #16 // 8-byte Folded Reload87; CHECK-NEXT: ret88entry:89 %0 = tail call target("aarch64.svcount") @llvm.aarch64.sve.ptrue.c8()90 %1 = tail call { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } @llvm.aarch64.sve.ld1.pn.x4.nxv2i64(target("aarch64.svcount") %0, ptr %ptr)91 %2 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %1, 092 %3 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %1, 193 %4 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %1, 294 %5 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %1, 395 %arrayidx2 = getelementptr inbounds i8, ptr %ptr, i64 %stride96 %6 = tail call { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } @llvm.aarch64.sve.ld1.pn.x4.nxv2i64(target("aarch64.svcount") %0, ptr %arrayidx2)97 %7 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %6, 098 %8 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %6, 199 %9 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %6, 2100 %10 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %6, 3101 %mul3 = shl i64 %stride, 1102 %arrayidx4 = getelementptr inbounds i8, ptr %ptr, i64 %mul3103 %11 = tail call { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } @llvm.aarch64.sve.ld1.pn.x4.nxv2i64(target("aarch64.svcount") %0, ptr %arrayidx4)104 %12 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %11, 0105 %13 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %11, 1106 %14 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %11, 2107 %15 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %11, 3108 %mul5 = mul i64 %stride, 3109 %arrayidx6 = getelementptr inbounds i8, ptr %ptr, i64 %mul5110 %16 = tail call { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } @llvm.aarch64.sve.ld1.pn.x4.nxv2i64(target("aarch64.svcount") %0, ptr %arrayidx6)111 %17 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %16, 0112 %18 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %16, 1113 %19 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %16, 2114 %20 = extractvalue { <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64> } %16, 3115 %res1 = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x4.nxv2i64(<vscale x 2 x i64> %2, <vscale x 2 x i64> %7, <vscale x 2 x i64> %12, <vscale x 2 x i64> %17)116 %res2 = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x4.nxv2i64(<vscale x 2 x i64> %3, <vscale x 2 x i64> %8, <vscale x 2 x i64> %13, <vscale x 2 x i64> %18)117 %res3 = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x4.nxv2i64(<vscale x 2 x i64> %4, <vscale x 2 x i64> %9, <vscale x 2 x i64> %14, <vscale x 2 x i64> %19)118 %res4 = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x4.nxv2i64(<vscale x 2 x i64> %5, <vscale x 2 x i64> %10, <vscale x 2 x i64> %15, <vscale x 2 x i64> %20)119 %ins1 = insertvalue { <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16> } poison, <vscale x 8 x i16> %res1, 0120 %ins2 = insertvalue { <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16> } %ins1, <vscale x 8 x i16> %res2, 1121 %ins3 = insertvalue { <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16> } %ins2, <vscale x 8 x i16> %res3, 2122 %ins4 = insertvalue { <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16> } %ins3, <vscale x 8 x i16> %res4, 3123 ret { <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16>, <vscale x 8 x i16> } %ins4124}125 126;127; UQCVT128;129 130; x2131define <vscale x 8 x i16> @multi_vector_qcvt_x2_u16_u32(<vscale x 4 x i32> %unused, <vscale x 4 x i32> %zn0, <vscale x 4 x i32> %zn1) {132; CHECK-LABEL: multi_vector_qcvt_x2_u16_u32:133; CHECK: // %bb.0:134; CHECK-NEXT: mov z3.d, z2.d135; CHECK-NEXT: mov z2.d, z1.d136; CHECK-NEXT: uqcvt z0.h, { z2.s, z3.s }137; CHECK-NEXT: ret138 %res = call <vscale x 8 x i16> @llvm.aarch64.sve.uqcvt.x2.nxv4i32(<vscale x 4 x i32> %zn0, <vscale x 4 x i32> %zn1)139 ret<vscale x 8 x i16> %res140}141 142; x4143define <vscale x 16 x i8> @multi_vector_qcvt_x4_u8_u32(<vscale x 4 x i32> %unused, <vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2, <vscale x 4 x i32> %zn3, <vscale x 4 x i32> %zn4) {144; CHECK-LABEL: multi_vector_qcvt_x4_u8_u32:145; CHECK: // %bb.0:146; CHECK-NEXT: mov z7.d, z4.d147; CHECK-NEXT: mov z6.d, z3.d148; CHECK-NEXT: mov z5.d, z2.d149; CHECK-NEXT: mov z4.d, z1.d150; CHECK-NEXT: uqcvt z0.b, { z4.s - z7.s }151; CHECK-NEXT: ret152 %res = call <vscale x 16 x i8> @llvm.aarch64.sve.uqcvt.x4.nxv4i32(<vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2, <vscale x 4 x i32> %zn3, <vscale x 4 x i32> %zn4)153 ret <vscale x 16 x i8> %res154}155 156define <vscale x 8 x i16> @multi_vector_qcvt_x4_u16_u64(<vscale x 2 x i64> %unused, <vscale x 2 x i64> %zn1, <vscale x 2 x i64> %zn2, <vscale x 2 x i64> %zn3, <vscale x 2 x i64> %zn4) {157; CHECK-LABEL: multi_vector_qcvt_x4_u16_u64:158; CHECK: // %bb.0:159; CHECK-NEXT: mov z7.d, z4.d160; CHECK-NEXT: mov z6.d, z3.d161; CHECK-NEXT: mov z5.d, z2.d162; CHECK-NEXT: mov z4.d, z1.d163; CHECK-NEXT: uqcvt z0.h, { z4.d - z7.d }164; CHECK-NEXT: ret165 %res = call <vscale x 8 x i16> @llvm.aarch64.sve.uqcvt.x4.nxv2i64(<vscale x 2 x i64> %zn1, <vscale x 2 x i64> %zn2, <vscale x 2 x i64> %zn3, <vscale x 2 x i64> %zn4)166 ret <vscale x 8 x i16> %res167}168 169;170; SQCVTU171;172 173; x2174define <vscale x 8 x i16 > @multi_vector_qcvt_x2_s16_u32(<vscale x 4 x i32> %unused, <vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2) {175; CHECK-LABEL: multi_vector_qcvt_x2_s16_u32:176; CHECK: // %bb.0:177; CHECK-NEXT: mov z3.d, z2.d178; CHECK-NEXT: mov z2.d, z1.d179; CHECK-NEXT: sqcvtu z0.h, { z2.s, z3.s }180; CHECK-NEXT: ret181 %res = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvtu.x2.nxv4i32(<vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2)182 ret <vscale x 8 x i16> %res183}184 185; x4186define <vscale x 16 x i8> @multi_vector_qcvt_x4_u8_s32(<vscale x 4 x i32> %unused, <vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2, <vscale x 4 x i32> %zn3, <vscale x 4 x i32> %zn4) {187; CHECK-LABEL: multi_vector_qcvt_x4_u8_s32:188; CHECK: // %bb.0:189; CHECK-NEXT: mov z7.d, z4.d190; CHECK-NEXT: mov z6.d, z3.d191; CHECK-NEXT: mov z5.d, z2.d192; CHECK-NEXT: mov z4.d, z1.d193; CHECK-NEXT: sqcvtu z0.b, { z4.s - z7.s }194; CHECK-NEXT: ret195 %res = call <vscale x 16 x i8> @llvm.aarch64.sve.sqcvtu.x4.nxv4i32(<vscale x 4 x i32> %zn1, <vscale x 4 x i32> %zn2, <vscale x 4 x i32> %zn3, <vscale x 4 x i32> %zn4)196 ret <vscale x 16 x i8> %res197}198 199define <vscale x 8 x i16> @multi_vector_qcvt_x4_u16_s64(<vscale x 2 x i64> %unused, <vscale x 2 x i64> %zn1, <vscale x 2 x i64> %zn2, <vscale x 2 x i64> %zn3, <vscale x 2 x i64> %zn4) {200; CHECK-LABEL: multi_vector_qcvt_x4_u16_s64:201; CHECK: // %bb.0:202; CHECK-NEXT: mov z7.d, z4.d203; CHECK-NEXT: mov z6.d, z3.d204; CHECK-NEXT: mov z5.d, z2.d205; CHECK-NEXT: mov z4.d, z1.d206; CHECK-NEXT: sqcvtu z0.h, { z4.d - z7.d }207; CHECK-NEXT: ret208 %res = call <vscale x 8 x i16> @llvm.aarch64.sve.sqcvtu.x4.nxv2i64(<vscale x 2 x i64> %zn1, <vscale x 2 x i64> %zn2, <vscale x 2 x i64> %zn3, <vscale x 2 x i64> %zn4)209 ret <vscale x 8 x i16> %res210}211 212declare <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x2.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>)213declare <vscale x 8 x i16> @llvm.aarch64.sve.uqcvt.x2.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>)214declare <vscale x 8 x i16> @llvm.aarch64.sve.sqcvtu.x2.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>)215declare <vscale x 16 x i8> @llvm.aarch64.sve.sqcvt.x4.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>)216declare <vscale x 8 x i16> @llvm.aarch64.sve.sqcvt.x4.nxv2i64(<vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>)217declare <vscale x 16 x i8> @llvm.aarch64.sve.uqcvt.x4.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>)218declare <vscale x 8 x i16> @llvm.aarch64.sve.uqcvt.x4.nxv2i64(<vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>)219declare <vscale x 16 x i8> @llvm.aarch64.sve.sqcvtu.x4.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>)220declare <vscale x 8 x i16> @llvm.aarch64.sve.sqcvtu.x4.nxv2i64(<vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>, <vscale x 2 x i64>)221