brintos

brintos / llvm-project-archived public Read only

0
0
Text · 36.2 KiB · 683b927 Raw
632 lines · plain
1; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 42; RUN: opt -passes=slp-vectorizer -slp-vectorize-non-power-of-2 -mtriple=arm64-apple-ios -S %s | FileCheck --check-prefixes=CHECK,NON-POW2 %s3; RUN: opt -passes=slp-vectorizer -slp-vectorize-non-power-of-2=false -mtriple=arm64-apple-ios -S %s | FileCheck --check-prefixes=CHECK,POW2-ONLY %s4 5%struct.zot = type { i32, i32, i32 }6 7define i1 @reorder_results(ptr %arg, i1 %arg1, ptr %arg2, i64 %arg3, ptr %arg4) {8; CHECK-LABEL: define i1 @reorder_results(9; CHECK-SAME: ptr [[ARG:%.*]], i1 [[ARG1:%.*]], ptr [[ARG2:%.*]], i64 [[ARG3:%.*]], ptr [[ARG4:%.*]]) {10; CHECK-NEXT:  bb:11; CHECK-NEXT:    [[LOAD:%.*]] = load ptr, ptr [[ARG4]], align 812; CHECK-NEXT:    [[LOAD4:%.*]] = load i32, ptr [[LOAD]], align 413; CHECK-NEXT:    [[GETELEMENTPTR:%.*]] = getelementptr i8, ptr [[LOAD]], i64 414; CHECK-NEXT:    [[LOAD5:%.*]] = load i32, ptr [[GETELEMENTPTR]], align 415; CHECK-NEXT:    [[GETELEMENTPTR6:%.*]] = getelementptr i8, ptr [[LOAD]], i64 816; CHECK-NEXT:    [[LOAD7:%.*]] = load i32, ptr [[GETELEMENTPTR6]], align 417; CHECK-NEXT:    br i1 [[ARG1]], label [[BB12:%.*]], label [[BB9:%.*]]18; CHECK:       bb8:19; CHECK-NEXT:    ret i1 false20; CHECK:       bb9:21; CHECK-NEXT:    [[FREEZE:%.*]] = freeze ptr [[ARG]]22; CHECK-NEXT:    store i32 [[LOAD4]], ptr [[FREEZE]], align 423; CHECK-NEXT:    [[GETELEMENTPTR10:%.*]] = getelementptr i8, ptr [[FREEZE]], i64 424; CHECK-NEXT:    store i32 [[LOAD7]], ptr [[GETELEMENTPTR10]], align 425; CHECK-NEXT:    [[GETELEMENTPTR11:%.*]] = getelementptr i8, ptr [[FREEZE]], i64 826; CHECK-NEXT:    store i32 [[LOAD5]], ptr [[GETELEMENTPTR11]], align 427; CHECK-NEXT:    br label [[BB8:%.*]]28; CHECK:       bb12:29; CHECK-NEXT:    [[GETELEMENTPTR13:%.*]] = getelementptr [[STRUCT_ZOT:%.*]], ptr [[ARG2]], i64 [[ARG3]]30; CHECK-NEXT:    store i32 [[LOAD4]], ptr [[GETELEMENTPTR13]], align 431; CHECK-NEXT:    [[GETELEMENTPTR14:%.*]] = getelementptr i8, ptr [[GETELEMENTPTR13]], i64 432; CHECK-NEXT:    store i32 [[LOAD7]], ptr [[GETELEMENTPTR14]], align 433; CHECK-NEXT:    [[GETELEMENTPTR15:%.*]] = getelementptr i8, ptr [[GETELEMENTPTR13]], i64 834; CHECK-NEXT:    store i32 [[LOAD5]], ptr [[GETELEMENTPTR15]], align 435; CHECK-NEXT:    br label [[BB8]]36;37bb:38  %load = load ptr, ptr %arg4, align 839  %load4 = load i32, ptr %load, align 440  %getelementptr = getelementptr i8, ptr %load, i64 441  %load5 = load i32, ptr %getelementptr, align 442  %getelementptr6 = getelementptr i8, ptr %load, i64 843  %load7 = load i32, ptr %getelementptr6, align 444  br i1 %arg1, label %bb12, label %bb945 46bb8:                                              ; preds = %bb12, %bb947  ret i1 false48 49bb9:                                              ; preds = %bb50  %freeze = freeze ptr %arg51  store i32 %load4, ptr %freeze, align 452  %getelementptr10 = getelementptr i8, ptr %freeze, i64 453  store i32 %load7, ptr %getelementptr10, align 454  %getelementptr11 = getelementptr i8, ptr %freeze, i64 855  store i32 %load5, ptr %getelementptr11, align 456  br label %bb857 58bb12:                                             ; preds = %bb59  %getelementptr13 = getelementptr %struct.zot, ptr %arg2, i64 %arg360  store i32 %load4, ptr %getelementptr13, align 461  %getelementptr14 = getelementptr i8, ptr %getelementptr13, i64 462  store i32 %load7, ptr %getelementptr14, align 463  %getelementptr15 = getelementptr i8, ptr %getelementptr13, i64 864  store i32 %load5, ptr %getelementptr15, align 465  br label %bb866}67 68define void @extract_mask(ptr %object, double %conv503, double %conv520) {69; CHECK-LABEL: define void @extract_mask(70; CHECK-SAME: ptr [[OBJECT:%.*]], double [[CONV503:%.*]], double [[CONV520:%.*]]) {71; CHECK-NEXT:  entry:72; CHECK-NEXT:    [[TMP0:%.*]] = load ptr, ptr [[OBJECT]], align 873; CHECK-NEXT:    [[BBOX483:%.*]] = getelementptr float, ptr [[TMP0]]74; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x float>, ptr [[BBOX483]], align 875; CHECK-NEXT:    [[TMP2:%.*]] = fpext <2 x float> [[TMP1]] to <2 x double>76; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> <i32 1, i32 0>77; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x double> [[TMP3]], double [[CONV503]], i32 078; CHECK-NEXT:    [[TMP5:%.*]] = fcmp ogt <2 x double> [[TMP4]], <double 0.000000e+00, double -2.000000e+10>79; CHECK-NEXT:    [[TMP6:%.*]] = select <2 x i1> [[TMP5]], <2 x double> [[TMP3]], <2 x double> <double 0.000000e+00, double -2.000000e+10>80; CHECK-NEXT:    [[TMP7:%.*]] = fsub <2 x double> zeroinitializer, [[TMP6]]81; CHECK-NEXT:    [[TMP8:%.*]] = fptrunc <2 x double> [[TMP7]] to <2 x float>82; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x float> [[TMP8]], i32 083; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x float> [[TMP8]], i32 184; CHECK-NEXT:    [[MUL646:%.*]] = fmul float [[TMP9]], [[TMP10]]85; CHECK-NEXT:    [[CMP663:%.*]] = fcmp olt float [[MUL646]], 0.000000e+0086; CHECK-NEXT:    br i1 [[CMP663]], label [[IF_THEN665:%.*]], label [[IF_END668:%.*]]87; CHECK:       if.then665:88; CHECK-NEXT:    [[ARRAYIDX656:%.*]] = getelementptr float, ptr [[OBJECT]], i64 1089; CHECK-NEXT:    [[BBOX651:%.*]] = getelementptr float, ptr [[OBJECT]]90; CHECK-NEXT:    [[CONV621:%.*]] = fptrunc double [[CONV520]] to float91; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <2 x double> [[TMP6]], <2 x double> poison, <2 x i32> <i32 poison, i32 0>92; CHECK-NEXT:    [[TMP12:%.*]] = insertelement <2 x double> [[TMP11]], double [[CONV503]], i32 093; CHECK-NEXT:    [[TMP13:%.*]] = fptrunc <2 x double> [[TMP12]] to <2 x float>94; CHECK-NEXT:    store <2 x float> [[TMP13]], ptr [[BBOX651]], align 895; CHECK-NEXT:    [[BBOX_SROA_8_0_BBOX666_SROA_IDX:%.*]] = getelementptr float, ptr [[OBJECT]], i64 296; CHECK-NEXT:    store float [[CONV621]], ptr [[BBOX_SROA_8_0_BBOX666_SROA_IDX]], align 897; CHECK-NEXT:    store <2 x float> [[TMP8]], ptr [[ARRAYIDX656]], align 898; CHECK-NEXT:    br label [[IF_END668]]99; CHECK:       if.end668:100; CHECK-NEXT:    ret void101;102entry:103  %0 = load ptr, ptr %object, align 8104  %bbox483 = getelementptr float, ptr %0105  %1 = load float, ptr %bbox483, align 8106  %conv486 = fpext float %1 to double107  %cmp487 = fcmp ogt double %conv486, -2.000000e+10108  %conv486.2 = select i1 %cmp487, double %conv486, double -2.000000e+10109  %arrayidx502 = getelementptr float, ptr %0, i64 1110  %2 = load float, ptr %arrayidx502, align 4111  %conv5033 = fpext float %2 to double112  %cmp504 = fcmp ogt double %conv503, 0.000000e+00113  %cond514 = select i1 %cmp504, double %conv5033, double 0.000000e+00114  %sub626 = fsub double 0.000000e+00, %conv486.2115  %conv627 = fptrunc double %sub626 to float116  %sub632 = fsub double 0.000000e+00, %cond514117  %conv633 = fptrunc double %sub632 to float118  %mul646 = fmul float %conv633, %conv627119  %cmp663 = fcmp olt float %mul646, 0.000000e+00120  br i1 %cmp663, label %if.then665, label %if.end668121 122if.then665:                                       ; preds = %entry123  %arrayidx656 = getelementptr float, ptr %object, i64 10124  %lengths652 = getelementptr float, ptr %object, i64 11125  %bbox651 = getelementptr float, ptr %object126  %conv621 = fptrunc double %conv520 to float127  %conv617 = fptrunc double %cond514 to float128  %conv613 = fptrunc double %conv503 to float129  store float %conv613, ptr %bbox651, align 8130  %bbox.sroa.6.0.bbox666.sroa_idx = getelementptr float, ptr %object, i64 1131  store float %conv617, ptr %bbox.sroa.6.0.bbox666.sroa_idx, align 4132  %bbox.sroa.8.0.bbox666.sroa_idx = getelementptr float, ptr %object, i64 2133  store float %conv621, ptr %bbox.sroa.8.0.bbox666.sroa_idx, align 8134  store float %conv627, ptr %lengths652, align 4135  store float %conv633, ptr %arrayidx656, align 8136  br label %if.end668137 138if.end668:                                        ; preds = %if.then665, %entry139  ret void140}141 142define void @gather_2(ptr %mat1, float %0, float %1) {143; NON-POW2-LABEL: define void @gather_2(144; NON-POW2-SAME: ptr [[MAT1:%.*]], float [[TMP0:%.*]], float [[TMP1:%.*]]) {145; NON-POW2-NEXT:  entry:146; NON-POW2-NEXT:    [[TMP2:%.*]] = insertelement <3 x float> poison, float [[TMP0]], i32 0147; NON-POW2-NEXT:    [[TMP3:%.*]] = shufflevector <3 x float> [[TMP2]], <3 x float> poison, <3 x i32> zeroinitializer148; NON-POW2-NEXT:    [[TMP4:%.*]] = insertelement <3 x float> <float 0.000000e+00, float poison, float poison>, float [[TMP1]], i32 1149; NON-POW2-NEXT:    [[TMP5:%.*]] = shufflevector <3 x float> [[TMP4]], <3 x float> poison, <3 x i32> <i32 0, i32 1, i32 1>150; NON-POW2-NEXT:    [[TMP6:%.*]] = call <3 x float> @llvm.fmuladd.v3f32(<3 x float> [[TMP3]], <3 x float> [[TMP5]], <3 x float> zeroinitializer)151; NON-POW2-NEXT:    [[ARRAYIDX163:%.*]] = getelementptr [4 x [4 x float]], ptr [[MAT1]], i64 0, i64 1152; NON-POW2-NEXT:    [[TMP7:%.*]] = fmul <3 x float> [[TMP6]], zeroinitializer153; NON-POW2-NEXT:    store <3 x float> [[TMP7]], ptr [[ARRAYIDX163]], align 4154; NON-POW2-NEXT:    ret void155;156; POW2-ONLY-LABEL: define void @gather_2(157; POW2-ONLY-SAME: ptr [[MAT1:%.*]], float [[TMP0:%.*]], float [[TMP1:%.*]]) {158; POW2-ONLY-NEXT:  entry:159; POW2-ONLY-NEXT:    [[TMP2:%.*]] = insertelement <2 x float> poison, float [[TMP0]], i32 0160; POW2-ONLY-NEXT:    [[TMP3:%.*]] = shufflevector <2 x float> [[TMP2]], <2 x float> poison, <2 x i32> zeroinitializer161; POW2-ONLY-NEXT:    [[TMP4:%.*]] = insertelement <2 x float> <float 0.000000e+00, float poison>, float [[TMP1]], i32 1162; POW2-ONLY-NEXT:    [[TMP5:%.*]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP3]], <2 x float> [[TMP4]], <2 x float> zeroinitializer)163; POW2-ONLY-NEXT:    [[TMP6:%.*]] = call float @llvm.fmuladd.f32(float [[TMP0]], float [[TMP1]], float 0.000000e+00)164; POW2-ONLY-NEXT:    [[TMP7:%.*]] = fmul float [[TMP6]], 0.000000e+00165; POW2-ONLY-NEXT:    [[ARRAYIDX163:%.*]] = getelementptr [4 x [4 x float]], ptr [[MAT1]], i64 0, i64 1166; POW2-ONLY-NEXT:    [[ARRAYIDX5_I_I_I280:%.*]] = getelementptr [4 x [4 x float]], ptr [[MAT1]], i64 0, i64 1, i64 2167; POW2-ONLY-NEXT:    [[TMP8:%.*]] = fmul <2 x float> [[TMP5]], zeroinitializer168; POW2-ONLY-NEXT:    store <2 x float> [[TMP8]], ptr [[ARRAYIDX163]], align 4169; POW2-ONLY-NEXT:    store float [[TMP7]], ptr [[ARRAYIDX5_I_I_I280]], align 4170; POW2-ONLY-NEXT:    ret void171;172entry:173  %2 = call float @llvm.fmuladd.f32(float %0, float 0.000000e+00, float 0.000000e+00)174  %3 = call float @llvm.fmuladd.f32(float %1, float %0, float 0.000000e+00)175  %4 = call float @llvm.fmuladd.f32(float %0, float %1, float 0.000000e+00)176  %5 = fmul float %2, 0.000000e+00177  %6 = fmul float %3, 0.000000e+00178  %7 = fmul float %4, 0.000000e+00179  %arrayidx163 = getelementptr [4 x [4 x float]], ptr %mat1, i64 0, i64 1180  %arrayidx2.i.i.i278 = getelementptr [4 x [4 x float]], ptr %mat1, i64 0, i64 1, i64 1181  %arrayidx5.i.i.i280 = getelementptr [4 x [4 x float]], ptr %mat1, i64 0, i64 1, i64 2182  store float %5, ptr %arrayidx163, align 4183  store float %6, ptr %arrayidx2.i.i.i278, align 4184  store float %7, ptr %arrayidx5.i.i.i280, align 4185  ret void186}187 188define i32 @reorder_indices_1(float %0) {189; NON-POW2-LABEL: define i32 @reorder_indices_1(190; NON-POW2-SAME: float [[TMP0:%.*]]) {191; NON-POW2-NEXT:  entry:192; NON-POW2-NEXT:    [[NOR1:%.*]] = alloca [0 x [3 x float]], i32 0, align 4193; NON-POW2-NEXT:    [[TMP1:%.*]] = load <3 x float>, ptr [[NOR1]], align 4194; NON-POW2-NEXT:    [[TMP2:%.*]] = shufflevector <3 x float> [[TMP1]], <3 x float> poison, <3 x i32> <i32 1, i32 2, i32 0>195; NON-POW2-NEXT:    [[TMP3:%.*]] = fneg <3 x float> [[TMP2]]196; NON-POW2-NEXT:    [[TMP4:%.*]] = insertelement <3 x float> poison, float [[TMP0]], i32 0197; NON-POW2-NEXT:    [[TMP5:%.*]] = shufflevector <3 x float> [[TMP4]], <3 x float> poison, <3 x i32> zeroinitializer198; NON-POW2-NEXT:    [[TMP6:%.*]] = fmul <3 x float> [[TMP3]], [[TMP5]]199; NON-POW2-NEXT:    [[TMP7:%.*]] = call <3 x float> @llvm.fmuladd.v3f32(<3 x float> [[TMP1]], <3 x float> zeroinitializer, <3 x float> [[TMP6]])200; NON-POW2-NEXT:    [[TMP8:%.*]] = call <3 x float> @llvm.fmuladd.v3f32(<3 x float> [[TMP5]], <3 x float> [[TMP7]], <3 x float> zeroinitializer)201; NON-POW2-NEXT:    [[TMP9:%.*]] = fmul <3 x float> [[TMP8]], zeroinitializer202; NON-POW2-NEXT:    store <3 x float> [[TMP9]], ptr [[NOR1]], align 4203; NON-POW2-NEXT:    ret i32 0204;205; POW2-ONLY-LABEL: define i32 @reorder_indices_1(206; POW2-ONLY-SAME: float [[TMP0:%.*]]) {207; POW2-ONLY-NEXT:  entry:208; POW2-ONLY-NEXT:    [[NOR1:%.*]] = alloca [0 x [3 x float]], i32 0, align 4209; POW2-ONLY-NEXT:    [[ARRAYIDX_I:%.*]] = getelementptr float, ptr [[NOR1]], i64 1210; POW2-ONLY-NEXT:    [[ARRAYIDX2_I265:%.*]] = getelementptr float, ptr [[NOR1]], i64 2211; POW2-ONLY-NEXT:    [[TMP1:%.*]] = load float, ptr [[ARRAYIDX2_I265]], align 4212; POW2-ONLY-NEXT:    [[TMP7:%.*]] = load <2 x float>, ptr [[ARRAYIDX_I]], align 4213; POW2-ONLY-NEXT:    [[TMP2:%.*]] = load <2 x float>, ptr [[NOR1]], align 4214; POW2-ONLY-NEXT:    [[TMP3:%.*]] = extractelement <2 x float> [[TMP2]], i32 0215; POW2-ONLY-NEXT:    [[TMP4:%.*]] = fneg float [[TMP3]]216; POW2-ONLY-NEXT:    [[NEG11_I:%.*]] = fmul float [[TMP4]], [[TMP0]]217; POW2-ONLY-NEXT:    [[TMP5:%.*]] = call float @llvm.fmuladd.f32(float [[TMP1]], float 0.000000e+00, float [[NEG11_I]])218; POW2-ONLY-NEXT:    [[TMP8:%.*]] = fneg <2 x float> [[TMP7]]219; POW2-ONLY-NEXT:    [[TMP9:%.*]] = insertelement <2 x float> poison, float [[TMP0]], i32 0220; POW2-ONLY-NEXT:    [[TMP10:%.*]] = shufflevector <2 x float> [[TMP9]], <2 x float> poison, <2 x i32> zeroinitializer221; POW2-ONLY-NEXT:    [[TMP11:%.*]] = fmul <2 x float> [[TMP8]], [[TMP10]]222; POW2-ONLY-NEXT:    [[TMP12:%.*]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP2]], <2 x float> zeroinitializer, <2 x float> [[TMP11]])223; POW2-ONLY-NEXT:    [[TMP13:%.*]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP10]], <2 x float> [[TMP12]], <2 x float> zeroinitializer)224; POW2-ONLY-NEXT:    [[TMP14:%.*]] = call float @llvm.fmuladd.f32(float [[TMP0]], float [[TMP5]], float 0.000000e+00)225; POW2-ONLY-NEXT:    [[TMP15:%.*]] = fmul <2 x float> [[TMP13]], zeroinitializer226; POW2-ONLY-NEXT:    [[MUL6_I_I_I:%.*]] = fmul float [[TMP14]], 0.000000e+00227; POW2-ONLY-NEXT:    store <2 x float> [[TMP15]], ptr [[NOR1]], align 4228; POW2-ONLY-NEXT:    store float [[MUL6_I_I_I]], ptr [[ARRAYIDX2_I265]], align 4229; POW2-ONLY-NEXT:    ret i32 0230;231entry:232  %nor1 = alloca [0 x [3 x float]], i32 0, align 4233  %arrayidx.i = getelementptr float, ptr %nor1, i64 1234  %1 = load float, ptr %arrayidx.i, align 4235  %arrayidx2.i265 = getelementptr float, ptr %nor1, i64 2236  %2 = load float, ptr %arrayidx2.i265, align 4237  %3 = fneg float %2238  %neg.i267 = fmul float %3, %0239  %4 = call float @llvm.fmuladd.f32(float %1, float 0.000000e+00, float %neg.i267)240  %5 = load float, ptr %nor1, align 4241  %6 = fneg float %5242  %neg11.i = fmul float %6, %0243  %7 = call float @llvm.fmuladd.f32(float %2, float 0.000000e+00, float %neg11.i)244  %8 = fneg float %1245  %neg18.i = fmul float %8, %0246  %9 = call float @llvm.fmuladd.f32(float %5, float 0.000000e+00, float %neg18.i)247  %10 = call float @llvm.fmuladd.f32(float %0, float %9, float 0.000000e+00)248  %11 = call float @llvm.fmuladd.f32(float %0, float %4, float 0.000000e+00)249  %12 = call float @llvm.fmuladd.f32(float %0, float %7, float 0.000000e+00)250  %mul.i.i.i = fmul float %10, 0.000000e+00251  %mul3.i.i.i = fmul float %11, 0.000000e+00252  %mul6.i.i.i = fmul float %12, 0.000000e+00253  store float %mul.i.i.i, ptr %nor1, align 4254  store float %mul3.i.i.i, ptr %arrayidx.i, align 4255  store float %mul6.i.i.i, ptr %arrayidx2.i265, align 4256  ret i32 0257}258 259define void @reorder_indices_2(ptr %spoint) {260; NON-POW2-LABEL: define void @reorder_indices_2(261; NON-POW2-SAME: ptr [[SPOINT:%.*]]) {262; NON-POW2-NEXT:  entry:263; NON-POW2-NEXT:    [[DSCO:%.*]] = getelementptr float, ptr [[SPOINT]], i64 0264; NON-POW2-NEXT:    [[TMP0:%.*]] = call <3 x float> @llvm.fmuladd.v3f32(<3 x float> zeroinitializer, <3 x float> zeroinitializer, <3 x float> zeroinitializer)265; NON-POW2-NEXT:    [[TMP1:%.*]] = fmul <3 x float> [[TMP0]], zeroinitializer266; NON-POW2-NEXT:    [[TMP2:%.*]] = shufflevector <3 x float> [[TMP1]], <3 x float> poison, <3 x i32> <i32 1, i32 2, i32 0>267; NON-POW2-NEXT:    store <3 x float> [[TMP2]], ptr [[DSCO]], align 4268; NON-POW2-NEXT:    ret void269;270; POW2-ONLY-LABEL: define void @reorder_indices_2(271; POW2-ONLY-SAME: ptr [[SPOINT:%.*]]) {272; POW2-ONLY-NEXT:  entry:273; POW2-ONLY-NEXT:    [[TMP0:%.*]] = extractelement <3 x float> zeroinitializer, i64 0274; POW2-ONLY-NEXT:    [[TMP1:%.*]] = tail call float @llvm.fmuladd.f32(float [[TMP0]], float 0.000000e+00, float 0.000000e+00)275; POW2-ONLY-NEXT:    [[MUL4_I461:%.*]] = fmul float [[TMP1]], 0.000000e+00276; POW2-ONLY-NEXT:    [[DSCO:%.*]] = getelementptr float, ptr [[SPOINT]], i64 0277; POW2-ONLY-NEXT:    [[TMP2:%.*]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> zeroinitializer, <2 x float> zeroinitializer, <2 x float> zeroinitializer)278; POW2-ONLY-NEXT:    [[TMP3:%.*]] = fmul <2 x float> [[TMP2]], zeroinitializer279; POW2-ONLY-NEXT:    store <2 x float> [[TMP3]], ptr [[DSCO]], align 4280; POW2-ONLY-NEXT:    [[ARRAYIDX5_I476:%.*]] = getelementptr float, ptr [[SPOINT]], i64 2281; POW2-ONLY-NEXT:    store float [[MUL4_I461]], ptr [[ARRAYIDX5_I476]], align 4282; POW2-ONLY-NEXT:    ret void283;284entry:285  %0 = extractelement <3 x float> zeroinitializer, i64 1286  %1 = extractelement <3 x float> zeroinitializer, i64 2287  %2 = extractelement <3 x float> zeroinitializer, i64 0288  %3 = tail call float @llvm.fmuladd.f32(float %0, float 0.000000e+00, float 0.000000e+00)289  %4 = tail call float @llvm.fmuladd.f32(float %1, float 0.000000e+00, float 0.000000e+00)290  %5 = tail call float @llvm.fmuladd.f32(float %2, float 0.000000e+00, float 0.000000e+00)291  %mul.i457 = fmul float %3, 0.000000e+00292  %mul2.i459 = fmul float %4, 0.000000e+00293  %mul4.i461 = fmul float %5, 0.000000e+00294  %dsco = getelementptr float, ptr %spoint, i64 0295  store float %mul.i457, ptr %dsco, align 4296  %arrayidx3.i474 = getelementptr float, ptr %spoint, i64 1297  store float %mul2.i459, ptr %arrayidx3.i474, align 4298  %arrayidx5.i476 = getelementptr float, ptr %spoint, i64 2299  store float %mul4.i461, ptr %arrayidx5.i476, align 4300  ret void301}302 303define void @reorder_indices_2x_load(ptr %png_ptr, ptr %info_ptr) {304; CHECK-LABEL: define void @reorder_indices_2x_load(305; CHECK-SAME: ptr [[PNG_PTR:%.*]], ptr [[INFO_PTR:%.*]]) {306; CHECK-NEXT:  entry:307; CHECK-NEXT:    [[BIT_DEPTH:%.*]] = getelementptr i8, ptr [[INFO_PTR]], i64 0308; CHECK-NEXT:    [[TMP0:%.*]] = load i8, ptr [[BIT_DEPTH]], align 4309; CHECK-NEXT:    [[COLOR_TYPE:%.*]] = getelementptr i8, ptr [[INFO_PTR]], i64 1310; CHECK-NEXT:    [[TMP1:%.*]] = load i8, ptr [[COLOR_TYPE]], align 1311; CHECK-NEXT:    [[BIT_DEPTH37_I:%.*]] = getelementptr i8, ptr [[PNG_PTR]], i64 11312; CHECK-NEXT:    store i8 [[TMP0]], ptr [[BIT_DEPTH37_I]], align 1313; CHECK-NEXT:    [[COLOR_TYPE39_I:%.*]] = getelementptr i8, ptr [[PNG_PTR]], i64 10314; CHECK-NEXT:    store i8 [[TMP1]], ptr [[COLOR_TYPE39_I]], align 2315; CHECK-NEXT:    [[USR_BIT_DEPTH_I:%.*]] = getelementptr i8, ptr [[PNG_PTR]], i64 12316; CHECK-NEXT:    store i8 [[TMP0]], ptr [[USR_BIT_DEPTH_I]], align 8317; CHECK-NEXT:    ret void318;319entry:320  %bit_depth = getelementptr i8, ptr %info_ptr, i64 0321  %0 = load i8, ptr %bit_depth, align 4322  %color_type = getelementptr i8, ptr %info_ptr, i64 1323  %1 = load i8, ptr %color_type, align 1324  %bit_depth37.i = getelementptr i8, ptr %png_ptr, i64 11325  store i8 %0, ptr %bit_depth37.i, align 1326  %color_type39.i = getelementptr i8, ptr %png_ptr, i64 10327  store i8 %1, ptr %color_type39.i, align 2328  %usr_bit_depth.i = getelementptr i8, ptr %png_ptr, i64 12329  store i8 %0, ptr %usr_bit_depth.i, align 8330  ret void331}332 333define void @reuse_shuffle_indidces_1(ptr %col, float %0, float %1) {334; NON-POW2-LABEL: define void @reuse_shuffle_indidces_1(335; NON-POW2-SAME: ptr [[COL:%.*]], float [[TMP0:%.*]], float [[TMP1:%.*]]) {336; NON-POW2-NEXT:  entry:337; NON-POW2-NEXT:    [[TMP2:%.*]] = insertelement <3 x float> poison, float [[TMP1]], i32 0338; NON-POW2-NEXT:    [[TMP3:%.*]] = insertelement <3 x float> [[TMP2]], float [[TMP0]], i32 1339; NON-POW2-NEXT:    [[TMP4:%.*]] = shufflevector <3 x float> [[TMP3]], <3 x float> poison, <3 x i32> <i32 0, i32 1, i32 1>340; NON-POW2-NEXT:    [[TMP5:%.*]] = fmul <3 x float> [[TMP4]], zeroinitializer341; NON-POW2-NEXT:    [[TMP6:%.*]] = fadd <3 x float> [[TMP5]], zeroinitializer342; NON-POW2-NEXT:    store <3 x float> [[TMP6]], ptr [[COL]], align 4343; NON-POW2-NEXT:    ret void344;345; POW2-ONLY-LABEL: define void @reuse_shuffle_indidces_1(346; POW2-ONLY-SAME: ptr [[COL:%.*]], float [[TMP0:%.*]], float [[TMP1:%.*]]) {347; POW2-ONLY-NEXT:  entry:348; POW2-ONLY-NEXT:    [[TMP2:%.*]] = insertelement <2 x float> poison, float [[TMP1]], i32 0349; POW2-ONLY-NEXT:    [[TMP3:%.*]] = insertelement <2 x float> [[TMP2]], float [[TMP0]], i32 1350; POW2-ONLY-NEXT:    [[TMP4:%.*]] = fmul <2 x float> [[TMP3]], zeroinitializer351; POW2-ONLY-NEXT:    [[TMP5:%.*]] = fadd <2 x float> [[TMP4]], zeroinitializer352; POW2-ONLY-NEXT:    store <2 x float> [[TMP5]], ptr [[COL]], align 4353; POW2-ONLY-NEXT:    [[ARRAYIDX33:%.*]] = getelementptr float, ptr [[COL]], i64 2354; POW2-ONLY-NEXT:    [[MUL38:%.*]] = fmul float [[TMP0]], 0.000000e+00355; POW2-ONLY-NEXT:    [[TMP6:%.*]] = fadd float [[MUL38]], 0.000000e+00356; POW2-ONLY-NEXT:    store float [[TMP6]], ptr [[ARRAYIDX33]], align 4357; POW2-ONLY-NEXT:    ret void358;359entry:360  %mul24 = fmul float %1, 0.000000e+00361  %2 = fadd float %mul24, 0.000000e+00362  store float %2, ptr %col, align 4363  %arrayidx26 = getelementptr float, ptr %col, i64 1364  %mul31 = fmul float %0, 0.000000e+00365  %3 = fadd float %mul31, 0.000000e+00366  store float %3, ptr %arrayidx26, align 4367  %arrayidx33 = getelementptr float, ptr %col, i64 2368  %mul38 = fmul float %0, 0.000000e+00369  %4 = fadd float %mul38, 0.000000e+00370  store float %4, ptr %arrayidx33, align 4371  ret void372}373 374define void @reuse_shuffle_indices_2(ptr %inertia, double %0) {375; CHECK-LABEL: define void @reuse_shuffle_indices_2(376; CHECK-SAME: ptr [[INERTIA:%.*]], double [[TMP0:%.*]]) {377; CHECK-NEXT:  entry:378; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> poison, double [[TMP0]], i32 0379; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer380; CHECK-NEXT:    [[TMP3:%.*]] = fptrunc <2 x double> [[TMP2]] to <2 x float>381; CHECK-NEXT:    [[TMP4:%.*]] = fmul <2 x float> [[TMP3]], zeroinitializer382; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 poison>383; CHECK-NEXT:    [[TMP6:%.*]] = fadd <4 x float> [[TMP5]], <float 0.000000e+00, float 0.000000e+00, float 0.000000e+00, float undef>384; CHECK-NEXT:    [[TMP7:%.*]] = fmul <4 x float> [[TMP6]], <float 0.000000e+00, float 0.000000e+00, float 0.000000e+00, float undef>385; CHECK-NEXT:    [[TMP8:%.*]] = fadd <4 x float> [[TMP7]], <float 0.000000e+00, float 0.000000e+00, float 0.000000e+00, float undef>386; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <4 x float> [[TMP8]], <4 x float> poison, <3 x i32> <i32 0, i32 1, i32 2>387; CHECK-NEXT:    store <3 x float> [[TMP9]], ptr [[INERTIA]], align 4388; CHECK-NEXT:    ret void389;390entry:391  %1 = insertelement <2 x double> poison, double %0, i32 0392  %2 = shufflevector <2 x double> %1, <2 x double> poison, <2 x i32> zeroinitializer393  %3 = fptrunc <2 x double> %2 to <2 x float>394  %4 = fmul <2 x float> %3, zeroinitializer395  %5 = shufflevector <2 x float> %4, <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 poison>396  %6 = fadd <4 x float> %5, <float 0.000000e+00, float 0.000000e+00, float 0.000000e+00, float undef>397  %7 = fmul <4 x float> %6, <float 0.000000e+00, float 0.000000e+00, float 0.000000e+00, float undef>398  %8 = fadd <4 x float> %7, <float 0.000000e+00, float 0.000000e+00, float 0.000000e+00, float undef>399  %9 = shufflevector <4 x float> %8, <4 x float> poison, <3 x i32> <i32 0, i32 1, i32 2>400  store <3 x float> %9, ptr %inertia, align 4401  ret void402}403 404define void @reuse_shuffle_indices_cost_crash_2(ptr %bezt, float %0) {405; NON-POW2-LABEL: define void @reuse_shuffle_indices_cost_crash_2(406; NON-POW2-SAME: ptr [[BEZT:%.*]], float [[TMP0:%.*]]) {407; NON-POW2-NEXT:  entry:408; NON-POW2-NEXT:    [[FNEG:%.*]] = fmul float [[TMP0]], 0.000000e+00409; NON-POW2-NEXT:    [[TMP1:%.*]] = insertelement <3 x float> poison, float [[FNEG]], i32 0410; NON-POW2-NEXT:    [[TMP2:%.*]] = shufflevector <3 x float> [[TMP1]], <3 x float> poison, <3 x i32> zeroinitializer411; NON-POW2-NEXT:    [[TMP3:%.*]] = insertelement <3 x float> <float poison, float poison, float 0.000000e+00>, float [[TMP0]], i32 0412; NON-POW2-NEXT:    [[TMP4:%.*]] = shufflevector <3 x float> [[TMP3]], <3 x float> poison, <3 x i32> <i32 0, i32 0, i32 2>413; NON-POW2-NEXT:    [[TMP5:%.*]] = call <3 x float> @llvm.fmuladd.v3f32(<3 x float> [[TMP2]], <3 x float> [[TMP4]], <3 x float> zeroinitializer)414; NON-POW2-NEXT:    store <3 x float> [[TMP5]], ptr [[BEZT]], align 4415; NON-POW2-NEXT:    ret void416;417; POW2-ONLY-LABEL: define void @reuse_shuffle_indices_cost_crash_2(418; POW2-ONLY-SAME: ptr [[BEZT:%.*]], float [[TMP0:%.*]]) {419; POW2-ONLY-NEXT:  entry:420; POW2-ONLY-NEXT:    [[FNEG:%.*]] = fmul float [[TMP0]], 0.000000e+00421; POW2-ONLY-NEXT:    [[TMP1:%.*]] = insertelement <2 x float> poison, float [[TMP0]], i32 0422; POW2-ONLY-NEXT:    [[TMP2:%.*]] = shufflevector <2 x float> [[TMP1]], <2 x float> poison, <2 x i32> zeroinitializer423; POW2-ONLY-NEXT:    [[TMP3:%.*]] = insertelement <2 x float> poison, float [[FNEG]], i32 0424; POW2-ONLY-NEXT:    [[TMP4:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <2 x i32> zeroinitializer425; POW2-ONLY-NEXT:    [[TMP5:%.*]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP2]], <2 x float> [[TMP4]], <2 x float> zeroinitializer)426; POW2-ONLY-NEXT:    store <2 x float> [[TMP5]], ptr [[BEZT]], align 4427; POW2-ONLY-NEXT:    [[TMP6:%.*]] = tail call float @llvm.fmuladd.f32(float [[FNEG]], float 0.000000e+00, float 0.000000e+00)428; POW2-ONLY-NEXT:    [[ARRAYIDX8_I831:%.*]] = getelementptr float, ptr [[BEZT]], i64 2429; POW2-ONLY-NEXT:    store float [[TMP6]], ptr [[ARRAYIDX8_I831]], align 4430; POW2-ONLY-NEXT:    ret void431;432entry:433  %fneg = fmul float %0, 0.000000e+00434  %1 = tail call float @llvm.fmuladd.f32(float %0, float %fneg, float 0.000000e+00)435  store float %1, ptr %bezt, align 4436  %2 = tail call float @llvm.fmuladd.f32(float %0, float %fneg, float 0.000000e+00)437  %arrayidx5.i = getelementptr float, ptr %bezt, i64 1438  store float %2, ptr %arrayidx5.i, align 4439  %3 = tail call float @llvm.fmuladd.f32(float %fneg, float 0.000000e+00, float 0.000000e+00)440  %arrayidx8.i831 = getelementptr float, ptr %bezt, i64 2441  store float %3, ptr %arrayidx8.i831, align 4442  ret void443}444 445define void @reuse_shuffle_indices_cost_crash_3(ptr %m, double %conv, double %conv2) {446; CHECK-LABEL: define void @reuse_shuffle_indices_cost_crash_3(447; CHECK-SAME: ptr [[M:%.*]], double [[CONV:%.*]], double [[CONV2:%.*]]) {448; CHECK-NEXT:  entry:449; CHECK-NEXT:    [[SUB19:%.*]] = fsub double 0.000000e+00, [[CONV2]]450; CHECK-NEXT:    [[CONV20:%.*]] = fptrunc double [[SUB19]] to float451; CHECK-NEXT:    store float [[CONV20]], ptr [[M]], align 4452; CHECK-NEXT:    [[ADD:%.*]] = fadd double [[CONV]], 0.000000e+00453; CHECK-NEXT:    [[CONV239:%.*]] = fptrunc double [[ADD]] to float454; CHECK-NEXT:    [[ARRAYIDX25:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 1455; CHECK-NEXT:    store float [[CONV239]], ptr [[ARRAYIDX25]], align 4456; CHECK-NEXT:    [[ADD26:%.*]] = fsub double [[CONV]], [[CONV]]457; CHECK-NEXT:    [[CONV27:%.*]] = fptrunc double [[ADD26]] to float458; CHECK-NEXT:    [[ARRAYIDX29:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 2459; CHECK-NEXT:    store float [[CONV27]], ptr [[ARRAYIDX29]], align 4460; CHECK-NEXT:    ret void461;462entry:463  %sub19 = fsub double 0.000000e+00, %conv2464  %conv20 = fptrunc double %sub19 to float465  store float %conv20, ptr %m, align 4466  %add = fadd double %conv, 0.000000e+00467  %conv239 = fptrunc double %add to float468  %arrayidx25 = getelementptr [4 x float], ptr %m, i64 0, i64 1469  store float %conv239, ptr %arrayidx25, align 4470  %add26 = fsub double %conv, %conv471  %conv27 = fptrunc double %add26 to float472  %arrayidx29 = getelementptr [4 x float], ptr %m, i64 0, i64 2473  store float %conv27, ptr %arrayidx29, align 4474  ret void475}476 477define void @reuse_shuffle_indices_cost_crash_4(double %conv7.i) {478; CHECK-LABEL: define void @reuse_shuffle_indices_cost_crash_4(479; CHECK-SAME: double [[CONV7_I:%.*]]) {480; CHECK-NEXT:  entry:481; CHECK-NEXT:    [[DATA_I111:%.*]] = alloca [0 x [0 x [0 x [3 x float]]]], i32 0, align 4482; CHECK-NEXT:    [[ARRAYIDX_2_I:%.*]] = getelementptr [3 x float], ptr [[DATA_I111]], i64 0, i64 2483; CHECK-NEXT:    [[MUL17_I_US:%.*]] = fmul double [[CONV7_I]], 0.000000e+00484; CHECK-NEXT:    [[MUL_2_I_I_US:%.*]] = fmul double [[MUL17_I_US]], 0.000000e+00485; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> poison, double [[CONV7_I]], i32 0486; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <2 x double> [[TMP0]], <2 x double> poison, <2 x i32> zeroinitializer487; CHECK-NEXT:    [[TMP2:%.*]] = fadd <2 x double> [[TMP1]], zeroinitializer488; CHECK-NEXT:    [[ADD_2_I_I_US:%.*]] = fadd double [[MUL_2_I_I_US]], 0.000000e+00489; CHECK-NEXT:    [[TMP3:%.*]] = fmul <2 x double> [[TMP2]], [[TMP1]]490; CHECK-NEXT:    [[TMP4:%.*]] = fadd <2 x double> [[TMP3]], zeroinitializer491; CHECK-NEXT:    [[TMP5:%.*]] = fptrunc <2 x double> [[TMP4]] to <2 x float>492; CHECK-NEXT:    store <2 x float> [[TMP5]], ptr [[DATA_I111]], align 4493; CHECK-NEXT:    [[CONV_2_I46_US:%.*]] = fptrunc double [[ADD_2_I_I_US]] to float494; CHECK-NEXT:    store float [[CONV_2_I46_US]], ptr [[ARRAYIDX_2_I]], align 4495; CHECK-NEXT:    [[CALL2_I_US:%.*]] = load volatile ptr, ptr [[DATA_I111]], align 8496; CHECK-NEXT:    ret void497;498entry:499  %data.i111 = alloca [0 x [0 x [0 x [3 x float]]]], i32 0, align 4500  %arrayidx.1.i = getelementptr [3 x float], ptr %data.i111, i64 0, i64 1501  %arrayidx.2.i = getelementptr [3 x float], ptr %data.i111, i64 0, i64 2502  %mul17.i.us = fmul double %conv7.i, 0.000000e+00503  %mul.2.i.i.us = fmul double %mul17.i.us, 0.000000e+00504  %add.i.i82.i.us = fadd double %conv7.i, 0.000000e+00505  %add.1.i.i84.i.us = fadd double %conv7.i, 0.000000e+00506  %mul.i.i91.i.us = fmul double %add.i.i82.i.us, %conv7.i507  %mul.1.i.i92.i.us = fmul double %add.1.i.i84.i.us, %conv7.i508  %add.i96.i.us = fadd double %mul.i.i91.i.us, 0.000000e+00509  %add.1.i.i.us = fadd double %mul.1.i.i92.i.us, 0.000000e+00510  %add.2.i.i.us = fadd double %mul.2.i.i.us, 0.000000e+00511  %conv.i42.us = fptrunc double %add.i96.i.us to float512  store float %conv.i42.us, ptr %data.i111, align 4513  %conv.1.i44.us = fptrunc double %add.1.i.i.us to float514  store float %conv.1.i44.us, ptr %arrayidx.1.i, align 4515  %conv.2.i46.us = fptrunc double %add.2.i.i.us to float516  store float %conv.2.i46.us, ptr %arrayidx.2.i, align 4517  %call2.i.us = load volatile ptr, ptr %data.i111, align 8518  ret void519}520 521define void @common_mask(ptr %m, double %conv, double %conv2) {522; CHECK-LABEL: define void @common_mask(523; CHECK-SAME: ptr [[M:%.*]], double [[CONV:%.*]], double [[CONV2:%.*]]) {524; CHECK-NEXT:  entry:525; CHECK-NEXT:    [[SUB19:%.*]] = fsub double [[CONV]], [[CONV]]526; CHECK-NEXT:    [[CONV20:%.*]] = fptrunc double [[SUB19]] to float527; CHECK-NEXT:    store float [[CONV20]], ptr [[M]], align 4528; CHECK-NEXT:    [[ADD:%.*]] = fadd double [[CONV2]], 0.000000e+00529; CHECK-NEXT:    [[CONV239:%.*]] = fptrunc double [[ADD]] to float530; CHECK-NEXT:    [[ARRAYIDX25:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 1531; CHECK-NEXT:    store float [[CONV239]], ptr [[ARRAYIDX25]], align 4532; CHECK-NEXT:    [[ADD26:%.*]] = fsub double 0.000000e+00, [[CONV]]533; CHECK-NEXT:    [[CONV27:%.*]] = fptrunc double [[ADD26]] to float534; CHECK-NEXT:    [[ARRAYIDX29:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 2535; CHECK-NEXT:    store float [[CONV27]], ptr [[ARRAYIDX29]], align 4536; CHECK-NEXT:    ret void537;538entry:539  %sub19 = fsub double %conv, %conv540  %conv20 = fptrunc double %sub19 to float541  store float %conv20, ptr %m, align 4542  %add = fadd double %conv2, 0.000000e+00543  %conv239 = fptrunc double %add to float544  %arrayidx25 = getelementptr [4 x float], ptr %m, i64 0, i64 1545  store float %conv239, ptr %arrayidx25, align 4546  %add26 = fsub double 0.000000e+00, %conv547  %conv27 = fptrunc double %add26 to float548  %arrayidx29 = getelementptr [4 x float], ptr %m, i64 0, i64 2549  store float %conv27, ptr %arrayidx29, align 4550  ret void551}552 553define void @vec3_extract(<3 x i16> %pixel.sroa.0.4.vec.insert606, ptr %call3.i536) {554; CHECK-LABEL: define void @vec3_extract(555; CHECK-SAME: <3 x i16> [[PIXEL_SROA_0_4_VEC_INSERT606:%.*]], ptr [[CALL3_I536:%.*]]) {556; CHECK-NEXT:  entry:557; CHECK-NEXT:    [[PIXEL_SROA_0_4_VEC_EXTRACT:%.*]] = extractelement <3 x i16> [[PIXEL_SROA_0_4_VEC_INSERT606]], i64 2558; CHECK-NEXT:    [[RED668:%.*]] = getelementptr i16, ptr [[CALL3_I536]], i64 2559; CHECK-NEXT:    store i16 [[PIXEL_SROA_0_4_VEC_EXTRACT]], ptr [[RED668]], align 2560; CHECK-NEXT:    [[PIXEL_SROA_0_2_VEC_EXTRACT:%.*]] = extractelement <3 x i16> [[PIXEL_SROA_0_4_VEC_INSERT606]], i64 1561; CHECK-NEXT:    [[GREEN670:%.*]] = getelementptr i16, ptr [[CALL3_I536]], i64 1562; CHECK-NEXT:    store i16 [[PIXEL_SROA_0_2_VEC_EXTRACT]], ptr [[GREEN670]], align 2563; CHECK-NEXT:    [[PIXEL_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <3 x i16> [[PIXEL_SROA_0_4_VEC_INSERT606]], i64 0564; CHECK-NEXT:    store i16 [[PIXEL_SROA_0_0_VEC_EXTRACT]], ptr [[CALL3_I536]], align 2565; CHECK-NEXT:    ret void566;567entry:568  %pixel.sroa.0.4.vec.extract = extractelement <3 x i16> %pixel.sroa.0.4.vec.insert606, i64 2569  %red668 = getelementptr i16, ptr %call3.i536, i64 2570  store i16 %pixel.sroa.0.4.vec.extract, ptr %red668, align 2571  %pixel.sroa.0.2.vec.extract = extractelement <3 x i16> %pixel.sroa.0.4.vec.insert606, i64 1572  %green670 = getelementptr i16, ptr %call3.i536, i64 1573  store i16 %pixel.sroa.0.2.vec.extract, ptr %green670, align 2574  %pixel.sroa.0.0.vec.extract = extractelement <3 x i16> %pixel.sroa.0.4.vec.insert606, i64 0575  store i16 %pixel.sroa.0.0.vec.extract, ptr %call3.i536, align 2576  ret void577}578 579define void @can_reorder_vec3_op_with_padding(ptr %A, <3 x float> %in) {580; NON-POW2-LABEL: define void @can_reorder_vec3_op_with_padding(581; NON-POW2-SAME: ptr [[A:%.*]], <3 x float> [[IN:%.*]]) {582; NON-POW2-NEXT:  entry:583; NON-POW2-NEXT:    [[TMP1:%.*]] = fsub <3 x float> [[IN]], [[IN]]584; NON-POW2-NEXT:    [[TMP2:%.*]] = call <3 x float> @llvm.fmuladd.v3f32(<3 x float> [[TMP1]], <3 x float> splat (float 2.000000e+00), <3 x float> splat (float 3.000000e+00))585; NON-POW2-NEXT:    [[TMP3:%.*]] = fmul <3 x float> [[TMP2]], splat (float 3.000000e+00)586; NON-POW2-NEXT:    [[TMP4:%.*]] = shufflevector <3 x float> [[TMP3]], <3 x float> poison, <3 x i32> <i32 1, i32 2, i32 0>587; NON-POW2-NEXT:    store <3 x float> [[TMP4]], ptr [[A]], align 4588; NON-POW2-NEXT:    ret void589;590; POW2-ONLY-LABEL: define void @can_reorder_vec3_op_with_padding(591; POW2-ONLY-SAME: ptr [[A:%.*]], <3 x float> [[IN:%.*]]) {592; POW2-ONLY-NEXT:  entry:593; POW2-ONLY-NEXT:    [[ARRAYIDX42_I:%.*]] = getelementptr float, ptr [[A]], i64 2594; POW2-ONLY-NEXT:    [[TMP0:%.*]] = extractelement <3 x float> [[IN]], i64 0595; POW2-ONLY-NEXT:    [[SUB_I362:%.*]] = fsub float [[TMP0]], [[TMP0]]596; POW2-ONLY-NEXT:    [[TMP1:%.*]] = call float @llvm.fmuladd.f32(float [[SUB_I362]], float 2.000000e+00, float 3.000000e+00)597; POW2-ONLY-NEXT:    [[MUL6_I_I_I_I:%.*]] = fmul float [[TMP1]], 3.000000e+00598; POW2-ONLY-NEXT:    [[TMP2:%.*]] = shufflevector <3 x float> [[IN]], <3 x float> poison, <2 x i32> <i32 1, i32 2>599; POW2-ONLY-NEXT:    [[TMP3:%.*]] = fsub <2 x float> [[TMP2]], [[TMP2]]600; POW2-ONLY-NEXT:    [[TMP4:%.*]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP3]], <2 x float> splat (float 2.000000e+00), <2 x float> splat (float 3.000000e+00))601; POW2-ONLY-NEXT:    [[TMP5:%.*]] = fmul <2 x float> [[TMP4]], splat (float 3.000000e+00)602; POW2-ONLY-NEXT:    store <2 x float> [[TMP5]], ptr [[A]], align 4603; POW2-ONLY-NEXT:    store float [[MUL6_I_I_I_I]], ptr [[ARRAYIDX42_I]], align 4604; POW2-ONLY-NEXT:    ret void605;606entry:607  %arrayidx42.i = getelementptr float, ptr %A, i64 2608  %arrayidx35.i = getelementptr float, ptr %A, i64 1609  %0 = extractelement <3 x float> %in, i64 0610  %1 = extractelement <3 x float> %in, i64 0611  %sub.i362 = fsub float %0, %1612  %2 = extractelement <3 x float> %in, i64 1613  %3 = extractelement <3 x float> %in, i64 1614  %sub5.i = fsub float %2, %3615  %4 = extractelement <3 x float> %in, i64 2616  %5 = extractelement <3 x float> %in, i64 2617  %sub9.i = fsub float %4, %5618  %6 = call float @llvm.fmuladd.f32(float %sub5.i, float 2.000000e+00, float 3.000000e+00)619  %7 = call float @llvm.fmuladd.f32(float %sub9.i, float 2.000000e+00, float 3.000000e+00)620  %8 = call float @llvm.fmuladd.f32(float %sub.i362, float 2.000000e+00, float 3.000000e+00)621  %mul.i.i.i.i373 = fmul float %6, 3.000000e+00622  %mul3.i.i.i.i = fmul float %7, 3.000000e+00623  %mul6.i.i.i.i = fmul float %8, 3.000000e+00624  store float %mul.i.i.i.i373, ptr %A, align 4625  store float %mul3.i.i.i.i, ptr %arrayidx35.i, align 4626  store float %mul6.i.i.i.i, ptr %arrayidx42.i, align 4627  ret void628}629 630declare float @llvm.fmuladd.f32(float, float, float)631declare double @llvm.fmuladd.f64(double, double, double)632