Home | History | Annotate | Download | only in X86
      1 ; RUN: opt < %s  -loop-vectorize -mtriple=x86_64-apple-macosx10.8.0 -mcpu=corei7 -dce -instcombine -S | FileCheck %s
      2 ; RUN: opt < %s  -loop-vectorize -mtriple=x86_64-apple-macosx10.8.0 -mcpu=corei7 -force-vector-interleave=0 -dce -instcombine -S | FileCheck %s -check-prefix=UNROLL
      3 
      4 target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f32:32:32-f64:64:64-v64:64:64-v128:128:128-a0:0:64-s0:64:64-f80:128:128-n8:16:32:64-S128"
      5 target triple = "x86_64-apple-macosx10.8.0"
      6 
      7 @b = common global [2048 x i32] zeroinitializer, align 16
      8 @c = common global [2048 x i32] zeroinitializer, align 16
      9 @a = common global [2048 x i32] zeroinitializer, align 16
     10 
     11 ; Select VF = 8;
     12 ;CHECK-LABEL: @example1(
     13 ;CHECK: load <4 x i32>
     14 ;CHECK: add nsw <4 x i32>
     15 ;CHECK: store <4 x i32>
     16 ;CHECK: ret void
     17 
     18 ;UNROLL-LABEL: @example1(
     19 ;UNROLL: load <4 x i32>
     20 ;UNROLL: load <4 x i32>
     21 ;UNROLL: add nsw <4 x i32>
     22 ;UNROLL: add nsw <4 x i32>
     23 ;UNROLL: store <4 x i32>
     24 ;UNROLL: store <4 x i32>
     25 ;UNROLL: ret void
     26 define void @example1() nounwind uwtable ssp {
     27   br label %1
     28 
     29 ; <label>:1                                       ; preds = %1, %0
     30   %indvars.iv = phi i64 [ 0, %0 ], [ %indvars.iv.next, %1 ]
     31   %2 = getelementptr inbounds [2048 x i32], [2048 x i32]* @b, i64 0, i64 %indvars.iv
     32   %3 = load i32, i32* %2, align 4
     33   %4 = getelementptr inbounds [2048 x i32], [2048 x i32]* @c, i64 0, i64 %indvars.iv
     34   %5 = load i32, i32* %4, align 4
     35   %6 = add nsw i32 %5, %3
     36   %7 = getelementptr inbounds [2048 x i32], [2048 x i32]* @a, i64 0, i64 %indvars.iv
     37   store i32 %6, i32* %7, align 4
     38   %indvars.iv.next = add i64 %indvars.iv, 1
     39   %lftr.wideiv = trunc i64 %indvars.iv.next to i32
     40   %exitcond = icmp eq i32 %lftr.wideiv, 256
     41   br i1 %exitcond, label %8, label %1
     42 
     43 ; <label>:8                                       ; preds = %1
     44   ret void
     45 }
     46 
     47 ; Select VF=4 because sext <8 x i1> to <8 x i32> is expensive.
     48 ;CHECK-LABEL: @example10b(
     49 ;CHECK: load <4 x i16>
     50 ;CHECK: sext <4 x i16>
     51 ;CHECK: store <4 x i32>
     52 ;CHECK: ret void
     53 ;UNROLL-LABEL: @example10b(
     54 ;UNROLL: load <4 x i16>
     55 ;UNROLL: load <4 x i16>
     56 ;UNROLL: store <4 x i32>
     57 ;UNROLL: store <4 x i32>
     58 ;UNROLL: ret void
     59 define void @example10b(i16* noalias nocapture %sa, i16* noalias nocapture %sb, i16* noalias nocapture %sc, i32* noalias nocapture %ia, i32* noalias nocapture %ib, i32* noalias nocapture %ic) nounwind uwtable ssp {
     60   br label %1
     61 
     62 ; <label>:1                                       ; preds = %1, %0
     63   %indvars.iv = phi i64 [ 0, %0 ], [ %indvars.iv.next, %1 ]
     64   %2 = getelementptr inbounds i16, i16* %sb, i64 %indvars.iv
     65   %3 = load i16, i16* %2, align 2
     66   %4 = sext i16 %3 to i32
     67   %5 = getelementptr inbounds i32, i32* %ia, i64 %indvars.iv
     68   store i32 %4, i32* %5, align 4
     69   %indvars.iv.next = add i64 %indvars.iv, 1
     70   %lftr.wideiv = trunc i64 %indvars.iv.next to i32
     71   %exitcond = icmp eq i32 %lftr.wideiv, 1024
     72   br i1 %exitcond, label %6, label %1
     73 
     74 ; <label>:6                                       ; preds = %1
     75   ret void
     76 }
     77 
     78