1*9880d681SAndroid Build Coastguard Worker; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py 2*9880d681SAndroid Build Coastguard Worker; RUN: llc < %s -mtriple=x86_64-apple-macosx -mattr=+avx2 -enable-unsafe-fp-math | FileCheck %s 3*9880d681SAndroid Build Coastguard Worker 4*9880d681SAndroid Build Coastguard Worker; Check that the ExeDepsFix pass correctly fixes the domain for broadcast instructions. 5*9880d681SAndroid Build Coastguard Worker; <rdar://problem/16354675> 6*9880d681SAndroid Build Coastguard Worker 7*9880d681SAndroid Build Coastguard Workerdefine <4 x float> @ExeDepsFix_broadcastss(<4 x float> %arg, <4 x float> %arg2) { 8*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss: 9*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 10*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vbroadcastss {{.*}}(%rip), %xmm2 11*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vandps %xmm2, %xmm0, %xmm0 12*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxps %xmm1, %xmm0, %xmm0 13*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 14*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <4 x float> %arg to <4 x i32> 15*9880d681SAndroid Build Coastguard Worker %and = and <4 x i32> %bitcast, <i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647> 16*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <4 x i32> %and to <4 x float> 17*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <4 x float> %floatcast, %arg2 18*9880d681SAndroid Build Coastguard Worker %max = select <4 x i1> %max_is_x, <4 x float> %floatcast, <4 x float> %arg2 19*9880d681SAndroid Build Coastguard Worker ret <4 x float> %max 20*9880d681SAndroid Build Coastguard Worker} 21*9880d681SAndroid Build Coastguard Worker 22*9880d681SAndroid Build Coastguard Workerdefine <8 x float> @ExeDepsFix_broadcastss256(<8 x float> %arg, <8 x float> %arg2) { 23*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss256: 24*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 25*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vbroadcastss {{.*}}(%rip), %ymm2 26*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vandps %ymm2, %ymm0, %ymm0 27*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxps %ymm1, %ymm0, %ymm0 28*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 29*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <8 x float> %arg to <8 x i32> 30*9880d681SAndroid Build Coastguard Worker %and = and <8 x i32> %bitcast, <i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647> 31*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <8 x i32> %and to <8 x float> 32*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <8 x float> %floatcast, %arg2 33*9880d681SAndroid Build Coastguard Worker %max = select <8 x i1> %max_is_x, <8 x float> %floatcast, <8 x float> %arg2 34*9880d681SAndroid Build Coastguard Worker ret <8 x float> %max 35*9880d681SAndroid Build Coastguard Worker} 36*9880d681SAndroid Build Coastguard Worker 37*9880d681SAndroid Build Coastguard Workerdefine <4 x float> @ExeDepsFix_broadcastss_inreg(<4 x float> %arg, <4 x float> %arg2, i32 %broadcastvalue) { 38*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss_inreg: 39*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 40*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmovd %edi, %xmm2 41*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vbroadcastss %xmm2, %xmm2 42*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vandps %xmm2, %xmm0, %xmm0 43*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxps %xmm1, %xmm0, %xmm0 44*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 45*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <4 x float> %arg to <4 x i32> 46*9880d681SAndroid Build Coastguard Worker %in = insertelement <4 x i32> undef, i32 %broadcastvalue, i32 0 47*9880d681SAndroid Build Coastguard Worker %mask = shufflevector <4 x i32> %in, <4 x i32> undef, <4 x i32> zeroinitializer 48*9880d681SAndroid Build Coastguard Worker %and = and <4 x i32> %bitcast, %mask 49*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <4 x i32> %and to <4 x float> 50*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <4 x float> %floatcast, %arg2 51*9880d681SAndroid Build Coastguard Worker %max = select <4 x i1> %max_is_x, <4 x float> %floatcast, <4 x float> %arg2 52*9880d681SAndroid Build Coastguard Worker ret <4 x float> %max 53*9880d681SAndroid Build Coastguard Worker} 54*9880d681SAndroid Build Coastguard Worker 55*9880d681SAndroid Build Coastguard Workerdefine <8 x float> @ExeDepsFix_broadcastss256_inreg(<8 x float> %arg, <8 x float> %arg2, i32 %broadcastvalue) { 56*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss256_inreg: 57*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 58*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmovd %edi, %xmm2 59*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vbroadcastss %xmm2, %ymm2 60*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vandps %ymm2, %ymm0, %ymm0 61*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxps %ymm1, %ymm0, %ymm0 62*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 63*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <8 x float> %arg to <8 x i32> 64*9880d681SAndroid Build Coastguard Worker %in = insertelement <8 x i32> undef, i32 %broadcastvalue, i32 0 65*9880d681SAndroid Build Coastguard Worker %mask = shufflevector <8 x i32> %in, <8 x i32> undef, <8 x i32> zeroinitializer 66*9880d681SAndroid Build Coastguard Worker %and = and <8 x i32> %bitcast, %mask 67*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <8 x i32> %and to <8 x float> 68*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <8 x float> %floatcast, %arg2 69*9880d681SAndroid Build Coastguard Worker %max = select <8 x i1> %max_is_x, <8 x float> %floatcast, <8 x float> %arg2 70*9880d681SAndroid Build Coastguard Worker ret <8 x float> %max 71*9880d681SAndroid Build Coastguard Worker} 72*9880d681SAndroid Build Coastguard Worker 73*9880d681SAndroid Build Coastguard Worker; In that case the broadcast is directly folded into vandpd. 74*9880d681SAndroid Build Coastguard Workerdefine <2 x double> @ExeDepsFix_broadcastsd(<2 x double> %arg, <2 x double> %arg2) { 75*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd: 76*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 77*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vandpd {{.*}}(%rip), %xmm0, %xmm0 78*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxpd %xmm1, %xmm0, %xmm0 79*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 80*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <2 x double> %arg to <2 x i64> 81*9880d681SAndroid Build Coastguard Worker %and = and <2 x i64> %bitcast, <i64 2147483647, i64 2147483647> 82*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <2 x i64> %and to <2 x double> 83*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <2 x double> %floatcast, %arg2 84*9880d681SAndroid Build Coastguard Worker %max = select <2 x i1> %max_is_x, <2 x double> %floatcast, <2 x double> %arg2 85*9880d681SAndroid Build Coastguard Worker ret <2 x double> %max 86*9880d681SAndroid Build Coastguard Worker} 87*9880d681SAndroid Build Coastguard Worker 88*9880d681SAndroid Build Coastguard Workerdefine <4 x double> @ExeDepsFix_broadcastsd256(<4 x double> %arg, <4 x double> %arg2) { 89*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd256: 90*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 91*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vbroadcastsd {{.*}}(%rip), %ymm2 92*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vandpd %ymm2, %ymm0, %ymm0 93*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxpd %ymm1, %ymm0, %ymm0 94*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 95*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <4 x double> %arg to <4 x i64> 96*9880d681SAndroid Build Coastguard Worker %and = and <4 x i64> %bitcast, <i64 2147483647, i64 2147483647, i64 2147483647, i64 2147483647> 97*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <4 x i64> %and to <4 x double> 98*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <4 x double> %floatcast, %arg2 99*9880d681SAndroid Build Coastguard Worker %max = select <4 x i1> %max_is_x, <4 x double> %floatcast, <4 x double> %arg2 100*9880d681SAndroid Build Coastguard Worker ret <4 x double> %max 101*9880d681SAndroid Build Coastguard Worker} 102*9880d681SAndroid Build Coastguard Worker 103*9880d681SAndroid Build Coastguard Worker; ExeDepsFix works top down, thus it coalesces vpunpcklqdq domain with 104*9880d681SAndroid Build Coastguard Worker; vpand and there is nothing more you can do to match vmaxpd. 105*9880d681SAndroid Build Coastguard Workerdefine <2 x double> @ExeDepsFix_broadcastsd_inreg(<2 x double> %arg, <2 x double> %arg2, i64 %broadcastvalue) { 106*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd_inreg: 107*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 108*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmovq %rdi, %xmm2 109*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vpbroadcastq %xmm2, %xmm2 110*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vpand %xmm2, %xmm0, %xmm0 111*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxpd %xmm1, %xmm0, %xmm0 112*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 113*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <2 x double> %arg to <2 x i64> 114*9880d681SAndroid Build Coastguard Worker %in = insertelement <2 x i64> undef, i64 %broadcastvalue, i32 0 115*9880d681SAndroid Build Coastguard Worker %mask = shufflevector <2 x i64> %in, <2 x i64> undef, <2 x i32> zeroinitializer 116*9880d681SAndroid Build Coastguard Worker %and = and <2 x i64> %bitcast, %mask 117*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <2 x i64> %and to <2 x double> 118*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <2 x double> %floatcast, %arg2 119*9880d681SAndroid Build Coastguard Worker %max = select <2 x i1> %max_is_x, <2 x double> %floatcast, <2 x double> %arg2 120*9880d681SAndroid Build Coastguard Worker ret <2 x double> %max 121*9880d681SAndroid Build Coastguard Worker} 122*9880d681SAndroid Build Coastguard Worker 123*9880d681SAndroid Build Coastguard Workerdefine <4 x double> @ExeDepsFix_broadcastsd256_inreg(<4 x double> %arg, <4 x double> %arg2, i64 %broadcastvalue) { 124*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd256_inreg: 125*9880d681SAndroid Build Coastguard Worker; CHECK: ## BB#0: 126*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmovq %rdi, %xmm2 127*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vbroadcastsd %xmm2, %ymm2 128*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vandpd %ymm2, %ymm0, %ymm0 129*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: vmaxpd %ymm1, %ymm0, %ymm0 130*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT: retq 131*9880d681SAndroid Build Coastguard Worker %bitcast = bitcast <4 x double> %arg to <4 x i64> 132*9880d681SAndroid Build Coastguard Worker %in = insertelement <4 x i64> undef, i64 %broadcastvalue, i32 0 133*9880d681SAndroid Build Coastguard Worker %mask = shufflevector <4 x i64> %in, <4 x i64> undef, <4 x i32> zeroinitializer 134*9880d681SAndroid Build Coastguard Worker %and = and <4 x i64> %bitcast, %mask 135*9880d681SAndroid Build Coastguard Worker %floatcast = bitcast <4 x i64> %and to <4 x double> 136*9880d681SAndroid Build Coastguard Worker %max_is_x = fcmp oge <4 x double> %floatcast, %arg2 137*9880d681SAndroid Build Coastguard Worker %max = select <4 x i1> %max_is_x, <4 x double> %floatcast, <4 x double> %arg2 138*9880d681SAndroid Build Coastguard Worker ret <4 x double> %max 139*9880d681SAndroid Build Coastguard Worker} 140