xref: /aosp_15_r20/external/llvm/test/CodeGen/X86/exedepsfix-broadcast.ll (revision 9880d6810fe72a1726cb53787c6711e909410d58)
1*9880d681SAndroid Build Coastguard Worker; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
2*9880d681SAndroid Build Coastguard Worker; RUN: llc < %s -mtriple=x86_64-apple-macosx -mattr=+avx2 -enable-unsafe-fp-math | FileCheck %s
3*9880d681SAndroid Build Coastguard Worker
4*9880d681SAndroid Build Coastguard Worker; Check that the ExeDepsFix pass correctly fixes the domain for broadcast instructions.
5*9880d681SAndroid Build Coastguard Worker; <rdar://problem/16354675>
6*9880d681SAndroid Build Coastguard Worker
7*9880d681SAndroid Build Coastguard Workerdefine <4 x float> @ExeDepsFix_broadcastss(<4 x float> %arg, <4 x float> %arg2) {
8*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss:
9*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
10*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vbroadcastss {{.*}}(%rip), %xmm2
11*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vandps %xmm2, %xmm0, %xmm0
12*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxps %xmm1, %xmm0, %xmm0
13*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
14*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <4 x float> %arg to <4 x i32>
15*9880d681SAndroid Build Coastguard Worker  %and = and <4 x i32> %bitcast, <i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647>
16*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <4 x i32> %and to <4 x float>
17*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <4 x float> %floatcast, %arg2
18*9880d681SAndroid Build Coastguard Worker  %max = select <4 x i1> %max_is_x, <4 x float> %floatcast, <4 x float> %arg2
19*9880d681SAndroid Build Coastguard Worker  ret <4 x float> %max
20*9880d681SAndroid Build Coastguard Worker}
21*9880d681SAndroid Build Coastguard Worker
22*9880d681SAndroid Build Coastguard Workerdefine <8 x float> @ExeDepsFix_broadcastss256(<8 x float> %arg, <8 x float> %arg2) {
23*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss256:
24*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
25*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vbroadcastss {{.*}}(%rip), %ymm2
26*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vandps %ymm2, %ymm0, %ymm0
27*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxps %ymm1, %ymm0, %ymm0
28*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
29*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <8 x float> %arg to <8 x i32>
30*9880d681SAndroid Build Coastguard Worker  %and = and <8 x i32> %bitcast, <i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647, i32 2147483647>
31*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <8 x i32> %and to <8 x float>
32*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <8 x float> %floatcast, %arg2
33*9880d681SAndroid Build Coastguard Worker  %max = select <8 x i1> %max_is_x, <8 x float> %floatcast, <8 x float> %arg2
34*9880d681SAndroid Build Coastguard Worker  ret <8 x float> %max
35*9880d681SAndroid Build Coastguard Worker}
36*9880d681SAndroid Build Coastguard Worker
37*9880d681SAndroid Build Coastguard Workerdefine <4 x float> @ExeDepsFix_broadcastss_inreg(<4 x float> %arg, <4 x float> %arg2, i32 %broadcastvalue) {
38*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss_inreg:
39*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
40*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmovd %edi, %xmm2
41*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vbroadcastss %xmm2, %xmm2
42*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vandps %xmm2, %xmm0, %xmm0
43*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxps %xmm1, %xmm0, %xmm0
44*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
45*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <4 x float> %arg to <4 x i32>
46*9880d681SAndroid Build Coastguard Worker  %in = insertelement <4 x i32> undef, i32 %broadcastvalue, i32 0
47*9880d681SAndroid Build Coastguard Worker  %mask = shufflevector <4 x i32> %in, <4 x i32> undef, <4 x i32> zeroinitializer
48*9880d681SAndroid Build Coastguard Worker  %and = and <4 x i32> %bitcast, %mask
49*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <4 x i32> %and to <4 x float>
50*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <4 x float> %floatcast, %arg2
51*9880d681SAndroid Build Coastguard Worker  %max = select <4 x i1> %max_is_x, <4 x float> %floatcast, <4 x float> %arg2
52*9880d681SAndroid Build Coastguard Worker  ret <4 x float> %max
53*9880d681SAndroid Build Coastguard Worker}
54*9880d681SAndroid Build Coastguard Worker
55*9880d681SAndroid Build Coastguard Workerdefine <8 x float> @ExeDepsFix_broadcastss256_inreg(<8 x float> %arg, <8 x float> %arg2, i32 %broadcastvalue) {
56*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastss256_inreg:
57*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
58*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmovd %edi, %xmm2
59*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vbroadcastss %xmm2, %ymm2
60*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vandps %ymm2, %ymm0, %ymm0
61*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxps %ymm1, %ymm0, %ymm0
62*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
63*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <8 x float> %arg to <8 x i32>
64*9880d681SAndroid Build Coastguard Worker  %in = insertelement <8 x i32> undef, i32 %broadcastvalue, i32 0
65*9880d681SAndroid Build Coastguard Worker  %mask = shufflevector <8 x i32> %in, <8 x i32> undef, <8 x i32> zeroinitializer
66*9880d681SAndroid Build Coastguard Worker  %and = and <8 x i32> %bitcast, %mask
67*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <8 x i32> %and to <8 x float>
68*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <8 x float> %floatcast, %arg2
69*9880d681SAndroid Build Coastguard Worker  %max = select <8 x i1> %max_is_x, <8 x float> %floatcast, <8 x float> %arg2
70*9880d681SAndroid Build Coastguard Worker  ret <8 x float> %max
71*9880d681SAndroid Build Coastguard Worker}
72*9880d681SAndroid Build Coastguard Worker
73*9880d681SAndroid Build Coastguard Worker; In that case the broadcast is directly folded into vandpd.
74*9880d681SAndroid Build Coastguard Workerdefine <2 x double> @ExeDepsFix_broadcastsd(<2 x double> %arg, <2 x double> %arg2) {
75*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd:
76*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
77*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vandpd {{.*}}(%rip), %xmm0, %xmm0
78*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxpd %xmm1, %xmm0, %xmm0
79*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
80*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <2 x double> %arg to <2 x i64>
81*9880d681SAndroid Build Coastguard Worker  %and = and <2 x i64> %bitcast, <i64 2147483647, i64 2147483647>
82*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <2 x i64> %and to <2 x double>
83*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <2 x double> %floatcast, %arg2
84*9880d681SAndroid Build Coastguard Worker  %max = select <2 x i1> %max_is_x, <2 x double> %floatcast, <2 x double> %arg2
85*9880d681SAndroid Build Coastguard Worker  ret <2 x double> %max
86*9880d681SAndroid Build Coastguard Worker}
87*9880d681SAndroid Build Coastguard Worker
88*9880d681SAndroid Build Coastguard Workerdefine <4 x double> @ExeDepsFix_broadcastsd256(<4 x double> %arg, <4 x double> %arg2) {
89*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd256:
90*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
91*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vbroadcastsd {{.*}}(%rip), %ymm2
92*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vandpd %ymm2, %ymm0, %ymm0
93*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxpd %ymm1, %ymm0, %ymm0
94*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
95*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <4 x double> %arg to <4 x i64>
96*9880d681SAndroid Build Coastguard Worker  %and = and <4 x i64> %bitcast, <i64 2147483647, i64 2147483647, i64 2147483647, i64 2147483647>
97*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <4 x i64> %and to <4 x double>
98*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <4 x double> %floatcast, %arg2
99*9880d681SAndroid Build Coastguard Worker  %max = select <4 x i1> %max_is_x, <4 x double> %floatcast, <4 x double> %arg2
100*9880d681SAndroid Build Coastguard Worker  ret <4 x double> %max
101*9880d681SAndroid Build Coastguard Worker}
102*9880d681SAndroid Build Coastguard Worker
103*9880d681SAndroid Build Coastguard Worker; ExeDepsFix works top down, thus it coalesces vpunpcklqdq domain with
104*9880d681SAndroid Build Coastguard Worker; vpand and there is nothing more you can do to match vmaxpd.
105*9880d681SAndroid Build Coastguard Workerdefine <2 x double> @ExeDepsFix_broadcastsd_inreg(<2 x double> %arg, <2 x double> %arg2, i64 %broadcastvalue) {
106*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd_inreg:
107*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
108*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmovq %rdi, %xmm2
109*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vpbroadcastq %xmm2, %xmm2
110*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vpand %xmm2, %xmm0, %xmm0
111*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxpd %xmm1, %xmm0, %xmm0
112*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
113*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <2 x double> %arg to <2 x i64>
114*9880d681SAndroid Build Coastguard Worker  %in = insertelement <2 x i64> undef, i64 %broadcastvalue, i32 0
115*9880d681SAndroid Build Coastguard Worker  %mask = shufflevector <2 x i64> %in, <2 x i64> undef, <2 x i32> zeroinitializer
116*9880d681SAndroid Build Coastguard Worker  %and = and <2 x i64> %bitcast, %mask
117*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <2 x i64> %and to <2 x double>
118*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <2 x double> %floatcast, %arg2
119*9880d681SAndroid Build Coastguard Worker  %max = select <2 x i1> %max_is_x, <2 x double> %floatcast, <2 x double> %arg2
120*9880d681SAndroid Build Coastguard Worker  ret <2 x double> %max
121*9880d681SAndroid Build Coastguard Worker}
122*9880d681SAndroid Build Coastguard Worker
123*9880d681SAndroid Build Coastguard Workerdefine <4 x double> @ExeDepsFix_broadcastsd256_inreg(<4 x double> %arg, <4 x double> %arg2, i64 %broadcastvalue) {
124*9880d681SAndroid Build Coastguard Worker; CHECK-LABEL: ExeDepsFix_broadcastsd256_inreg:
125*9880d681SAndroid Build Coastguard Worker; CHECK:       ## BB#0:
126*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmovq %rdi, %xmm2
127*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vbroadcastsd %xmm2, %ymm2
128*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vandpd %ymm2, %ymm0, %ymm0
129*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    vmaxpd %ymm1, %ymm0, %ymm0
130*9880d681SAndroid Build Coastguard Worker; CHECK-NEXT:    retq
131*9880d681SAndroid Build Coastguard Worker  %bitcast = bitcast <4 x double> %arg to <4 x i64>
132*9880d681SAndroid Build Coastguard Worker  %in = insertelement <4 x i64> undef, i64 %broadcastvalue, i32 0
133*9880d681SAndroid Build Coastguard Worker  %mask = shufflevector <4 x i64> %in, <4 x i64> undef, <4 x i32> zeroinitializer
134*9880d681SAndroid Build Coastguard Worker  %and = and <4 x i64> %bitcast, %mask
135*9880d681SAndroid Build Coastguard Worker  %floatcast = bitcast <4 x i64> %and to <4 x double>
136*9880d681SAndroid Build Coastguard Worker  %max_is_x = fcmp oge <4 x double> %floatcast, %arg2
137*9880d681SAndroid Build Coastguard Worker  %max = select <4 x i1> %max_is_x, <4 x double> %floatcast, <4 x double> %arg2
138*9880d681SAndroid Build Coastguard Worker  ret <4 x double> %max
139*9880d681SAndroid Build Coastguard Worker}
140