LLVM 24.0.0git
AMDGPUISelLowering.cpp
Go to the documentation of this file.
1//===-- AMDGPUISelLowering.cpp - AMDGPU Common DAG lowering functions -----===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This is the parent TargetLowering class for hardware code gen
11/// targets.
12//
13//===----------------------------------------------------------------------===//
14
15#include "AMDGPUISelLowering.h"
16#include "AMDGPU.h"
17#include "AMDGPUInstrInfo.h"
19#include "AMDGPUMemoryUtils.h"
26#include "llvm/IR/IntrinsicsAMDGPU.h"
30
31using namespace llvm;
32
33#define GET_CALLING_CONV_IMPL
34#include "AMDGPUGenCallingConv.inc"
35
37 "amdgpu-bypass-slow-div",
38 cl::desc("Skip 64-bit divide for dynamic 32-bit values"),
39 cl::init(true));
40
41// Find a larger type to do a load / store of a vector with.
43 unsigned StoreSize = VT.getStoreSizeInBits();
44 if (StoreSize <= 32)
45 return EVT::getIntegerVT(Ctx, StoreSize);
46
47 if (StoreSize % 32 == 0)
48 return EVT::getVectorVT(Ctx, MVT::i32, StoreSize / 32);
49
50 return VT;
51}
52
56
58 // In order for this to be a signed 24-bit value, bit 23, must
59 // be a sign bit.
60 return DAG.ComputeMaxSignificantBits(Op);
61}
62
64 const TargetSubtargetInfo &STI,
65 const AMDGPUSubtarget &AMDGPUSTI)
66 : TargetLowering(TM, STI), Subtarget(&AMDGPUSTI) {
67 // Always lower memset, memcpy, and memmove intrinsics to load/store
68 // instructions, rather then generating calls to memset, mempcy or memmove.
72
73 // Enable ganging up loads and stores in the memcpy DAG lowering.
75
76 // Lower floating point store/load to integer store/load to reduce the number
77 // of patterns in tablegen.
79 AddPromotedToType(ISD::LOAD, MVT::f32, MVT::i32);
80
82 AddPromotedToType(ISD::LOAD, MVT::v2f32, MVT::v2i32);
83
85 AddPromotedToType(ISD::LOAD, MVT::v3f32, MVT::v3i32);
86
88 AddPromotedToType(ISD::LOAD, MVT::v4f32, MVT::v4i32);
89
91 AddPromotedToType(ISD::LOAD, MVT::v5f32, MVT::v5i32);
92
94 AddPromotedToType(ISD::LOAD, MVT::v6f32, MVT::v6i32);
95
97 AddPromotedToType(ISD::LOAD, MVT::v7f32, MVT::v7i32);
98
100 AddPromotedToType(ISD::LOAD, MVT::v8f32, MVT::v8i32);
101
103 AddPromotedToType(ISD::LOAD, MVT::v9f32, MVT::v9i32);
104
105 setOperationAction(ISD::LOAD, MVT::v10f32, Promote);
106 AddPromotedToType(ISD::LOAD, MVT::v10f32, MVT::v10i32);
107
108 setOperationAction(ISD::LOAD, MVT::v11f32, Promote);
109 AddPromotedToType(ISD::LOAD, MVT::v11f32, MVT::v11i32);
110
111 setOperationAction(ISD::LOAD, MVT::v12f32, Promote);
112 AddPromotedToType(ISD::LOAD, MVT::v12f32, MVT::v12i32);
113
114 setOperationAction(ISD::LOAD, MVT::v16f32, Promote);
115 AddPromotedToType(ISD::LOAD, MVT::v16f32, MVT::v16i32);
116
117 setOperationAction(ISD::LOAD, MVT::v32f32, Promote);
118 AddPromotedToType(ISD::LOAD, MVT::v32f32, MVT::v32i32);
119
121 AddPromotedToType(ISD::LOAD, MVT::i64, MVT::v2i32);
122
124 AddPromotedToType(ISD::LOAD, MVT::v2i64, MVT::v4i32);
125
127 AddPromotedToType(ISD::LOAD, MVT::f64, MVT::v2i32);
128
130 AddPromotedToType(ISD::LOAD, MVT::v2f64, MVT::v4i32);
131
133 AddPromotedToType(ISD::LOAD, MVT::v3i64, MVT::v6i32);
134
136 AddPromotedToType(ISD::LOAD, MVT::v4i64, MVT::v8i32);
137
139 AddPromotedToType(ISD::LOAD, MVT::v3f64, MVT::v6i32);
140
142 AddPromotedToType(ISD::LOAD, MVT::v4f64, MVT::v8i32);
143
145 AddPromotedToType(ISD::LOAD, MVT::v8i64, MVT::v16i32);
146
148 AddPromotedToType(ISD::LOAD, MVT::v8f64, MVT::v16i32);
149
150 setOperationAction(ISD::LOAD, MVT::v16i64, Promote);
151 AddPromotedToType(ISD::LOAD, MVT::v16i64, MVT::v32i32);
152
153 setOperationAction(ISD::LOAD, MVT::v16f64, Promote);
154 AddPromotedToType(ISD::LOAD, MVT::v16f64, MVT::v32i32);
155
157 AddPromotedToType(ISD::LOAD, MVT::i128, MVT::v4i32);
158
159 // TODO: Would be better to consume as directly legal
161 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f32, MVT::i32);
162
164 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f64, MVT::i64);
165
167 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f16, MVT::i16);
168
170 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::bf16, MVT::i16);
171
173 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::v2f32, MVT::i64);
174
176 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f32, MVT::i32);
177
179 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f64, MVT::i64);
180
182 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f16, MVT::i16);
183
185 AddPromotedToType(ISD::ATOMIC_STORE, MVT::bf16, MVT::i16);
186
188 AddPromotedToType(ISD::ATOMIC_STORE, MVT::v2f32, MVT::i64);
189
190 // There are no 64-bit extloads. These should be done as a 32-bit extload and
191 // an extension to 64-bit.
192 for (MVT VT : MVT::integer_valuetypes())
194 Expand);
195
196 for (MVT VT : MVT::integer_valuetypes()) {
197 if (VT == MVT::i64)
198 continue;
199
200 for (auto Op : {ISD::SEXTLOAD, ISD::ZEXTLOAD, ISD::EXTLOAD}) {
201 setLoadExtAction(Op, VT, MVT::i1, Promote);
202 setLoadExtAction(Op, VT, MVT::i8, Legal);
203 setLoadExtAction(Op, VT, MVT::i16, Legal);
204 setLoadExtAction(Op, VT, MVT::i32, Expand);
205 }
206 }
207
209 for (auto MemVT :
210 {MVT::v2i8, MVT::v4i8, MVT::v2i16, MVT::v3i16, MVT::v4i16})
212 Expand);
213
214 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::f16, Expand);
215 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::bf16, Expand);
216 setLoadExtAction(ISD::EXTLOAD, MVT::v2f32, MVT::v2f16, Expand);
217 setLoadExtAction(ISD::EXTLOAD, MVT::v2f32, MVT::v2bf16, Expand);
218 setLoadExtAction(ISD::EXTLOAD, MVT::v3f32, MVT::v3f16, Expand);
219 setLoadExtAction(ISD::EXTLOAD, MVT::v3f32, MVT::v3bf16, Expand);
220 setLoadExtAction(ISD::EXTLOAD, MVT::v4f32, MVT::v4f16, Expand);
221 setLoadExtAction(ISD::EXTLOAD, MVT::v4f32, MVT::v4bf16, Expand);
222 setLoadExtAction(ISD::EXTLOAD, MVT::v8f32, MVT::v8f16, Expand);
223 setLoadExtAction(ISD::EXTLOAD, MVT::v8f32, MVT::v8bf16, Expand);
224 setLoadExtAction(ISD::EXTLOAD, MVT::v16f32, MVT::v16f16, Expand);
225 setLoadExtAction(ISD::EXTLOAD, MVT::v16f32, MVT::v16bf16, Expand);
226 setLoadExtAction(ISD::EXTLOAD, MVT::v32f32, MVT::v32f16, Expand);
227 setLoadExtAction(ISD::EXTLOAD, MVT::v32f32, MVT::v32bf16, Expand);
228
229 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand);
230 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f32, Expand);
231 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3f32, Expand);
232 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4f32, Expand);
233 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8f32, Expand);
234 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16f32, Expand);
235
236 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f16, Expand);
237 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::bf16, Expand);
238 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f16, Expand);
239 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2bf16, Expand);
240 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3f16, Expand);
241 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3bf16, Expand);
242 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4f16, Expand);
243 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4bf16, Expand);
244 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8f16, Expand);
245 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8bf16, Expand);
246 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16f16, Expand);
247 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16bf16, Expand);
248
250 AddPromotedToType(ISD::STORE, MVT::f32, MVT::i32);
251
253 AddPromotedToType(ISD::STORE, MVT::v2f32, MVT::v2i32);
254
256 AddPromotedToType(ISD::STORE, MVT::v3f32, MVT::v3i32);
257
259 AddPromotedToType(ISD::STORE, MVT::v4f32, MVT::v4i32);
260
262 AddPromotedToType(ISD::STORE, MVT::v5f32, MVT::v5i32);
263
265 AddPromotedToType(ISD::STORE, MVT::v6f32, MVT::v6i32);
266
268 AddPromotedToType(ISD::STORE, MVT::v7f32, MVT::v7i32);
269
271 AddPromotedToType(ISD::STORE, MVT::v8f32, MVT::v8i32);
272
274 AddPromotedToType(ISD::STORE, MVT::v9f32, MVT::v9i32);
275
277 AddPromotedToType(ISD::STORE, MVT::v10f32, MVT::v10i32);
278
280 AddPromotedToType(ISD::STORE, MVT::v11f32, MVT::v11i32);
281
283 AddPromotedToType(ISD::STORE, MVT::v12f32, MVT::v12i32);
284
286 AddPromotedToType(ISD::STORE, MVT::v16f32, MVT::v16i32);
287
289 AddPromotedToType(ISD::STORE, MVT::v32f32, MVT::v32i32);
290
292 AddPromotedToType(ISD::STORE, MVT::i64, MVT::v2i32);
293
295 AddPromotedToType(ISD::STORE, MVT::v2i64, MVT::v4i32);
296
298 AddPromotedToType(ISD::STORE, MVT::f64, MVT::v2i32);
299
301 AddPromotedToType(ISD::STORE, MVT::v2f64, MVT::v4i32);
302
304 AddPromotedToType(ISD::STORE, MVT::v3i64, MVT::v6i32);
305
307 AddPromotedToType(ISD::STORE, MVT::v3f64, MVT::v6i32);
308
310 AddPromotedToType(ISD::STORE, MVT::v4i64, MVT::v8i32);
311
313 AddPromotedToType(ISD::STORE, MVT::v4f64, MVT::v8i32);
314
316 AddPromotedToType(ISD::STORE, MVT::v8i64, MVT::v16i32);
317
319 AddPromotedToType(ISD::STORE, MVT::v8f64, MVT::v16i32);
320
322 AddPromotedToType(ISD::STORE, MVT::v16i64, MVT::v32i32);
323
325 AddPromotedToType(ISD::STORE, MVT::v16f64, MVT::v32i32);
326
328 AddPromotedToType(ISD::STORE, MVT::i128, MVT::v4i32);
329
330 setTruncStoreAction(MVT::i64, MVT::i1, Expand);
331 setTruncStoreAction(MVT::i64, MVT::i8, Expand);
332 setTruncStoreAction(MVT::i64, MVT::i16, Expand);
333 setTruncStoreAction(MVT::i64, MVT::i32, Expand);
334
335 setTruncStoreAction(MVT::v2i64, MVT::v2i1, Expand);
336 setTruncStoreAction(MVT::v2i64, MVT::v2i8, Expand);
337 setTruncStoreAction(MVT::v2i64, MVT::v2i16, Expand);
338 setTruncStoreAction(MVT::v2i64, MVT::v2i32, Expand);
339
340 setTruncStoreAction(MVT::f32, MVT::bf16, Expand);
341 setTruncStoreAction(MVT::f32, MVT::f16, Expand);
342 setTruncStoreAction(MVT::v2f32, MVT::v2bf16, Expand);
343 setTruncStoreAction(MVT::v2f32, MVT::v2f16, Expand);
344 setTruncStoreAction(MVT::v3f32, MVT::v3bf16, Expand);
345 setTruncStoreAction(MVT::v3f32, MVT::v3f16, Expand);
346 setTruncStoreAction(MVT::v4f32, MVT::v4bf16, Expand);
347 setTruncStoreAction(MVT::v4f32, MVT::v4f16, Expand);
348 setTruncStoreAction(MVT::v6f32, MVT::v6f16, Expand);
349 setTruncStoreAction(MVT::v8f32, MVT::v8bf16, Expand);
350 setTruncStoreAction(MVT::v8f32, MVT::v8f16, Expand);
351 setTruncStoreAction(MVT::v16f32, MVT::v16bf16, Expand);
352 setTruncStoreAction(MVT::v16f32, MVT::v16f16, Expand);
353 setTruncStoreAction(MVT::v32f32, MVT::v32bf16, Expand);
354 setTruncStoreAction(MVT::v32f32, MVT::v32f16, Expand);
355
356 setTruncStoreAction(MVT::f64, MVT::bf16, Expand);
357 setTruncStoreAction(MVT::f64, MVT::f16, Expand);
358 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
359
360 setTruncStoreAction(MVT::v2f64, MVT::v2f32, Expand);
361 setTruncStoreAction(MVT::v2f64, MVT::v2bf16, Expand);
362 setTruncStoreAction(MVT::v2f64, MVT::v2f16, Expand);
363
364 setTruncStoreAction(MVT::v3i32, MVT::v3i8, Expand);
365
366 setTruncStoreAction(MVT::v3i64, MVT::v3i32, Expand);
367 setTruncStoreAction(MVT::v3i64, MVT::v3i16, Expand);
368 setTruncStoreAction(MVT::v3i64, MVT::v3i8, Expand);
369 setTruncStoreAction(MVT::v3i64, MVT::v3i1, Expand);
370 setTruncStoreAction(MVT::v3f64, MVT::v3f32, Expand);
371 setTruncStoreAction(MVT::v3f64, MVT::v3bf16, Expand);
372 setTruncStoreAction(MVT::v3f64, MVT::v3f16, Expand);
373
374 setTruncStoreAction(MVT::v4i64, MVT::v4i32, Expand);
375 setTruncStoreAction(MVT::v4i64, MVT::v4i16, Expand);
376 setTruncStoreAction(MVT::v4f64, MVT::v4f32, Expand);
377 setTruncStoreAction(MVT::v4f64, MVT::v4bf16, Expand);
378 setTruncStoreAction(MVT::v4f64, MVT::v4f16, Expand);
379
380 setTruncStoreAction(MVT::v5i32, MVT::v5i1, Expand);
381 setTruncStoreAction(MVT::v5i32, MVT::v5i8, Expand);
382 setTruncStoreAction(MVT::v5i32, MVT::v5i16, Expand);
383
384 setTruncStoreAction(MVT::v6i32, MVT::v6i1, Expand);
385 setTruncStoreAction(MVT::v6i32, MVT::v6i8, Expand);
386 setTruncStoreAction(MVT::v6i32, MVT::v6i16, Expand);
387
388 setTruncStoreAction(MVT::v7i32, MVT::v7i1, Expand);
389 setTruncStoreAction(MVT::v7i32, MVT::v7i8, Expand);
390 setTruncStoreAction(MVT::v7i32, MVT::v7i16, Expand);
391
392 setTruncStoreAction(MVT::v8f64, MVT::v8f32, Expand);
393 setTruncStoreAction(MVT::v8f64, MVT::v8bf16, Expand);
394 setTruncStoreAction(MVT::v8f64, MVT::v8f16, Expand);
395
396 setTruncStoreAction(MVT::v16f64, MVT::v16f32, Expand);
397 setTruncStoreAction(MVT::v16f64, MVT::v16bf16, Expand);
398 setTruncStoreAction(MVT::v16f64, MVT::v16f16, Expand);
399 setTruncStoreAction(MVT::v16i64, MVT::v16i16, Expand);
400 setTruncStoreAction(MVT::v16i64, MVT::v16i8, Expand);
401 setTruncStoreAction(MVT::v16i64, MVT::v16i8, Expand);
402 setTruncStoreAction(MVT::v16i64, MVT::v16i1, Expand);
403
404 setOperationAction(ISD::Constant, {MVT::i32, MVT::i64}, Legal);
405 setOperationAction(ISD::ConstantFP, {MVT::f32, MVT::f64}, Legal);
406
408
409 // For R600, this is totally unsupported, just custom lower to produce an
410 // error.
412
413 // Library functions. These default to Expand, but we have instructions
414 // for them.
417 {MVT::f16, MVT::f32}, Legal);
419
421 setOperationAction(ISD::FROUND, {MVT::f32, MVT::f64}, Custom);
423 {MVT::f16, MVT::f32, MVT::f64}, Expand);
424
427 Custom);
429
430 setOperationAction(ISD::FNEARBYINT, {MVT::f16, MVT::f32, MVT::f64}, Custom);
431
432 setOperationAction(ISD::FRINT, {MVT::f16, MVT::f32, MVT::f64}, Custom);
433
434 setOperationAction({ISD::LRINT, ISD::LLRINT}, {MVT::f16, MVT::f32, MVT::f64},
435 Expand);
436
437 setOperationAction(ISD::FREM, {MVT::f16, MVT::f32, MVT::f64}, Expand);
438 setOperationAction(ISD::IS_FPCLASS, {MVT::f32, MVT::f64}, Legal);
440
442 Custom);
443
444 setOperationAction(ISD::FCANONICALIZE, {MVT::f32, MVT::f64}, Legal);
445
446 // FIXME: These IS_FPCLASS vector fp types are marked custom so it reaches
447 // scalarization code. Can be removed when IS_FPCLASS expand isn't called by
448 // default unless marked custom/legal.
450 {MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32,
451 MVT::v6f32, MVT::v7f32, MVT::v8f32, MVT::v16f32,
452 MVT::v2f64, MVT::v3f64, MVT::v4f64, MVT::v8f64,
453 MVT::v16f64},
454 Custom);
455
456 // Expand to fneg + fadd.
458
460 {MVT::v3i32, MVT::v3f32, MVT::v4i32, MVT::v4f32,
461 MVT::v5i32, MVT::v5f32, MVT::v6i32, MVT::v6f32,
462 MVT::v7i32, MVT::v7f32, MVT::v8i32, MVT::v8f32,
463 MVT::v9i32, MVT::v9f32, MVT::v10i32, MVT::v10f32,
464 MVT::v11i32, MVT::v11f32, MVT::v12i32, MVT::v12f32},
465 Custom);
466
469 {MVT::v2f32, MVT::v2i32, MVT::v3f32, MVT::v3i32, MVT::v4f32,
470 MVT::v4i32, MVT::v5f32, MVT::v5i32, MVT::v6f32, MVT::v6i32,
471 MVT::v7f32, MVT::v7i32, MVT::v8f32, MVT::v8i32, MVT::v9f32,
472 MVT::v9i32, MVT::v10i32, MVT::v10f32, MVT::v11i32, MVT::v11f32,
473 MVT::v12i32, MVT::v12f32, MVT::v16i32, MVT::v32f32, MVT::v32i32,
474 MVT::v2f64, MVT::v2i64, MVT::v3f64, MVT::v3i64, MVT::v4f64,
475 MVT::v4i64, MVT::v8f64, MVT::v8i64, MVT::v16f64, MVT::v16i64},
476 Custom);
477
479 Expand);
480 setOperationAction(ISD::FP_TO_FP16, {MVT::f64, MVT::f32}, Custom);
481
482 const MVT ScalarIntVTs[] = { MVT::i32, MVT::i64 };
483 for (MVT VT : ScalarIntVTs) {
484 // These should use [SU]DIVREM, so set them to expand
486 Expand);
487
488 // GPU does not have divrem function for signed or unsigned.
490
491 // GPU does not have [S|U]MUL_LOHI functions as a single instruction.
493
495
497 Expand);
498 }
499
500 // The hardware supports 32-bit FSHR, but not FSHL.
502
503 setOperationAction({ISD::ROTL, ISD::ROTR}, {MVT::i32, MVT::i64}, Expand);
504
506
511 MVT::i64, Custom);
513
515 Legal);
516
519 MVT::i64, Custom);
520
521 for (auto VT : {MVT::i8, MVT::i16})
523
524 static const MVT::SimpleValueType VectorIntTypes[] = {
525 MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32, MVT::v6i32, MVT::v7i32,
526 MVT::v9i32, MVT::v10i32, MVT::v11i32, MVT::v12i32};
527
528 for (MVT VT : VectorIntTypes) {
529 // Expand the following operations for the current type by default.
530 // clang-format off
550 VT, Expand);
551 // clang-format on
552 }
553
554 static const MVT::SimpleValueType FloatVectorTypes[] = {
555 MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32, MVT::v6f32, MVT::v7f32,
556 MVT::v9f32, MVT::v10f32, MVT::v11f32, MVT::v12f32};
557
558 for (MVT VT : FloatVectorTypes) {
571 VT, Expand);
572 }
573
574 // This causes using an unrolled select operation rather than expansion with
575 // bit operations. This is in general better, but the alternative using BFI
576 // instructions may be better if the select sources are SGPRs.
578 AddPromotedToType(ISD::SELECT, MVT::v2f32, MVT::v2i32);
579
581 AddPromotedToType(ISD::SELECT, MVT::v3f32, MVT::v3i32);
582
584 AddPromotedToType(ISD::SELECT, MVT::v4f32, MVT::v4i32);
585
587 AddPromotedToType(ISD::SELECT, MVT::v5f32, MVT::v5i32);
588
590 AddPromotedToType(ISD::SELECT, MVT::v6f32, MVT::v6i32);
591
593 AddPromotedToType(ISD::SELECT, MVT::v7f32, MVT::v7i32);
594
596 AddPromotedToType(ISD::SELECT, MVT::v9f32, MVT::v9i32);
597
599 AddPromotedToType(ISD::SELECT, MVT::v10f32, MVT::v10i32);
600
602 AddPromotedToType(ISD::SELECT, MVT::v11f32, MVT::v11i32);
603
605 AddPromotedToType(ISD::SELECT, MVT::v12f32, MVT::v12i32);
606
608 setJumpIsExpensive(true);
609
612
614
615 // We want to find all load dependencies for long chains of stores to enable
616 // merging into very wide vectors. The problem is with vectors with > 4
617 // elements. MergeConsecutiveStores will attempt to merge these because x8/x16
618 // vectors are a legal type, even though we have to split the loads
619 // usually. When we can more precisely specify load legality per address
620 // space, we should be able to make FindBetterChain/MergeConsecutiveStores
621 // smarter so that they can figure out what to do in 2 iterations without all
622 // N > 4 stores on the same chain.
624
625 // memcpy/memmove/memset are expanded in the IR, so we shouldn't need to worry
626 // about these during lowering.
627 MaxStoresPerMemcpy = 0xffffffff;
628 MaxStoresPerMemmove = 0xffffffff;
629 MaxStoresPerMemset = 0xffffffff;
630
631 // The expansion for 64-bit division is enormous.
633 addBypassSlowDiv(64, 32);
634
645
649}
650
651//===----------------------------------------------------------------------===//
652// Target Information
653//===----------------------------------------------------------------------===//
654
656static bool fnegFoldsIntoOpcode(unsigned Opc) {
657 switch (Opc) {
658 case ISD::FADD:
659 case ISD::FSUB:
660 case ISD::FMUL:
661 case ISD::FMA:
662 case ISD::FMAD:
663 case ISD::FMINNUM:
664 case ISD::FMAXNUM:
667 case ISD::FMINIMUM:
668 case ISD::FMAXIMUM:
669 case ISD::FMINIMUMNUM:
670 case ISD::FMAXIMUMNUM:
671 case ISD::SELECT:
672 case ISD::FSIN:
673 case ISD::FTRUNC:
674 case ISD::FRINT:
675 case ISD::FNEARBYINT:
676 case ISD::FROUNDEVEN:
678 case AMDGPUISD::RCP:
679 case AMDGPUISD::RCP_LEGACY:
680 case AMDGPUISD::RCP_IFLAG:
681 case AMDGPUISD::SIN_HW:
682 case AMDGPUISD::FMUL_LEGACY:
683 case AMDGPUISD::FMIN_LEGACY:
684 case AMDGPUISD::FMAX_LEGACY:
685 case AMDGPUISD::FMED3:
686 // TODO: handle llvm.amdgcn.fma.legacy
687 return true;
688 case ISD::BITCAST:
689 llvm_unreachable("bitcast is special cased");
690 default:
691 return false;
692 }
693}
694
695static bool fnegFoldsIntoOp(const SDNode *N) {
696 unsigned Opc = N->getOpcode();
697 if (Opc == ISD::BITCAST) {
698 // TODO: Is there a benefit to checking the conditions performFNegCombine
699 // does? We don't for the other cases.
700 SDValue BCSrc = N->getOperand(0);
701 if (BCSrc.getOpcode() == ISD::BUILD_VECTOR) {
702 return BCSrc.getNumOperands() == 2 &&
703 BCSrc.getOperand(1).getValueSizeInBits() == 32;
704 }
705
706 return BCSrc.getOpcode() == ISD::SELECT && BCSrc.getValueType() == MVT::f32;
707 }
708
709 return fnegFoldsIntoOpcode(Opc);
710}
711
712/// \p returns true if the operation will definitely need to use a 64-bit
713/// encoding, and thus will use a VOP3 encoding regardless of the source
714/// modifiers.
716static bool opMustUseVOP3Encoding(const SDNode *N, MVT VT) {
717 return (N->getNumOperands() > 2 && N->getOpcode() != ISD::SELECT) ||
718 VT == MVT::f64;
719}
720
721/// Return true if v_cndmask_b32 will support fabs/fneg source modifiers for the
722/// type for ISD::SELECT.
724static bool selectSupportsSourceMods(const SDNode *N) {
725 // TODO: Only applies if select will be vector
726 return N->getValueType(0) == MVT::f32;
727}
728
729// Most FP instructions support source modifiers, but this could be refined
730// slightly.
732static bool hasSourceMods(const SDNode *N) {
733 if (isa<MemSDNode>(N))
734 return false;
735
736 switch (N->getOpcode()) {
737 case ISD::CopyToReg:
738 case ISD::FDIV:
739 case ISD::FREM:
740 case ISD::INLINEASM:
742 case AMDGPUISD::DIV_SCALE:
744
745 // TODO: Should really be looking at the users of the bitcast. These are
746 // problematic because bitcasts are used to legalize all stores to integer
747 // types.
748 case ISD::BITCAST:
749 return false;
751 switch (N->getConstantOperandVal(0)) {
752 case Intrinsic::amdgcn_interp_p1:
753 case Intrinsic::amdgcn_interp_p2:
754 case Intrinsic::amdgcn_interp_mov:
755 case Intrinsic::amdgcn_interp_p1_f16:
756 case Intrinsic::amdgcn_interp_p2_f16:
757 return false;
758 default:
759 return true;
760 }
761 }
762 case ISD::SELECT:
764 default:
765 return true;
766 }
767}
768
770 unsigned CostThreshold) {
771 // Some users (such as 3-operand FMA/MAD) must use a VOP3 encoding, and thus
772 // it is truly free to use a source modifier in all cases. If there are
773 // multiple users but for each one will necessitate using VOP3, there will be
774 // a code size increase. Try to avoid increasing code size unless we know it
775 // will save on the instruction count.
776 unsigned NumMayIncreaseSize = 0;
777 MVT VT = N->getValueType(0).getScalarType().getSimpleVT();
778
779 assert(!N->use_empty());
780
781 // XXX - Should this limit number of uses to check?
782 for (const SDNode *U : N->users()) {
783 if (!hasSourceMods(U))
784 return false;
785
786 if (!opMustUseVOP3Encoding(U, VT)) {
787 if (++NumMayIncreaseSize > CostThreshold)
788 return false;
789 }
790 }
791
792 return true;
793}
794
796 ISD::NodeType ExtendKind) const {
797 assert(!VT.isVector() && "only scalar expected");
798
799 // Round to the next multiple of 32-bits.
800 unsigned Size = VT.getSizeInBits();
801 if (Size <= 32)
802 return MVT::i32;
803 return EVT::getIntegerVT(Context, 32 * ((Size + 31) / 32));
804}
805
807 return 32;
808}
809
811 return true;
812}
813
814// The backend supports 32 and 64 bit floating point immediates.
815// FIXME: Why are we reporting vectors of FP immediates as legal?
817 bool ForCodeSize) const {
818 return isTypeLegal(VT.getScalarType());
819}
820
821// We don't want to shrink f64 / f32 constants.
823 EVT ScalarVT = VT.getScalarType();
824 return (ScalarVT != MVT::f32 && ScalarVT != MVT::f64);
825}
826
828 SDNode *N, ISD::LoadExtType ExtTy, EVT NewVT,
829 std::optional<unsigned> ByteOffset) const {
830 // TODO: This may be worth removing. Check regression tests for diffs.
831 if (!TargetLoweringBase::shouldReduceLoadWidth(N, ExtTy, NewVT, ByteOffset))
832 return false;
833
834 unsigned NewSize = NewVT.getStoreSizeInBits();
835
836 // If we are reducing to a 32-bit load or a smaller multi-dword load,
837 // this is always better.
838 if (NewSize >= 32)
839 return true;
840
841 EVT OldVT = N->getValueType(0);
842 unsigned OldSize = OldVT.getStoreSizeInBits();
843
845 unsigned AS = MN->getAddressSpace();
846 // Do not shrink an aligned scalar load to sub-dword.
847 // Scalar engine cannot do sub-dword loads.
848 // TODO: Update this for GFX12 which does have scalar sub-dword loads.
849 if (OldSize >= 32 && NewSize < 32 && MN->getAlign() >= Align(4) &&
853 MN->isInvariant())) &&
855 return false;
856
857 // Don't produce extloads from sub 32-bit types. SI doesn't have scalar
858 // extloads, so doing one requires using a buffer_load. In cases where we
859 // still couldn't use a scalar load, using the wider load shouldn't really
860 // hurt anything.
861
862 // If the old size already had to be an extload, there's no harm in continuing
863 // to reduce the width.
864 return (OldSize < 32);
865}
866
868 const SelectionDAG &DAG,
869 const MachineMemOperand &MMO) const {
870
871 assert(LoadTy.getSizeInBits() == CastTy.getSizeInBits());
872
873 if (LoadTy.getScalarType() == MVT::i32)
874 return false;
875
876 unsigned LScalarSize = LoadTy.getScalarSizeInBits();
877 unsigned CastScalarSize = CastTy.getScalarSizeInBits();
878
879 if ((LScalarSize >= CastScalarSize) && (CastScalarSize < 32))
880 return false;
881
882 unsigned Fast = 0;
884 CastTy, MMO, &Fast) &&
885 Fast;
886}
887
888// SI+ has instructions for cttz / ctlz for 32-bit values. This is probably also
889// profitable with the expansion for 64-bit since it's generally good to
890// speculate things.
892 return true;
893}
894
896 return true;
897}
898
900 switch (N->getOpcode()) {
901 case ISD::EntryToken:
902 case ISD::TokenFactor:
903 return true;
905 unsigned IntrID = N->getConstantOperandVal(0);
907 }
909 unsigned IntrID = N->getConstantOperandVal(1);
911 }
912 case ISD::LOAD:
913 if (cast<LoadSDNode>(N)->getMemOperand()->getAddrSpace() ==
915 return true;
916 return false;
917 case AMDGPUISD::SETCC: // ballot-style instruction
918 return true;
919 }
920 return false;
921}
922
924 SDValue Op, SelectionDAG &DAG, bool LegalOperations, bool ForCodeSize,
925 NegatibleCost &Cost, unsigned Depth) const {
926
927 switch (Op.getOpcode()) {
928 case ISD::FMA:
929 case ISD::FMAD: {
930 // Negating a fma is not free if it has users without source mods.
931 if (!allUsesHaveSourceMods(Op.getNode()))
932 return SDValue();
933 break;
934 }
935 case AMDGPUISD::RCP: {
936 SDValue Src = Op.getOperand(0);
937 EVT VT = Op.getValueType();
938 SDLoc SL(Op);
939
940 SDValue NegSrc = getNegatedExpression(Src, DAG, LegalOperations,
941 ForCodeSize, Cost, Depth + 1);
942 if (NegSrc)
943 return DAG.getNode(AMDGPUISD::RCP, SL, VT, NegSrc, Op->getFlags());
944 return SDValue();
945 }
946 default:
947 break;
948 }
949
950 return TargetLowering::getNegatedExpression(Op, DAG, LegalOperations,
951 ForCodeSize, Cost, Depth);
952}
953
954//===---------------------------------------------------------------------===//
955// Target Properties
956//===---------------------------------------------------------------------===//
957
960
961 // Packed operations do not have a fabs modifier.
962 // Report this based on the end legalized type.
963 return VT == MVT::f32 || VT == MVT::f64 || VT == MVT::f16 || VT == MVT::bf16;
964}
965
968 // Report this based on the end legalized type.
969 VT = VT.getScalarType();
970 return VT == MVT::f32 || VT == MVT::f64 || VT == MVT::f16 || VT == MVT::bf16;
971}
972
974 unsigned NumElem,
975 unsigned AS) const {
976 return true;
977}
978
980 // There are few operations which truly have vector input operands. Any vector
981 // operation is going to involve operations on each component, and a
982 // build_vector will be a copy per element, so it always makes sense to use a
983 // build_vector input in place of the extracted element to avoid a copy into a
984 // super register.
985 //
986 // We should probably only do this if all users are extracts only, but this
987 // should be the common case.
988 return true;
989}
990
992 // Truncate is just accessing a subregister.
993
994 unsigned SrcSize = Source.getSizeInBits();
995 unsigned DestSize = Dest.getSizeInBits();
996
997 return DestSize < SrcSize && DestSize % 32 == 0 ;
998}
999
1001 // Truncate is just accessing a subregister.
1002
1003 unsigned SrcSize = Source->getScalarSizeInBits();
1004 unsigned DestSize = Dest->getScalarSizeInBits();
1005
1006 if (DestSize== 16 && Subtarget->has16BitInsts())
1007 return SrcSize >= 32;
1008
1009 return DestSize < SrcSize && DestSize % 32 == 0;
1010}
1011
1013 unsigned SrcSize = Src->getScalarSizeInBits();
1014 unsigned DestSize = Dest->getScalarSizeInBits();
1015
1016 if (SrcSize == 16 && Subtarget->has16BitInsts())
1017 return DestSize >= 32;
1018
1019 return SrcSize == 32 && DestSize == 64;
1020}
1021
1023 // Any register load of a 64-bit value really requires 2 32-bit moves. For all
1024 // practical purposes, the extra mov 0 to load a 64-bit is free. As used,
1025 // this will enable reducing 64-bit operations the 32-bit, which is always
1026 // good.
1027
1028 if (Src == MVT::i16)
1029 return Dest == MVT::i32 ||Dest == MVT::i64 ;
1030
1031 return Src == MVT::i32 && Dest == MVT::i64;
1032}
1033
1035 EVT DestVT) const {
1036 switch (N->getOpcode()) {
1037 case ISD::ABS:
1038 case ISD::ADD:
1039 case ISD::SUB:
1040 case ISD::SHL:
1041 case ISD::SRL:
1042 case ISD::SRA:
1043 case ISD::AND:
1044 case ISD::OR:
1045 case ISD::XOR:
1046 case ISD::MUL:
1047 case ISD::SETCC:
1048 case ISD::SELECT:
1049 case ISD::SMIN:
1050 case ISD::SMAX:
1051 case ISD::UMIN:
1052 case ISD::UMAX:
1053 case ISD::USUBSAT:
1054 case ISD::UADDSAT:
1055 if (isTypeLegal(MVT::i16) &&
1056 (!DestVT.isVector() ||
1057 !isOperationLegal(ISD::ADD, MVT::v2i16))) { // Check if VOP3P
1058 // Don't narrow back down to i16 if promoted to i32 already.
1059 if (!N->isDivergent() && DestVT.isInteger() &&
1060 DestVT.getScalarSizeInBits() > 1 &&
1061 DestVT.getScalarSizeInBits() <= 16 &&
1062 SrcVT.getScalarSizeInBits() > 16) {
1063 return false;
1064 }
1065 }
1066 return true;
1067 default:
1068 break;
1069 }
1070
1071 // There aren't really 64-bit registers, but pairs of 32-bit ones and only a
1072 // limited number of native 64-bit operations. Shrinking an operation to fit
1073 // in a single 32-bit register should always be helpful. As currently used,
1074 // this is much less general than the name suggests, and is only used in
1075 // places trying to reduce the sizes of loads. Shrinking loads to < 32-bits is
1076 // not profitable, and may actually be harmful.
1077 if (isa<LoadSDNode>(N))
1078 return SrcVT.getSizeInBits() > 32 && DestVT.getSizeInBits() == 32;
1079
1080 return true;
1081}
1082
1084 const SDNode* N, CombineLevel Level) const {
1085 assert((N->getOpcode() == ISD::SHL || N->getOpcode() == ISD::SRA ||
1086 N->getOpcode() == ISD::SRL) &&
1087 "Expected shift op");
1088
1089 SDValue ShiftLHS = N->getOperand(0);
1090 if (!ShiftLHS->hasOneUse())
1091 return false;
1092
1093 if (ShiftLHS.getOpcode() == ISD::SIGN_EXTEND &&
1094 !ShiftLHS.getOperand(0)->hasOneUse())
1095 return false;
1096
1097 // Always commute pre-type legalization and right shifts.
1098 // We're looking for shl(or(x,y),z) patterns.
1100 N->getOpcode() != ISD::SHL || N->getOperand(0).getOpcode() != ISD::OR)
1101 return true;
1102
1103 // If only user is a i32 right-shift, then don't destroy a BFE pattern.
1104 if (N->getValueType(0) == MVT::i32 && N->hasOneUse() &&
1105 (N->user_begin()->getOpcode() == ISD::SRA ||
1106 N->user_begin()->getOpcode() == ISD::SRL))
1107 return false;
1108
1109 // Don't destroy or(shl(load_zext(),c), load_zext()) patterns.
1110 auto IsShiftAndLoad = [](SDValue LHS, SDValue RHS) {
1111 if (LHS.getOpcode() != ISD::SHL)
1112 return false;
1113 auto *RHSLd = dyn_cast<LoadSDNode>(RHS);
1114 auto *LHS0 = dyn_cast<LoadSDNode>(LHS.getOperand(0));
1115 auto *LHS1 = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
1116 return LHS0 && LHS1 && RHSLd && LHS0->getExtensionType() == ISD::ZEXTLOAD &&
1117 LHS1->getAPIntValue() == LHS0->getMemoryVT().getScalarSizeInBits() &&
1118 RHSLd->getExtensionType() == ISD::ZEXTLOAD;
1119 };
1120 SDValue LHS = N->getOperand(0).getOperand(0);
1121 SDValue RHS = N->getOperand(0).getOperand(1);
1122 return !(IsShiftAndLoad(LHS, RHS) || IsShiftAndLoad(RHS, LHS));
1123}
1124
1125//===---------------------------------------------------------------------===//
1126// TargetLowering Callbacks
1127//===---------------------------------------------------------------------===//
1128
1130 bool IsVarArg) {
1131 switch (CC) {
1139 return CC_AMDGPU;
1142 return CC_AMDGPU_CS_CHAIN;
1143 case CallingConv::C:
1144 case CallingConv::Fast:
1145 case CallingConv::Cold:
1146 return CC_AMDGPU_Func;
1149 return CC_SI_Gfx;
1152 default:
1153 reportFatalUsageError("unsupported calling convention for call");
1154 }
1155}
1156
1158 bool IsVarArg) {
1159 switch (CC) {
1162 llvm_unreachable("kernels should not be handled here");
1172 return RetCC_SI_Shader;
1175 return RetCC_SI_Gfx;
1176 case CallingConv::C:
1177 case CallingConv::Fast:
1178 case CallingConv::Cold:
1179 return RetCC_AMDGPU_Func;
1180 default:
1181 reportFatalUsageError("unsupported calling convention");
1182 }
1183}
1184
1185/// The SelectionDAGBuilder will automatically promote function arguments
1186/// with illegal types. However, this does not work for the AMDGPU targets
1187/// since the function arguments are stored in memory as these illegal types.
1188/// In order to handle this properly we need to get the original types sizes
1189/// from the LLVM IR Function and fixup the ISD:InputArg values before
1190/// passing them to AnalyzeFormalArguments()
1191
1192/// When the SelectionDAGBuilder computes the Ins, it takes care of splitting
1193/// input values across multiple registers. Each item in the Ins array
1194/// represents a single value that will be stored in registers. Ins[x].VT is
1195/// the value type of the value that will be stored in the register, so
1196/// whatever SDNode we lower the argument to needs to be this type.
1197///
1198/// In order to correctly lower the arguments we need to know the size of each
1199/// argument. Since Ins[x].VT gives us the size of the register that will
1200/// hold the value, we need to look at Ins[x].ArgVT to see the 'real' type
1201/// for the original function argument so that we can deduce the correct memory
1202/// type to use for Ins[x]. In most cases the correct memory type will be
1203/// Ins[x].ArgVT. However, this will not always be the case. If, for example,
1204/// we have a kernel argument of type v8i8, this argument will be split into
1205/// 8 parts and each part will be represented by its own item in the Ins array.
1206/// For each part the Ins[x].ArgVT will be the v8i8, which is the full type of
1207/// the argument before it was split. From this, we deduce that the memory type
1208/// for each individual part is i8. We pass the memory type as LocVT to the
1209/// calling convention analysis function and the register type (Ins[x].VT) as
1210/// the ValVT.
1212 CCState &State,
1213 const SmallVectorImpl<ISD::InputArg> &Ins) const {
1214 const MachineFunction &MF = State.getMachineFunction();
1215 const Function &Fn = MF.getFunction();
1216 LLVMContext &Ctx = Fn.getContext();
1217 const unsigned ExplicitOffset = Subtarget->getExplicitKernelArgOffset();
1219
1220 Align MaxAlign = Align(1);
1221 uint64_t ExplicitArgOffset = 0;
1222 const DataLayout &DL = Fn.getDataLayout();
1223
1224 unsigned InIndex = 0;
1225
1226 for (const Argument &Arg : Fn.args()) {
1227 const bool IsByRef = Arg.hasByRefAttr();
1228 Type *BaseArgTy = Arg.getType();
1229 Type *MemArgTy = IsByRef ? Arg.getParamByRefType() : BaseArgTy;
1230 Align Alignment = DL.getValueOrABITypeAlignment(
1231 IsByRef ? Arg.getParamAlign() : std::nullopt, MemArgTy);
1232 MaxAlign = std::max(Alignment, MaxAlign);
1233 uint64_t AllocSize = DL.getTypeAllocSize(MemArgTy);
1234
1235 uint64_t ArgOffset = alignTo(ExplicitArgOffset, Alignment) + ExplicitOffset;
1236 ExplicitArgOffset = alignTo(ExplicitArgOffset, Alignment) + AllocSize;
1237
1238 // We're basically throwing away everything passed into us and starting over
1239 // to get accurate in-memory offsets. The "PartOffset" is completely useless
1240 // to us as computed in Ins.
1241 //
1242 // We also need to figure out what type legalization is trying to do to get
1243 // the correct memory offsets.
1244
1245 SmallVector<EVT, 16> ValueVTs;
1247 ComputeValueVTs(*this, DL, BaseArgTy, ValueVTs, /*MemVTs=*/nullptr,
1248 &Offsets, ArgOffset);
1249
1250 for (unsigned Value = 0, NumValues = ValueVTs.size();
1251 Value != NumValues; ++Value) {
1252 uint64_t BasePartOffset = Offsets[Value];
1253
1254 EVT ArgVT = ValueVTs[Value];
1255 EVT MemVT = ArgVT;
1256 MVT RegisterVT = getRegisterTypeForCallingConv(Ctx, CC, ArgVT);
1257 unsigned NumRegs = getNumRegistersForCallingConv(Ctx, CC, ArgVT);
1258
1259 if (NumRegs == 1) {
1260 // This argument is not split, so the IR type is the memory type.
1261 if (ArgVT.isExtended()) {
1262 // We have an extended type, like i24, so we should just use the
1263 // register type.
1264 MemVT = RegisterVT;
1265 } else {
1266 MemVT = ArgVT;
1267 }
1268 } else if (ArgVT.isVector() && RegisterVT.isVector() &&
1269 ArgVT.getScalarType() == RegisterVT.getScalarType()) {
1270 assert(ArgVT.getVectorNumElements() > RegisterVT.getVectorNumElements());
1271 // We have a vector value which has been split into a vector with
1272 // the same scalar type, but fewer elements. This should handle
1273 // all the floating-point vector types.
1274 MemVT = RegisterVT;
1275 } else if (ArgVT.isVector() &&
1276 ArgVT.getVectorNumElements() == NumRegs) {
1277 // This arg has been split so that each element is stored in a separate
1278 // register.
1279 MemVT = ArgVT.getScalarType();
1280 } else if (ArgVT.isExtended()) {
1281 // We have an extended type, like i65.
1282 MemVT = RegisterVT;
1283 } else {
1284 unsigned MemoryBits = ArgVT.getStoreSizeInBits() / NumRegs;
1285 assert(ArgVT.getStoreSizeInBits() % NumRegs == 0);
1286 if (RegisterVT.isInteger()) {
1287 MemVT = EVT::getIntegerVT(State.getContext(), MemoryBits);
1288 } else if (RegisterVT.isVector()) {
1289 assert(!RegisterVT.getScalarType().isFloatingPoint());
1290 unsigned NumElements = RegisterVT.getVectorNumElements();
1291 assert(MemoryBits % NumElements == 0);
1292 // This vector type has been split into another vector type with
1293 // a different elements size.
1294 EVT ScalarVT = EVT::getIntegerVT(State.getContext(),
1295 MemoryBits / NumElements);
1296 MemVT = EVT::getVectorVT(State.getContext(), ScalarVT, NumElements);
1297 } else {
1298 llvm_unreachable("cannot deduce memory type.");
1299 }
1300 }
1301
1302 // Convert one element vectors to scalar.
1303 if (MemVT.isVector() && MemVT.getVectorNumElements() == 1)
1304 MemVT = MemVT.getScalarType();
1305
1306 // Round up vec3/vec5 argument.
1307 if (MemVT.isVector() && !MemVT.isPow2VectorType()) {
1308 MemVT = MemVT.getPow2VectorType(State.getContext());
1309 } else if (!MemVT.isSimple() && !MemVT.isVector()) {
1310 MemVT = MemVT.getRoundIntegerType(State.getContext());
1311 }
1312
1313 unsigned PartOffset = 0;
1314 for (unsigned i = 0; i != NumRegs; ++i) {
1315 State.addLoc(CCValAssign::getCustomMem(InIndex++, RegisterVT,
1316 BasePartOffset + PartOffset,
1317 MemVT.getSimpleVT(),
1319 PartOffset += MemVT.getStoreSize();
1320 }
1321 }
1322 }
1323}
1324
1326 SDValue Chain, CallingConv::ID CallConv,
1327 bool isVarArg,
1329 const SmallVectorImpl<SDValue> &OutVals,
1330 const SDLoc &DL, SelectionDAG &DAG) const {
1331 // FIXME: Fails for r600 tests
1332 //assert(!isVarArg && Outs.empty() && OutVals.empty() &&
1333 // "wave terminate should not have return values");
1334 return DAG.getNode(AMDGPUISD::ENDPGM, DL, MVT::Other, Chain);
1335}
1336
1337//===---------------------------------------------------------------------===//
1338// Target specific lowering
1339//===---------------------------------------------------------------------===//
1340
1341/// Selects the correct CCAssignFn for a given CallingConvention value.
1346
1351
1353 SelectionDAG &DAG,
1354 MachineFrameInfo &MFI,
1355 int ClobberedFI) const {
1356 SmallVector<SDValue, 8> ArgChains;
1357 int64_t FirstByte = MFI.getObjectOffset(ClobberedFI);
1358 int64_t LastByte = FirstByte + MFI.getObjectSize(ClobberedFI) - 1;
1359
1360 // Include the original chain at the beginning of the list. When this is
1361 // used by target LowerCall hooks, this helps legalize find the
1362 // CALLSEQ_BEGIN node.
1363 ArgChains.push_back(Chain);
1364
1365 // Add a chain value for each stack argument corresponding
1366 for (SDNode *U : DAG.getEntryNode().getNode()->users()) {
1367 if (LoadSDNode *L = dyn_cast<LoadSDNode>(U)) {
1368 if (FrameIndexSDNode *FI = dyn_cast<FrameIndexSDNode>(L->getBasePtr())) {
1369 if (FI->getIndex() < 0) {
1370 int64_t InFirstByte = MFI.getObjectOffset(FI->getIndex());
1371 int64_t InLastByte = InFirstByte;
1372 InLastByte += MFI.getObjectSize(FI->getIndex()) - 1;
1373
1374 if ((InFirstByte <= FirstByte && FirstByte <= InLastByte) ||
1375 (FirstByte <= InFirstByte && InFirstByte <= LastByte))
1376 ArgChains.push_back(SDValue(L, 1));
1377 }
1378 }
1379 }
1380 }
1381
1382 // Build a tokenfactor for all the chains.
1383 return DAG.getNode(ISD::TokenFactor, SDLoc(Chain), MVT::Other, ArgChains);
1384}
1385
1388 StringRef Reason) const {
1389 SDValue Callee = CLI.Callee;
1390 SelectionDAG &DAG = CLI.DAG;
1391
1392 const Function &Fn = DAG.getMachineFunction().getFunction();
1393
1394 StringRef FuncName("<unknown>");
1395
1397 FuncName = G->getSymbol();
1398 else if (const GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee))
1399 FuncName = G->getGlobal()->getName();
1400
1401 DAG.getContext()->diagnose(
1402 DiagnosticInfoUnsupported(Fn, Reason + FuncName, CLI.DL.getDebugLoc()));
1403
1404 if (!CLI.IsTailCall) {
1405 for (ISD::InputArg &Arg : CLI.Ins)
1406 InVals.push_back(DAG.getPOISON(Arg.VT));
1407 }
1408
1409 // FIXME: Hack because R600 doesn't handle callseq pseudos yet.
1410 if (getTargetMachine().getTargetTriple().getArch() == Triple::r600)
1411 return CLI.Chain;
1412
1413 SDValue Chain = DAG.getCALLSEQ_START(CLI.Chain, 0, 0, CLI.DL);
1414 return DAG.getCALLSEQ_END(Chain, 0, 0, /*InGlue=*/SDValue(), CLI.DL);
1415}
1416
1418 SmallVectorImpl<SDValue> &InVals) const {
1419 return lowerUnhandledCall(CLI, InVals, "unsupported call to function ");
1420}
1421
1423 SelectionDAG &DAG) const {
1424 const Function &Fn = DAG.getMachineFunction().getFunction();
1425
1427 Fn, "unsupported dynamic alloca", SDLoc(Op).getDebugLoc()));
1428 auto Ops = {DAG.getConstant(0, SDLoc(), Op.getValueType()), Op.getOperand(0)};
1429 return DAG.getMergeValues(Ops, SDLoc());
1430}
1431
1433 SelectionDAG &DAG) const {
1434 switch (Op.getOpcode()) {
1435 default:
1436 Op->print(errs(), &DAG);
1437 llvm_unreachable("Custom lowering code for this "
1438 "instruction is not implemented yet!");
1439 break;
1441 case ISD::CONCAT_VECTORS: return LowerCONCAT_VECTORS(Op, DAG);
1443 case ISD::UDIVREM: return LowerUDIVREM(Op, DAG);
1444 case ISD::SDIVREM:
1445 return LowerSDIVREM(Op, DAG);
1446 case ISD::FCEIL: return LowerFCEIL(Op, DAG);
1447 case ISD::FTRUNC: return LowerFTRUNC(Op, DAG);
1448 case ISD::FRINT: return LowerFRINT(Op, DAG);
1449 case ISD::FNEARBYINT: return LowerFNEARBYINT(Op, DAG);
1450 case ISD::FROUNDEVEN:
1451 return LowerFROUNDEVEN(Op, DAG);
1452 case ISD::FROUND: return LowerFROUND(Op, DAG);
1453 case ISD::FFLOOR: return LowerFFLOOR(Op, DAG);
1454 case ISD::FLOG2:
1455 return LowerFLOG2(Op, DAG);
1456 case ISD::FLOG:
1457 case ISD::FLOG10:
1458 return LowerFLOGCommon(Op, DAG);
1459 case ISD::FEXP:
1460 case ISD::FEXP10:
1461 return lowerFEXP(Op, DAG);
1462 case ISD::FEXP2:
1463 return lowerFEXP2(Op, DAG);
1464 case ISD::SINT_TO_FP: return LowerSINT_TO_FP(Op, DAG);
1465 case ISD::UINT_TO_FP: return LowerUINT_TO_FP(Op, DAG);
1466 case ISD::FP_TO_FP16: return LowerFP_TO_FP16(Op, DAG);
1467 case ISD::FP_TO_SINT:
1468 case ISD::FP_TO_UINT:
1469 return LowerFP_TO_INT(Op, DAG);
1472 return LowerFP_TO_INT_SAT(Op, DAG);
1473 case ISD::CTTZ:
1475 case ISD::CTLZ:
1477 return LowerCTLZ_CTTZ(Op, DAG);
1478 case ISD::CTLS:
1479 return LowerCTLS(Op, DAG);
1481 }
1482 return Op;
1483}
1484
1487 SelectionDAG &DAG) const {
1488 switch (N->getOpcode()) {
1490 // Different parts of legalization seem to interpret which type of
1491 // sign_extend_inreg is the one to check for custom lowering. The extended
1492 // from type is what really matters, but some places check for custom
1493 // lowering of the result type. This results in trying to use
1494 // ReplaceNodeResults to sext_in_reg to an illegal type, so we'll just do
1495 // nothing here and let the illegal result integer be handled normally.
1496 return;
1497 case ISD::FLOG2:
1498 if (SDValue Lowered = LowerFLOG2(SDValue(N, 0), DAG))
1499 Results.push_back(Lowered);
1500 return;
1501 case ISD::FLOG:
1502 case ISD::FLOG10:
1503 if (SDValue Lowered = LowerFLOGCommon(SDValue(N, 0), DAG))
1504 Results.push_back(Lowered);
1505 return;
1506 case ISD::FEXP2:
1507 if (SDValue Lowered = lowerFEXP2(SDValue(N, 0), DAG))
1508 Results.push_back(Lowered);
1509 return;
1510 case ISD::FEXP:
1511 case ISD::FEXP10:
1512 if (SDValue Lowered = lowerFEXP(SDValue(N, 0), DAG))
1513 Results.push_back(Lowered);
1514 return;
1515 case ISD::CTLZ:
1517 if (auto Lowered = lowerCTLZResults(SDValue(N, 0u), DAG))
1518 Results.push_back(Lowered);
1519 return;
1520 default:
1521 return;
1522 }
1523}
1524
1526 SelectionDAG &DAG) const {
1528 SDLoc SL(Op);
1529 EVT VT = Op.getValueType();
1530 return DAG.getTargetBlockAddress(BA->getBlockAddress(), VT, BA->getOffset(),
1531 BA->getTargetFlags());
1532}
1533
1535 SDValue Op,
1536 SelectionDAG &DAG) const {
1537
1538 const DataLayout &DL = DAG.getDataLayout();
1540 const GlobalValue *GV = G->getGlobal();
1541
1542 if (!MFI->isModuleEntryFunction()) {
1543 bool IsNamedBarrier = AMDGPU::isNamedBarrier(*cast<GlobalVariable>(GV));
1544 std::optional<uint32_t> Address =
1546 if (!Address && IsNamedBarrier)
1547 llvm_unreachable("named barrier should have an assigned address");
1548 if (Address) {
1549 if (IsNamedBarrier) {
1550 unsigned BarCnt = cast<GlobalVariable>(GV)->getGlobalSize(DL) / 16;
1551 MFI->recordNumNamedBarriers(Address.value(), BarCnt);
1552 }
1553 // A constant byte offset (e.g. from a GEP into an array of named
1554 // barriers) folds directly into the fixed LDS address.
1555 return DAG.getConstant(*Address + G->getOffset(), SDLoc(Op),
1556 Op.getValueType());
1557 }
1558 }
1559
1560 if (G->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS ||
1561 G->getAddressSpace() == AMDGPUAS::REGION_ADDRESS) {
1562 if (!MFI->isModuleEntryFunction() &&
1563 GV->getName() != "llvm.amdgcn.module.lds" &&
1565 SDLoc DL(Op);
1566 const Function &Fn = DAG.getMachineFunction().getFunction();
1568 Fn, "local memory global used by non-kernel function",
1569 DL.getDebugLoc(), DS_Warning));
1570
1571 // We currently don't have a way to correctly allocate LDS objects that
1572 // aren't directly associated with a kernel. We do force inlining of
1573 // functions that use local objects. However, if these dead functions are
1574 // not eliminated, we don't want a compile time error. Just emit a warning
1575 // and a trap, since there should be no callable path here.
1576 SDValue Trap = DAG.getNode(ISD::TRAP, DL, MVT::Other, DAG.getEntryNode());
1577 SDValue OutputChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
1578 Trap, DAG.getRoot());
1579 DAG.setRoot(OutputChain);
1580 return DAG.getPOISON(Op.getValueType());
1581 }
1582
1583 // TODO: We could emit code to handle the initialization somewhere.
1584 // We ignore the initializer for now and legalize it to allow selection.
1585 // The initializer will anyway get errored out during assembly emission.
1586 unsigned Offset = MFI->allocateLDSGlobal(DL, *cast<GlobalVariable>(GV));
1587 // A constant byte offset (e.g. from a GEP into an array of named barriers)
1588 // folds directly into the allocated LDS address.
1589 return DAG.getConstant(Offset + G->getOffset(), SDLoc(Op),
1590 Op.getValueType());
1591 }
1592 return SDValue();
1593}
1594
1596 SelectionDAG &DAG) const {
1598 SDLoc SL(Op);
1599
1600 EVT VT = Op.getValueType();
1601 if (VT.getVectorElementType().getSizeInBits() < 32) {
1602 unsigned OpBitSize = Op.getOperand(0).getValueType().getSizeInBits();
1603 if (OpBitSize >= 32 && OpBitSize % 32 == 0) {
1604 unsigned NewNumElt = OpBitSize / 32;
1605 EVT NewEltVT = (NewNumElt == 1) ? MVT::i32
1607 MVT::i32, NewNumElt);
1608 for (const SDUse &U : Op->ops()) {
1609 SDValue In = U.get();
1610 SDValue NewIn = DAG.getNode(ISD::BITCAST, SL, NewEltVT, In);
1611 if (NewNumElt > 1)
1612 DAG.ExtractVectorElements(NewIn, Args);
1613 else
1614 Args.push_back(NewIn);
1615 }
1616
1617 EVT NewVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32,
1618 NewNumElt * Op.getNumOperands());
1619 SDValue BV = DAG.getBuildVector(NewVT, SL, Args);
1620 return DAG.getNode(ISD::BITCAST, SL, VT, BV);
1621 }
1622 }
1623
1624 for (const SDUse &U : Op->ops())
1625 DAG.ExtractVectorElements(U.get(), Args);
1626
1627 return DAG.getBuildVector(Op.getValueType(), SL, Args);
1628}
1629
1631 SelectionDAG &DAG) const {
1632 SDLoc SL(Op);
1634 unsigned Start = Op.getConstantOperandVal(1);
1635 EVT VT = Op.getValueType();
1636 EVT SrcVT = Op.getOperand(0).getValueType();
1637
1638 if (VT.getScalarSizeInBits() == 16 && Start % 2 == 0) {
1639 unsigned NumElt = VT.getVectorNumElements();
1640 unsigned NumSrcElt = SrcVT.getVectorNumElements();
1641 assert(NumElt % 2 == 0 && NumSrcElt % 2 == 0 && "expect legal types");
1642
1643 // Extract 32-bit registers at a time.
1644 EVT NewSrcVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumSrcElt / 2);
1645 EVT NewVT = NumElt == 2
1646 ? MVT::i32
1647 : EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumElt / 2);
1648 SDValue Tmp = DAG.getNode(ISD::BITCAST, SL, NewSrcVT, Op.getOperand(0));
1649
1650 DAG.ExtractVectorElements(Tmp, Args, Start / 2, NumElt / 2);
1651 if (NumElt == 2)
1652 Tmp = Args[0];
1653 else
1654 Tmp = DAG.getBuildVector(NewVT, SL, Args);
1655
1656 return DAG.getNode(ISD::BITCAST, SL, VT, Tmp);
1657 }
1658
1659 DAG.ExtractVectorElements(Op.getOperand(0), Args, Start,
1661
1662 return DAG.getBuildVector(Op.getValueType(), SL, Args);
1663}
1664
1665// TODO: Handle fabs too
1667 if (Val.getOpcode() == ISD::FNEG)
1668 return Val.getOperand(0);
1669
1670 return Val;
1671}
1672
1673// SelectionDAG twin of AMDGPUCombinerHelper::canIgnoreLegacyMinMaxTies.
1675 SDNodeFlags Flags, SDValue LHS,
1676 SDValue RHS) {
1677 return Flags.hasNoSignedZeros() || DAG.isKnownNeverLogicalZero(LHS) ||
1679}
1680
1682 const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True,
1683 SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const {
1684 SelectionDAG &DAG = DCI.DAG;
1685 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(CC)->get();
1686 assert(CCOpcode != ISD::SETCC_INVALID && "Invalid setcc condcode!");
1687
1688 switch (CCOpcode) {
1689 case ISD::SETOLE:
1690 case ISD::SETOLT:
1691 case ISD::SETLE:
1692 case ISD::SETLT:
1693 case ISD::SETOGE:
1694 case ISD::SETOGT:
1695 case ISD::SETGE:
1696 case ISD::SETGT:
1697 // Only do this after legalization to avoid interfering with other combines
1698 // which might occur.
1700 !DCI.isCalledByLegalizer())
1701 return SDValue();
1702 break;
1703 default:
1704 break;
1705 }
1706
1707 // Canonicalize so the select returns the compare's LHS on a true predicate.
1708 if (LHS != True)
1709 CCOpcode = ISD::getSetCCInverse(CCOpcode, VT);
1710
1711 unsigned Opc;
1712 bool Swap; // Emit (rhs, lhs) instead of (lhs, rhs).
1713 switch (CCOpcode) {
1714 case ISD::SETOLT:
1715 case ISD::SETLT:
1716 case ISD::SETOLE:
1717 Opc = AMDGPUISD::FMIN_LEGACY;
1718 Swap = false;
1719 break;
1720 case ISD::SETULE:
1721 case ISD::SETLE:
1722 case ISD::SETULT:
1723 Opc = AMDGPUISD::FMIN_LEGACY;
1724 Swap = true;
1725 break;
1726 case ISD::SETOGE:
1727 case ISD::SETGE:
1728 case ISD::SETOGT:
1729 Opc = AMDGPUISD::FMAX_LEGACY;
1730 Swap = false;
1731 break;
1732 case ISD::SETUGT:
1733 case ISD::SETGT:
1734 case ISD::SETUGE:
1735 Opc = AMDGPUISD::FMAX_LEGACY;
1736 Swap = true;
1737 break;
1738 default:
1739 return SDValue();
1740 }
1741
1742 // For these predicates the NaN-correct operand order is the signed zero
1743 // tie-incorrect one, so the fold needs the tie to be unobservable.
1744 if ((CCOpcode == ISD::SETOLE || CCOpcode == ISD::SETULT ||
1745 CCOpcode == ISD::SETOGT || CCOpcode == ISD::SETUGE) &&
1746 !canIgnoreLegacyMinMaxTies(DAG, Flags, LHS, RHS))
1747 return SDValue();
1748
1749 if (Swap)
1750 std::swap(LHS, RHS);
1751 return DAG.getNode(Opc, DL, VT, LHS, RHS, Flags);
1752}
1753
1754/// Generate Min/Max node
1756 const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True,
1757 SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const {
1758 if ((LHS == True && RHS == False) || (LHS == False && RHS == True))
1759 return combineFMinMaxLegacyImpl(DL, VT, LHS, RHS, True, False, CC, Flags,
1760 DCI);
1761
1762 SelectionDAG &DAG = DCI.DAG;
1763
1764 // If we can't directly match this, try to see if we can fold an fneg to
1765 // match.
1766
1769 SDValue NegTrue = peekFNeg(True);
1770
1771 // Undo the combine foldFreeOpFromSelect does if it helps us match the
1772 // fmin/fmax.
1773 //
1774 // select (fcmp olt (lhs, K)), (fneg lhs), -K
1775 // -> fneg (fmin_legacy lhs, K)
1776 //
1777 // TODO: Use getNegatedExpression
1778 if (LHS == NegTrue && CFalse && CRHS) {
1779 APFloat NegRHS = neg(CRHS->getValueAPF());
1780 if (NegRHS == CFalse->getValueAPF()) {
1781 SDValue Combined = combineFMinMaxLegacyImpl(DL, VT, LHS, RHS, NegTrue,
1782 False, CC, Flags, DCI);
1783 if (Combined)
1784 return DAG.getNode(ISD::FNEG, DL, VT, Combined);
1785 return SDValue();
1786 }
1787 }
1788
1789 return SDValue();
1790}
1791
1792std::pair<SDValue, SDValue>
1794 SDLoc SL(Op);
1795
1796 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1797
1798 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
1799 const SDValue One = DAG.getConstant(1, SL, MVT::i32);
1800
1801 SDValue Lo = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, Zero);
1802 SDValue Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, One);
1803
1804 return std::pair(Lo, Hi);
1805}
1806
1808 SDLoc SL(Op);
1809
1810 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1811 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
1812 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, Zero);
1813}
1814
1816 SDLoc SL(Op);
1817
1818 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1819 const SDValue One = DAG.getConstant(1, SL, MVT::i32);
1820 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, One);
1821}
1822
1823// Split a vector type into two parts. The first part is a power of two vector.
1824// The second part is whatever is left over, and is a scalar if it would
1825// otherwise be a 1-vector.
1826std::pair<EVT, EVT>
1828 EVT LoVT, HiVT;
1829 EVT EltVT = VT.getVectorElementType();
1830 unsigned NumElts = VT.getVectorNumElements();
1831 unsigned LoNumElts = PowerOf2Ceil((NumElts + 1) / 2);
1832 LoVT = EVT::getVectorVT(*DAG.getContext(), EltVT, LoNumElts);
1833 HiVT = NumElts - LoNumElts == 1
1834 ? EltVT
1835 : EVT::getVectorVT(*DAG.getContext(), EltVT, NumElts - LoNumElts);
1836 return std::pair(LoVT, HiVT);
1837}
1838
1839// Split a vector value into two parts of types LoVT and HiVT. HiVT could be
1840// scalar.
1841std::pair<SDValue, SDValue>
1843 const EVT &LoVT, const EVT &HiVT,
1844 SelectionDAG &DAG) const {
1845 EVT VT = N.getValueType();
1847 (HiVT.isVector() ? HiVT.getVectorNumElements() : 1) <=
1848 VT.getVectorNumElements() &&
1849 "More vector elements requested than available!");
1851 DAG.getVectorIdxConstant(0, DL));
1852
1853 unsigned LoNumElts = LoVT.getVectorNumElements();
1854
1855 if (HiVT.isVector()) {
1856 unsigned HiNumElts = HiVT.getVectorNumElements();
1857 if ((VT.getVectorNumElements() % HiNumElts) == 0) {
1858 // Avoid creating an extract_subvector with an index that isn't a multiple
1859 // of the result type.
1861 DAG.getConstant(LoNumElts, DL, MVT::i32));
1862 return {Lo, Hi};
1863 }
1864
1866 DAG.ExtractVectorElements(N, Elts, /*Start=*/LoNumElts,
1867 /*Count=*/HiNumElts);
1868 SDValue Hi = DAG.getBuildVector(HiVT, DL, Elts);
1869 return {Lo, Hi};
1870 }
1871
1873 DAG.getVectorIdxConstant(LoNumElts, DL));
1874 return {Lo, Hi};
1875}
1876
1878 SelectionDAG &DAG) const {
1880 EVT VT = Op.getValueType();
1881 SDLoc SL(Op);
1882
1883
1884 // If this is a 2 element vector, we really want to scalarize and not create
1885 // weird 1 element vectors.
1886 if (VT.getVectorNumElements() == 2) {
1887 SDValue Ops[2];
1888 std::tie(Ops[0], Ops[1]) = scalarizeVectorLoad(Load, DAG);
1889 return DAG.getMergeValues(Ops, SL);
1890 }
1891
1892 SDValue BasePtr = Load->getBasePtr();
1893 EVT MemVT = Load->getMemoryVT();
1894
1895 const MachinePointerInfo &SrcValue = Load->getMemOperand()->getPointerInfo();
1896
1897 EVT LoVT, HiVT;
1898 EVT LoMemVT, HiMemVT;
1899 SDValue Lo, Hi;
1900
1901 std::tie(LoVT, HiVT) = getSplitDestVTs(VT, DAG);
1902 std::tie(LoMemVT, HiMemVT) = getSplitDestVTs(MemVT, DAG);
1903 std::tie(Lo, Hi) = splitVector(Op, SL, LoVT, HiVT, DAG);
1904
1905 unsigned Size = LoMemVT.getStoreSize();
1906 Align BaseAlign = Load->getAlign();
1907 Align HiAlign = commonAlignment(BaseAlign, Size);
1908
1909 SDValue LoLoad = DAG.getExtLoad(
1910 Load->getExtensionType(), SL, LoVT, Load->getChain(), BasePtr, SrcValue,
1911 LoMemVT, BaseAlign, Load->getMemOperand()->getFlags(), Load->getAAInfo());
1912 SDValue HiPtr = DAG.getObjectPtrOffset(SL, BasePtr, TypeSize::getFixed(Size));
1913 SDValue HiLoad = DAG.getExtLoad(
1914 Load->getExtensionType(), SL, HiVT, Load->getChain(), HiPtr,
1915 SrcValue.getWithOffset(LoMemVT.getStoreSize()), HiMemVT, HiAlign,
1916 Load->getMemOperand()->getFlags(), Load->getAAInfo());
1917
1918 SDValue Join;
1919 if (LoVT == HiVT) {
1920 // This is the case that the vector is power of two so was evenly split.
1921 Join = DAG.getNode(ISD::CONCAT_VECTORS, SL, VT, LoLoad, HiLoad);
1922 } else {
1923 Join = DAG.getNode(ISD::INSERT_SUBVECTOR, SL, VT, DAG.getPOISON(VT), LoLoad,
1924 DAG.getVectorIdxConstant(0, SL));
1925 Join = DAG.getNode(
1927 VT, Join, HiLoad,
1929 }
1930
1931 SDValue Ops[] = {Join, DAG.getNode(ISD::TokenFactor, SL, MVT::Other,
1932 LoLoad.getValue(1), HiLoad.getValue(1))};
1933
1934 return DAG.getMergeValues(Ops, SL);
1935}
1936
1938 SelectionDAG &DAG) const {
1940 EVT VT = Op.getValueType();
1941 SDValue BasePtr = Load->getBasePtr();
1942 EVT MemVT = Load->getMemoryVT();
1943 SDLoc SL(Op);
1944 const MachinePointerInfo &SrcValue = Load->getMemOperand()->getPointerInfo();
1945 Align BaseAlign = Load->getAlign();
1946 unsigned NumElements = MemVT.getVectorNumElements();
1947
1948 // Widen from vec3 to vec4 when the load is at least 8-byte aligned
1949 // or 16-byte fully dereferenceable. Otherwise, split the vector load.
1950 if (NumElements != 3 ||
1951 (BaseAlign < Align(8) &&
1952 !SrcValue.isDereferenceable(16, *DAG.getContext(), DAG.getDataLayout())))
1953 return SplitVectorLoad(Op, DAG);
1954
1955 assert(NumElements == 3);
1956
1957 EVT WideVT =
1959 EVT WideMemVT =
1961 SDValue WideLoad = DAG.getExtLoad(
1962 Load->getExtensionType(), SL, WideVT, Load->getChain(), BasePtr, SrcValue,
1963 WideMemVT, BaseAlign, Load->getMemOperand()->getFlags());
1964 return DAG.getMergeValues(
1965 {DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, VT, WideLoad,
1966 DAG.getVectorIdxConstant(0, SL)),
1967 WideLoad.getValue(1)},
1968 SL);
1969}
1970
1972 SelectionDAG &DAG) const {
1974 SDValue Val = Store->getValue();
1975 EVT VT = Val.getValueType();
1976
1977 // If this is a 2 element vector, we really want to scalarize and not create
1978 // weird 1 element vectors.
1979 if (VT.getVectorNumElements() == 2)
1980 return scalarizeVectorStore(Store, DAG);
1981
1982 EVT MemVT = Store->getMemoryVT();
1983 SDValue Chain = Store->getChain();
1984 SDValue BasePtr = Store->getBasePtr();
1985 SDLoc SL(Op);
1986
1987 EVT LoVT, HiVT;
1988 EVT LoMemVT, HiMemVT;
1989 SDValue Lo, Hi;
1990
1991 std::tie(LoVT, HiVT) = getSplitDestVTs(VT, DAG);
1992 std::tie(LoMemVT, HiMemVT) = getSplitDestVTs(MemVT, DAG);
1993 std::tie(Lo, Hi) = splitVector(Val, SL, LoVT, HiVT, DAG);
1994
1995 SDValue HiPtr = DAG.getObjectPtrOffset(SL, BasePtr, LoMemVT.getStoreSize());
1996
1997 const MachinePointerInfo &SrcValue = Store->getMemOperand()->getPointerInfo();
1998 Align BaseAlign = Store->getAlign();
1999 unsigned Size = LoMemVT.getStoreSize();
2000 Align HiAlign = commonAlignment(BaseAlign, Size);
2001
2002 SDValue LoStore =
2003 DAG.getTruncStore(Chain, SL, Lo, BasePtr, SrcValue, LoMemVT, BaseAlign,
2004 Store->getMemOperand()->getFlags(), Store->getAAInfo());
2005 SDValue HiStore = DAG.getTruncStore(
2006 Chain, SL, Hi, HiPtr, SrcValue.getWithOffset(Size), HiMemVT, HiAlign,
2007 Store->getMemOperand()->getFlags(), Store->getAAInfo());
2008
2009 return DAG.getNode(ISD::TokenFactor, SL, MVT::Other, LoStore, HiStore);
2010}
2011
2012// This is a shortcut for integer division because we have fast i32<->f32
2013// conversions, and fast f32 reciprocal instructions.
2015 bool Sign) const {
2016 SDLoc DL(Op);
2017 EVT VT = Op.getValueType();
2018 assert(VT == MVT::i32 && "LowerDIVREMToFloat expects an i32");
2019
2020 SDValue LHS = Op.getOperand(0);
2021 SDValue RHS = Op.getOperand(1);
2022 MVT IntVT = MVT::i32;
2023 MVT FltVT = MVT::f32;
2024
2025 unsigned LHSSignBits;
2026 unsigned RHSSignBits;
2027 if (Sign) {
2028 LHSSignBits = DAG.ComputeNumSignBits(LHS);
2029 RHSSignBits = DAG.ComputeNumSignBits(RHS);
2030 if (LHSSignBits < 9 || RHSSignBits < 9)
2031 return SDValue();
2032 } else {
2033 KnownBits LHSKnown = DAG.computeKnownBits(LHS);
2034 KnownBits RHSKnown = DAG.computeKnownBits(RHS);
2035
2036 LHSSignBits = LHSKnown.countMinLeadingZeros();
2037 RHSSignBits = RHSKnown.countMinLeadingZeros();
2038 }
2039
2040 unsigned BitSize = VT.getSizeInBits();
2041 unsigned SignBits = std::min(LHSSignBits, RHSSignBits);
2042 unsigned DivBits = BitSize - SignBits;
2043 if (Sign)
2044 ++DivBits;
2045
2046 // In order to avoid problems due to 1 ulp accuracy issues with v_rcp_f32,
2047 // limit LowerDIVREMToFloat to:
2048 // [-0x400000,0x3FFFFF] for Sign
2049 // [ 0x000000,0x3FFFFF] for !Sign
2050 // This matches what is done in expandDivRemToFloatImpl.
2051 if (DivBits > (Sign ? 23 : 22))
2052 return SDValue();
2053
2056
2057 // int ia = (int)LHS;
2058 SDValue ia = LHS;
2059
2060 // int ib, (int)RHS;
2061 SDValue ib = RHS;
2062
2063 // The calculation:
2064 // fq = fa*recip(fb)
2065 // may be too small due to the 1ulp accuracy in the recip
2066 // operation and rounding issues. Since fq is truncated to produce
2067 // an integer value it may be too small by one. This is
2068 // dealt with by incrementing fa by 1ulp:
2069 // fq = (fa+1ulp)*recip(fb)
2070 // This will increase fa's magnitude by at most 0.5
2071 // (i.e. when fabs(fa)==0x400000 the LSB of the mantissa represents 0.5).
2072 // Thus, this method is safe since fa must be incremented by at least 1.0
2073 // for the quotient to increase by one.
2074 SDValue fa = DAG.getNode(ToFp, DL, FltVT, ia);
2075 SDValue faAsInt = DAG.getNode(ISD::BITCAST, DL, MVT::i32, fa);
2076 SDValue faIncremented = DAG.getNode(ISD::ADD, DL, MVT::i32, faAsInt,
2077 DAG.getConstant(1, DL, MVT::i32));
2078 fa = DAG.getNode(ISD::BITCAST, DL, FltVT, faIncremented);
2079
2080 // float fb = (float)ib;
2081 SDValue fb = DAG.getNode(ToFp, DL, FltVT, ib);
2082
2083 SDValue fq = DAG.getNode(ISD::FMUL, DL, FltVT,
2084 fa, DAG.getNode(AMDGPUISD::RCP, DL, FltVT, fb));
2085
2086 // fq = trunc(fq);
2087 fq = DAG.getNode(ISD::FTRUNC, DL, FltVT, fq);
2088
2089 // int iq = (int)fq;
2090 SDValue Div = DAG.getNode(ToInt, DL, IntVT, fq);
2091
2092 // Rem needs compensation, it's easier to recompute it
2093 SDValue Rem = DAG.getNode(ISD::MUL, DL, VT, Div, RHS);
2094 Rem = DAG.getNode(ISD::SUB, DL, VT, LHS, Rem);
2095
2096 return DAG.getMergeValues({ Div, Rem }, DL);
2097}
2098
2100 SelectionDAG &DAG,
2102 SDLoc DL(Op);
2103 EVT VT = Op.getValueType();
2104
2105 assert(VT == MVT::i64 && "LowerUDIVREM64 expects an i64");
2106
2107 EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
2108
2109 SDValue One = DAG.getConstant(1, DL, HalfVT);
2110 SDValue Zero = DAG.getConstant(0, DL, HalfVT);
2111
2112 //HiLo split
2113 SDValue LHS_Lo, LHS_Hi;
2114 SDValue LHS = Op.getOperand(0);
2115 std::tie(LHS_Lo, LHS_Hi) = DAG.SplitScalar(LHS, DL, HalfVT, HalfVT);
2116
2117 SDValue RHS_Lo, RHS_Hi;
2118 SDValue RHS = Op.getOperand(1);
2119 std::tie(RHS_Lo, RHS_Hi) = DAG.SplitScalar(RHS, DL, HalfVT, HalfVT);
2120
2121 if (DAG.MaskedValueIsZero(RHS, APInt::getHighBitsSet(64, 32)) &&
2122 DAG.MaskedValueIsZero(LHS, APInt::getHighBitsSet(64, 32))) {
2123
2124 SDValue Res = DAG.getNode(ISD::UDIVREM, DL, DAG.getVTList(HalfVT, HalfVT),
2125 LHS_Lo, RHS_Lo);
2126
2127 SDValue DIV = DAG.getBuildVector(MVT::v2i32, DL, {Res.getValue(0), Zero});
2128 SDValue REM = DAG.getBuildVector(MVT::v2i32, DL, {Res.getValue(1), Zero});
2129
2130 Results.push_back(DAG.getNode(ISD::BITCAST, DL, MVT::i64, DIV));
2131 Results.push_back(DAG.getNode(ISD::BITCAST, DL, MVT::i64, REM));
2132 return;
2133 }
2134
2135 if (isTypeLegal(MVT::i64)) {
2136 // The algorithm here is based on ideas from "Software Integer Division",
2137 // Tom Rodeheffer, August 2008.
2138
2141
2142 // Compute denominator reciprocal.
2143 unsigned FMAD =
2144 !Subtarget->hasMadMacF32Insts() ? (unsigned)ISD::FMA
2147 : (unsigned)AMDGPUISD::FMAD_FTZ;
2148
2149 SDValue Cvt_Lo = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::f32, RHS_Lo);
2150 SDValue Cvt_Hi = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::f32, RHS_Hi);
2151 SDValue Mad1 = DAG.getNode(FMAD, DL, MVT::f32, Cvt_Hi,
2152 DAG.getConstantFP(APInt(32, 0x4f800000).bitsToFloat(), DL, MVT::f32),
2153 Cvt_Lo);
2154 SDValue Rcp = DAG.getNode(AMDGPUISD::RCP, DL, MVT::f32, Mad1);
2155 SDValue Mul1 = DAG.getNode(ISD::FMUL, DL, MVT::f32, Rcp,
2156 DAG.getConstantFP(APInt(32, 0x5f7ffffc).bitsToFloat(), DL, MVT::f32));
2157 SDValue Mul2 = DAG.getNode(ISD::FMUL, DL, MVT::f32, Mul1,
2158 DAG.getConstantFP(APInt(32, 0x2f800000).bitsToFloat(), DL, MVT::f32));
2159 SDValue Trunc = DAG.getNode(ISD::FTRUNC, DL, MVT::f32, Mul2);
2160 SDValue Mad2 = DAG.getNode(FMAD, DL, MVT::f32, Trunc,
2161 DAG.getConstantFP(APInt(32, 0xcf800000).bitsToFloat(), DL, MVT::f32),
2162 Mul1);
2163 SDValue Rcp_Lo = DAG.getNode(ISD::FP_TO_UINT, DL, HalfVT, Mad2);
2164 SDValue Rcp_Hi = DAG.getNode(ISD::FP_TO_UINT, DL, HalfVT, Trunc);
2165 SDValue Rcp64 = DAG.getBitcast(VT,
2166 DAG.getBuildVector(MVT::v2i32, DL, {Rcp_Lo, Rcp_Hi}));
2167
2168 SDValue Zero64 = DAG.getConstant(0, DL, VT);
2169 SDValue One64 = DAG.getConstant(1, DL, VT);
2170 SDValue Zero1 = DAG.getConstant(0, DL, MVT::i1);
2171 SDVTList HalfCarryVT = DAG.getVTList(HalfVT, MVT::i1);
2172
2173 // First round of UNR (Unsigned integer Newton-Raphson).
2174 SDValue Neg_RHS = DAG.getNode(ISD::SUB, DL, VT, Zero64, RHS);
2175 SDValue Mullo1 = DAG.getNode(ISD::MUL, DL, VT, Neg_RHS, Rcp64);
2176 SDValue Mulhi1 = DAG.getNode(ISD::MULHU, DL, VT, Rcp64, Mullo1);
2177 SDValue Mulhi1_Lo, Mulhi1_Hi;
2178 std::tie(Mulhi1_Lo, Mulhi1_Hi) =
2179 DAG.SplitScalar(Mulhi1, DL, HalfVT, HalfVT);
2180 SDValue Add1_Lo = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Rcp_Lo,
2181 Mulhi1_Lo, Zero1);
2182 SDValue Add1_Hi = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Rcp_Hi,
2183 Mulhi1_Hi, Add1_Lo.getValue(1));
2184 SDValue Add1 = DAG.getBitcast(VT,
2185 DAG.getBuildVector(MVT::v2i32, DL, {Add1_Lo, Add1_Hi}));
2186
2187 // Second round of UNR.
2188 SDValue Mullo2 = DAG.getNode(ISD::MUL, DL, VT, Neg_RHS, Add1);
2189 SDValue Mulhi2 = DAG.getNode(ISD::MULHU, DL, VT, Add1, Mullo2);
2190 SDValue Mulhi2_Lo, Mulhi2_Hi;
2191 std::tie(Mulhi2_Lo, Mulhi2_Hi) =
2192 DAG.SplitScalar(Mulhi2, DL, HalfVT, HalfVT);
2193 SDValue Add2_Lo = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Add1_Lo,
2194 Mulhi2_Lo, Zero1);
2195 SDValue Add2_Hi = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Add1_Hi,
2196 Mulhi2_Hi, Add2_Lo.getValue(1));
2197 SDValue Add2 = DAG.getBitcast(VT,
2198 DAG.getBuildVector(MVT::v2i32, DL, {Add2_Lo, Add2_Hi}));
2199
2200 SDValue Mulhi3 = DAG.getNode(ISD::MULHU, DL, VT, LHS, Add2);
2201
2202 SDValue Mul3 = DAG.getNode(ISD::MUL, DL, VT, RHS, Mulhi3);
2203
2204 SDValue Mul3_Lo, Mul3_Hi;
2205 std::tie(Mul3_Lo, Mul3_Hi) = DAG.SplitScalar(Mul3, DL, HalfVT, HalfVT);
2206 SDValue Sub1_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, LHS_Lo,
2207 Mul3_Lo, Zero1);
2208 SDValue Sub1_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, LHS_Hi,
2209 Mul3_Hi, Sub1_Lo.getValue(1));
2210 SDValue Sub1_Mi = DAG.getNode(ISD::SUB, DL, HalfVT, LHS_Hi, Mul3_Hi);
2211 SDValue Sub1 = DAG.getBitcast(VT,
2212 DAG.getBuildVector(MVT::v2i32, DL, {Sub1_Lo, Sub1_Hi}));
2213
2214 SDValue MinusOne = DAG.getConstant(0xffffffffu, DL, HalfVT);
2215 SDValue C1 = DAG.getSelectCC(DL, Sub1_Hi, RHS_Hi, MinusOne, Zero,
2216 ISD::SETUGE);
2217 SDValue C2 = DAG.getSelectCC(DL, Sub1_Lo, RHS_Lo, MinusOne, Zero,
2218 ISD::SETUGE);
2219 SDValue C3 = DAG.getSelectCC(DL, Sub1_Hi, RHS_Hi, C2, C1, ISD::SETEQ);
2220
2221 // TODO: Here and below portions of the code can be enclosed into if/endif.
2222 // Currently control flow is unconditional and we have 4 selects after
2223 // potential endif to substitute PHIs.
2224
2225 // if C3 != 0 ...
2226 SDValue Sub2_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub1_Lo,
2227 RHS_Lo, Zero1);
2228 SDValue Sub2_Mi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub1_Mi,
2229 RHS_Hi, Sub1_Lo.getValue(1));
2230 SDValue Sub2_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Mi,
2231 Zero, Sub2_Lo.getValue(1));
2232 SDValue Sub2 = DAG.getBitcast(VT,
2233 DAG.getBuildVector(MVT::v2i32, DL, {Sub2_Lo, Sub2_Hi}));
2234
2235 SDValue Add3 = DAG.getNode(ISD::ADD, DL, VT, Mulhi3, One64);
2236
2237 SDValue C4 = DAG.getSelectCC(DL, Sub2_Hi, RHS_Hi, MinusOne, Zero,
2238 ISD::SETUGE);
2239 SDValue C5 = DAG.getSelectCC(DL, Sub2_Lo, RHS_Lo, MinusOne, Zero,
2240 ISD::SETUGE);
2241 SDValue C6 = DAG.getSelectCC(DL, Sub2_Hi, RHS_Hi, C5, C4, ISD::SETEQ);
2242
2243 // if (C6 != 0)
2244 SDValue Add4 = DAG.getNode(ISD::ADD, DL, VT, Add3, One64);
2245
2246 SDValue Sub3_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Lo,
2247 RHS_Lo, Zero1);
2248 SDValue Sub3_Mi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Mi,
2249 RHS_Hi, Sub2_Lo.getValue(1));
2250 SDValue Sub3_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub3_Mi,
2251 Zero, Sub3_Lo.getValue(1));
2252 SDValue Sub3 = DAG.getBitcast(VT,
2253 DAG.getBuildVector(MVT::v2i32, DL, {Sub3_Lo, Sub3_Hi}));
2254
2255 // endif C6
2256 // endif C3
2257
2258 SDValue Sel1 = DAG.getSelectCC(DL, C6, Zero, Add4, Add3, ISD::SETNE);
2259 SDValue Div = DAG.getSelectCC(DL, C3, Zero, Sel1, Mulhi3, ISD::SETNE);
2260
2261 SDValue Sel2 = DAG.getSelectCC(DL, C6, Zero, Sub3, Sub2, ISD::SETNE);
2262 SDValue Rem = DAG.getSelectCC(DL, C3, Zero, Sel2, Sub1, ISD::SETNE);
2263
2264 Results.push_back(Div);
2265 Results.push_back(Rem);
2266
2267 return;
2268 }
2269
2270 // r600 expandion.
2271 // Get Speculative values
2272 SDValue DIV_Part = DAG.getNode(ISD::UDIV, DL, HalfVT, LHS_Hi, RHS_Lo);
2273 SDValue REM_Part = DAG.getNode(ISD::UREM, DL, HalfVT, LHS_Hi, RHS_Lo);
2274
2275 SDValue REM_Lo = DAG.getSelectCC(DL, RHS_Hi, Zero, REM_Part, LHS_Hi, ISD::SETEQ);
2276 SDValue REM = DAG.getBuildVector(MVT::v2i32, DL, {REM_Lo, Zero});
2277 REM = DAG.getNode(ISD::BITCAST, DL, MVT::i64, REM);
2278
2279 SDValue DIV_Hi = DAG.getSelectCC(DL, RHS_Hi, Zero, DIV_Part, Zero, ISD::SETEQ);
2280 SDValue DIV_Lo = Zero;
2281
2282 const unsigned halfBitWidth = HalfVT.getSizeInBits();
2283
2284 for (unsigned i = 0; i < halfBitWidth; ++i) {
2285 const unsigned bitPos = halfBitWidth - i - 1;
2286 SDValue POS = DAG.getConstant(bitPos, DL, HalfVT);
2287 // Get value of high bit
2288 SDValue HBit = DAG.getNode(ISD::SRL, DL, HalfVT, LHS_Lo, POS);
2289 HBit = DAG.getNode(ISD::AND, DL, HalfVT, HBit, One);
2290 HBit = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, HBit);
2291
2292 // Shift
2293 REM = DAG.getNode(ISD::SHL, DL, VT, REM, DAG.getConstant(1, DL, VT));
2294 // Add LHS high bit
2295 REM = DAG.getNode(ISD::OR, DL, VT, REM, HBit);
2296
2297 SDValue BIT = DAG.getConstant(1ULL << bitPos, DL, HalfVT);
2298 SDValue realBIT = DAG.getSelectCC(DL, REM, RHS, BIT, Zero, ISD::SETUGE);
2299
2300 DIV_Lo = DAG.getNode(ISD::OR, DL, HalfVT, DIV_Lo, realBIT);
2301
2302 // Update REM
2303 SDValue REM_sub = DAG.getNode(ISD::SUB, DL, VT, REM, RHS);
2304 REM = DAG.getSelectCC(DL, REM, RHS, REM_sub, REM, ISD::SETUGE);
2305 }
2306
2307 SDValue DIV = DAG.getBuildVector(MVT::v2i32, DL, {DIV_Lo, DIV_Hi});
2308 DIV = DAG.getNode(ISD::BITCAST, DL, MVT::i64, DIV);
2309 Results.push_back(DIV);
2310 Results.push_back(REM);
2311}
2312
2314 SelectionDAG &DAG) const {
2315 SDLoc DL(Op);
2316 EVT VT = Op.getValueType();
2317
2318 if (VT == MVT::i64) {
2320 LowerUDIVREM64(Op, DAG, Results);
2321 return DAG.getMergeValues(Results, DL);
2322 }
2323
2324 if (VT == MVT::i32) {
2325 if (SDValue Res = LowerDIVREMToFloat(Op, DAG, false))
2326 return Res;
2327 }
2328
2329 SDValue X = Op.getOperand(0);
2330 SDValue Y = Op.getOperand(1);
2331
2332 // See AMDGPUCodeGenPrepare::expandDivRem32 for a description of the
2333 // algorithm used here.
2334
2335 // Initial estimate of inv(y).
2336 SDValue Z = DAG.getNode(AMDGPUISD::URECIP, DL, VT, Y);
2337
2338 // One round of UNR.
2339 SDValue NegY = DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), Y);
2340 SDValue NegYZ = DAG.getNode(ISD::MUL, DL, VT, NegY, Z);
2341 Z = DAG.getNode(ISD::ADD, DL, VT, Z,
2342 DAG.getNode(ISD::MULHU, DL, VT, Z, NegYZ));
2343
2344 // Quotient/remainder estimate.
2345 SDValue Q = DAG.getNode(ISD::MULHU, DL, VT, X, Z);
2346 SDValue R =
2347 DAG.getNode(ISD::SUB, DL, VT, X, DAG.getNode(ISD::MUL, DL, VT, Q, Y));
2348
2349 // First quotient/remainder refinement.
2350 EVT CCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
2351 SDValue One = DAG.getConstant(1, DL, VT);
2352 SDValue Cond = DAG.getSetCC(DL, CCVT, R, Y, ISD::SETUGE);
2353 Q = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2354 DAG.getNode(ISD::ADD, DL, VT, Q, One), Q);
2355 R = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2356 DAG.getNode(ISD::SUB, DL, VT, R, Y), R);
2357
2358 // Second quotient/remainder refinement.
2359 Cond = DAG.getSetCC(DL, CCVT, R, Y, ISD::SETUGE);
2360 Q = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2361 DAG.getNode(ISD::ADD, DL, VT, Q, One), Q);
2362 R = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2363 DAG.getNode(ISD::SUB, DL, VT, R, Y), R);
2364
2365 return DAG.getMergeValues({Q, R}, DL);
2366}
2367
2369 SelectionDAG &DAG) const {
2370 SDLoc DL(Op);
2371 EVT VT = Op.getValueType();
2372
2373 SDValue LHS = Op.getOperand(0);
2374 SDValue RHS = Op.getOperand(1);
2375
2376 SDValue Zero = DAG.getConstant(0, DL, VT);
2377 SDValue NegOne = DAG.getAllOnesConstant(DL, VT);
2378
2379 if (VT == MVT::i32) {
2380 if (SDValue Res = LowerDIVREMToFloat(Op, DAG, true))
2381 return Res;
2382 }
2383
2384 // LHS must have > 33 sign-bits to ensure that LHS != -2147483648
2385 // Otherwise 32-bit division cannot be used safely.
2386 // -2147483648/1 and -2147483648/-1 are not equal,
2387 // but they produce the same lower 32-bit result.
2388 if (VT == MVT::i64 && DAG.ComputeNumSignBits(LHS) > 33 &&
2389 DAG.ComputeNumSignBits(RHS) > 32) {
2390 EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
2391
2392 //HiLo split
2393 SDValue LHS_Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, HalfVT, LHS, Zero);
2394 SDValue RHS_Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, HalfVT, RHS, Zero);
2395 SDValue DIVREM = DAG.getNode(ISD::SDIVREM, DL, DAG.getVTList(HalfVT, HalfVT),
2396 LHS_Lo, RHS_Lo);
2397 SDValue Res[2] = {
2398 DAG.getNode(ISD::SIGN_EXTEND, DL, VT, DIVREM.getValue(0)),
2399 DAG.getNode(ISD::SIGN_EXTEND, DL, VT, DIVREM.getValue(1))
2400 };
2401 return DAG.getMergeValues(Res, DL);
2402 }
2403
2404 SDValue LHSign = DAG.getSelectCC(DL, LHS, Zero, NegOne, Zero, ISD::SETLT);
2405 SDValue RHSign = DAG.getSelectCC(DL, RHS, Zero, NegOne, Zero, ISD::SETLT);
2406 SDValue DSign = DAG.getNode(ISD::XOR, DL, VT, LHSign, RHSign);
2407 SDValue RSign = LHSign; // Remainder sign is the same as LHS
2408
2409 LHS = DAG.getNode(ISD::ADD, DL, VT, LHS, LHSign);
2410 RHS = DAG.getNode(ISD::ADD, DL, VT, RHS, RHSign);
2411
2412 LHS = DAG.getNode(ISD::XOR, DL, VT, LHS, LHSign);
2413 RHS = DAG.getNode(ISD::XOR, DL, VT, RHS, RHSign);
2414
2415 SDValue Div = DAG.getNode(ISD::UDIVREM, DL, DAG.getVTList(VT, VT), LHS, RHS);
2416 SDValue Rem = Div.getValue(1);
2417
2418 Div = DAG.getNode(ISD::XOR, DL, VT, Div, DSign);
2419 Rem = DAG.getNode(ISD::XOR, DL, VT, Rem, RSign);
2420
2421 Div = DAG.getNode(ISD::SUB, DL, VT, Div, DSign);
2422 Rem = DAG.getNode(ISD::SUB, DL, VT, Rem, RSign);
2423
2424 SDValue Res[2] = {
2425 Div,
2426 Rem
2427 };
2428 return DAG.getMergeValues(Res, DL);
2429}
2430
2432 SDLoc SL(Op);
2433 SDValue Src = Op.getOperand(0);
2434
2435 // result = trunc(src)
2436 // if (src > 0.0 && src != result)
2437 // result += 1.0
2438
2439 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, MVT::f64, Src);
2440
2441 const SDValue Zero = DAG.getConstantFP(0.0, SL, MVT::f64);
2442 const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f64);
2443
2444 EVT SetCCVT =
2445 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2446
2447 SDValue Lt0 = DAG.getSetCC(SL, SetCCVT, Src, Zero, ISD::SETOGT);
2448 SDValue NeTrunc = DAG.getSetCC(SL, SetCCVT, Src, Trunc, ISD::SETONE);
2449 SDValue And = DAG.getNode(ISD::AND, SL, SetCCVT, Lt0, NeTrunc);
2450
2451 SDValue Add = DAG.getNode(ISD::SELECT, SL, MVT::f64, And, One, Zero);
2452 // TODO: Should this propagate fast-math-flags?
2453 return DAG.getNode(ISD::FADD, SL, MVT::f64, Trunc, Add);
2454}
2455
2457 SelectionDAG &DAG) {
2458 const unsigned FractBits = 52;
2459 const unsigned ExpBits = 11;
2460
2461 SDValue ExpPart = DAG.getNode(AMDGPUISD::BFE_U32, SL, MVT::i32,
2462 Hi,
2463 DAG.getConstant(FractBits - 32, SL, MVT::i32),
2464 DAG.getConstant(ExpBits, SL, MVT::i32));
2465 SDValue Exp = DAG.getNode(ISD::SUB, SL, MVT::i32, ExpPart,
2466 DAG.getConstant(1023, SL, MVT::i32));
2467
2468 return Exp;
2469}
2470
2472 SDLoc SL(Op);
2473 SDValue Src = Op.getOperand(0);
2474
2475 assert(Op.getValueType() == MVT::f64);
2476
2477 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
2478
2479 // Extract the upper half, since this is where we will find the sign and
2480 // exponent.
2481 SDValue Hi = getHiHalf64(Src, DAG);
2482
2483 SDValue Exp = extractF64Exponent(Hi, SL, DAG);
2484
2485 const unsigned FractBits = 52;
2486
2487 // Extract the sign bit.
2488 const SDValue SignBitMask = DAG.getConstant(UINT32_C(1) << 31, SL, MVT::i32);
2489 SDValue SignBit = DAG.getNode(ISD::AND, SL, MVT::i32, Hi, SignBitMask);
2490
2491 // Extend back to 64-bits.
2492 SDValue SignBit64 = DAG.getBuildVector(MVT::v2i32, SL, {Zero, SignBit});
2493 SignBit64 = DAG.getNode(ISD::BITCAST, SL, MVT::i64, SignBit64);
2494
2495 SDValue BcInt = DAG.getNode(ISD::BITCAST, SL, MVT::i64, Src);
2496 const SDValue FractMask
2497 = DAG.getConstant((UINT64_C(1) << FractBits) - 1, SL, MVT::i64);
2498
2499 SDValue Shr = DAG.getNode(ISD::SRA, SL, MVT::i64, FractMask, Exp);
2500 SDValue Not = DAG.getNOT(SL, Shr, MVT::i64);
2501 SDValue Tmp0 = DAG.getNode(ISD::AND, SL, MVT::i64, BcInt, Not);
2502
2503 EVT SetCCVT =
2504 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::i32);
2505
2506 const SDValue FiftyOne = DAG.getConstant(FractBits - 1, SL, MVT::i32);
2507
2508 SDValue ExpLt0 = DAG.getSetCC(SL, SetCCVT, Exp, Zero, ISD::SETLT);
2509 SDValue ExpGt51 = DAG.getSetCC(SL, SetCCVT, Exp, FiftyOne, ISD::SETGT);
2510
2511 SDValue Tmp1 = DAG.getNode(ISD::SELECT, SL, MVT::i64, ExpLt0, SignBit64, Tmp0);
2512 SDValue Tmp2 = DAG.getNode(ISD::SELECT, SL, MVT::i64, ExpGt51, BcInt, Tmp1);
2513
2514 return DAG.getNode(ISD::BITCAST, SL, MVT::f64, Tmp2);
2515}
2516
2518 SelectionDAG &DAG) const {
2519 SDLoc SL(Op);
2520 SDValue Src = Op.getOperand(0);
2521
2522 assert(Op.getValueType() == MVT::f64);
2523
2524 APFloat C1Val(APFloat::IEEEdouble(), "0x1.0p+52");
2525 SDValue C1 = DAG.getConstantFP(C1Val, SL, MVT::f64);
2526 SDValue CopySign = DAG.getNode(ISD::FCOPYSIGN, SL, MVT::f64, C1, Src);
2527
2528 // TODO: Should this propagate fast-math-flags?
2529
2530 SDValue Tmp1 = DAG.getNode(ISD::FADD, SL, MVT::f64, Src, CopySign);
2531 SDValue Tmp2 = DAG.getNode(ISD::FSUB, SL, MVT::f64, Tmp1, CopySign);
2532
2533 SDValue Fabs = DAG.getNode(ISD::FABS, SL, MVT::f64, Src);
2534
2535 APFloat C2Val(APFloat::IEEEdouble(), "0x1.fffffffffffffp+51");
2536 SDValue C2 = DAG.getConstantFP(C2Val, SL, MVT::f64);
2537
2538 EVT SetCCVT =
2539 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2540 SDValue Cond = DAG.getSetCC(SL, SetCCVT, Fabs, C2, ISD::SETOGT);
2541
2542 return DAG.getSelect(SL, MVT::f64, Cond, Src, Tmp2);
2543}
2544
2546 SelectionDAG &DAG) const {
2547 // FNEARBYINT and FRINT are the same, except in their handling of FP
2548 // exceptions. Those aren't really meaningful for us, and OpenCL only has
2549 // rint, so just treat them as equivalent.
2550 return DAG.getNode(ISD::FROUNDEVEN, SDLoc(Op), Op.getValueType(),
2551 Op.getOperand(0));
2552}
2553
2555 auto VT = Op.getValueType();
2556 auto Arg = Op.getOperand(0u);
2557 return DAG.getNode(ISD::FROUNDEVEN, SDLoc(Op), VT, Arg);
2558}
2559
2560// XXX - May require not supporting f32 denormals?
2561
2562// Don't handle v2f16. The extra instructions to scalarize and repack around the
2563// compare and vselect end up producing worse code than scalarizing the whole
2564// operation.
2566 SDLoc SL(Op);
2567 SDValue X = Op.getOperand(0);
2568 EVT VT = Op.getValueType();
2569
2570 SDValue T = DAG.getNode(ISD::FTRUNC, SL, VT, X);
2571
2572 // TODO: Should this propagate fast-math-flags?
2573
2574 SDValue Diff = DAG.getNode(ISD::FSUB, SL, VT, X, T);
2575
2576 SDValue AbsDiff = DAG.getNode(ISD::FABS, SL, VT, Diff);
2577
2578 const SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
2579 const SDValue One = DAG.getConstantFP(1.0, SL, VT);
2580
2581 EVT SetCCVT =
2582 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
2583
2584 const SDValue Half = DAG.getConstantFP(0.5, SL, VT);
2585 SDValue Cmp = DAG.getSetCC(SL, SetCCVT, AbsDiff, Half, ISD::SETOGE);
2586 SDValue OneOrZeroFP = DAG.getNode(ISD::SELECT, SL, VT, Cmp, One, Zero);
2587
2588 SDValue SignedOffset = DAG.getNode(ISD::FCOPYSIGN, SL, VT, OneOrZeroFP, X);
2589 return DAG.getNode(ISD::FADD, SL, VT, T, SignedOffset);
2590}
2591
2593 SDLoc SL(Op);
2594 SDValue Src = Op.getOperand(0);
2595
2596 // result = trunc(src);
2597 // if (src < 0.0 && src != result)
2598 // result += -1.0.
2599
2600 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, MVT::f64, Src);
2601
2602 const SDValue Zero = DAG.getConstantFP(0.0, SL, MVT::f64);
2603 const SDValue NegOne = DAG.getConstantFP(-1.0, SL, MVT::f64);
2604
2605 EVT SetCCVT =
2606 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2607
2608 SDValue Lt0 = DAG.getSetCC(SL, SetCCVT, Src, Zero, ISD::SETOLT);
2609 SDValue NeTrunc = DAG.getSetCC(SL, SetCCVT, Src, Trunc, ISD::SETONE);
2610 SDValue And = DAG.getNode(ISD::AND, SL, SetCCVT, Lt0, NeTrunc);
2611
2612 SDValue Add = DAG.getNode(ISD::SELECT, SL, MVT::f64, And, NegOne, Zero);
2613 // TODO: Should this propagate fast-math-flags?
2614 return DAG.getNode(ISD::FADD, SL, MVT::f64, Trunc, Add);
2615}
2616
2617/// Return true if it's known that \p Src can never be an f32 denormal value.
2619 switch (Src.getOpcode()) {
2620 case ISD::FP_EXTEND:
2621 return Src.getOperand(0).getValueType() == MVT::f16;
2622 case ISD::FP16_TO_FP:
2623 case ISD::FFREXP:
2624 case ISD::FSQRT:
2625 case AMDGPUISD::LOG:
2626 case AMDGPUISD::EXP:
2627 return true;
2629 unsigned IntrinsicID = Src.getConstantOperandVal(0);
2630 switch (IntrinsicID) {
2631 case Intrinsic::amdgcn_frexp_mant:
2632 case Intrinsic::amdgcn_log:
2633 case Intrinsic::amdgcn_log_clamp:
2634 case Intrinsic::amdgcn_exp2:
2635 case Intrinsic::amdgcn_sqrt:
2636 return true;
2637 default:
2638 return false;
2639 }
2640 }
2641 default:
2642 return false;
2643 }
2644
2645 llvm_unreachable("covered opcode switch");
2646}
2647
2649 SDNodeFlags Flags) {
2650 return Flags.hasApproximateFuncs();
2651}
2652
2661
2663 SDValue Src,
2664 SDNodeFlags Flags) const {
2665 SDLoc SL(Src);
2666 EVT VT = Src.getValueType();
2667 const fltSemantics &Semantics = VT.getFltSemantics();
2668 SDValue SmallestNormal =
2669 DAG.getConstantFP(APFloat::getSmallestNormalized(Semantics), SL, VT);
2670
2671 // Want to scale denormals up, but negatives and 0 work just as well on the
2672 // scaled path.
2673 SDValue IsLtSmallestNormal = DAG.getSetCC(
2674 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Src,
2675 SmallestNormal, ISD::SETOLT);
2676
2677 return IsLtSmallestNormal;
2678}
2679
2681 SDNodeFlags Flags) const {
2682 SDLoc SL(Src);
2683 EVT VT = Src.getValueType();
2684 const fltSemantics &Semantics = VT.getFltSemantics();
2685 SDValue Inf = DAG.getConstantFP(APFloat::getInf(Semantics), SL, VT);
2686
2687 SDValue Fabs = DAG.getNode(ISD::FABS, SL, VT, Src, Flags);
2688 SDValue IsFinite = DAG.getSetCC(
2689 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Fabs,
2690 Inf, ISD::SETOLT);
2691 return IsFinite;
2692}
2693
2694/// If denormal handling is required return the scaled input to FLOG2, and the
2695/// check for denormal range. Otherwise, return null values.
2696std::pair<SDValue, SDValue>
2698 SDValue Src, SDNodeFlags Flags) const {
2699 if (!needsDenormHandlingF32(DAG, Src, Flags))
2700 return {};
2701
2702 MVT VT = MVT::f32;
2703 const fltSemantics &Semantics = APFloat::IEEEsingle();
2704 SDValue SmallestNormal =
2705 DAG.getConstantFP(APFloat::getSmallestNormalized(Semantics), SL, VT);
2706
2707 SDValue IsLtSmallestNormal = DAG.getSetCC(
2708 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Src,
2709 SmallestNormal, ISD::SETOLT);
2710
2711 SDValue Scale32 = DAG.getConstantFP(0x1.0p+32, SL, VT);
2712 SDValue One = DAG.getConstantFP(1.0, SL, VT);
2713 SDValue ScaleFactor =
2714 DAG.getNode(ISD::SELECT, SL, VT, IsLtSmallestNormal, Scale32, One, Flags);
2715
2716 SDValue ScaledInput = DAG.getNode(ISD::FMUL, SL, VT, Src, ScaleFactor, Flags);
2717 return {ScaledInput, IsLtSmallestNormal};
2718}
2719
2721 // v_log_f32 is good enough for OpenCL, except it doesn't handle denormals.
2722 // If we have to handle denormals, scale up the input and adjust the result.
2723
2724 // scaled = x * (is_denormal ? 0x1.0p+32 : 1.0)
2725 // log2 = amdgpu_log2 - (is_denormal ? 32.0 : 0.0)
2726
2727 SDLoc SL(Op);
2728 EVT VT = Op.getValueType();
2729 SDValue Src = Op.getOperand(0);
2730 SDNodeFlags Flags = Op->getFlags();
2731
2732 if (VT == MVT::f16) {
2733 // Nothing in half is a denormal when promoted to f32.
2734 assert(!isTypeLegal(VT));
2735 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src, Flags);
2736 SDValue Log = DAG.getNode(AMDGPUISD::LOG, SL, MVT::f32, Ext, Flags);
2737 return DAG.getNode(ISD::FP_ROUND, SL, VT, Log,
2738 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
2739 }
2740
2741 auto [ScaledInput, IsLtSmallestNormal] =
2742 getScaledLogInput(DAG, SL, Src, Flags);
2743 if (!ScaledInput)
2744 return DAG.getNode(AMDGPUISD::LOG, SL, VT, Src, Flags);
2745
2746 SDValue Log2 = DAG.getNode(AMDGPUISD::LOG, SL, VT, ScaledInput, Flags);
2747
2748 SDValue ThirtyTwo = DAG.getConstantFP(32.0, SL, VT);
2749 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
2750 SDValue ResultOffset =
2751 DAG.getNode(ISD::SELECT, SL, VT, IsLtSmallestNormal, ThirtyTwo, Zero);
2752 return DAG.getNode(ISD::FSUB, SL, VT, Log2, ResultOffset, Flags);
2753}
2754
2755static SDValue getMad(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue X,
2756 SDValue Y, SDValue C, SDNodeFlags Flags = SDNodeFlags()) {
2757 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, X, Y, Flags);
2758 return DAG.getNode(ISD::FADD, SL, VT, Mul, C, Flags);
2759}
2760
2762 SelectionDAG &DAG) const {
2763 SDValue X = Op.getOperand(0);
2764 EVT VT = Op.getValueType();
2765 SDNodeFlags Flags = Op->getFlags();
2766 SDLoc DL(Op);
2767 const bool IsLog10 = Op.getOpcode() == ISD::FLOG10;
2768 assert(IsLog10 || Op.getOpcode() == ISD::FLOG);
2769
2770 if (VT == MVT::f16 || Flags.hasApproximateFuncs()) {
2771 // TODO: The direct f16 path is 1.79 ulp for f16. This should be used
2772 // depending on !fpmath metadata.
2773
2774 bool PromoteToF32 = VT == MVT::f16 && (!Flags.hasApproximateFuncs() ||
2775 !isTypeLegal(MVT::f16));
2776
2777 if (PromoteToF32) {
2778 // Log and multiply in f32 is always good enough for f16.
2779 X = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, X, Flags);
2780 }
2781
2782 SDValue Lowered = LowerFLOGUnsafe(X, DL, DAG, IsLog10, Flags);
2783 if (PromoteToF32) {
2784 return DAG.getNode(ISD::FP_ROUND, DL, VT, Lowered,
2785 DAG.getTargetConstant(0, DL, MVT::i32), Flags);
2786 }
2787
2788 return Lowered;
2789 }
2790
2791 SDValue ScaledInput, IsScaled;
2792 if (VT == MVT::f16)
2793 X = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, X, Flags);
2794 else {
2795 std::tie(ScaledInput, IsScaled) = getScaledLogInput(DAG, DL, X, Flags);
2796 if (ScaledInput)
2797 X = ScaledInput;
2798 }
2799
2800 SDValue Y = DAG.getNode(AMDGPUISD::LOG, DL, VT, X, Flags);
2801
2802 SDValue R;
2803 if (Subtarget->hasFastFMAF32()) {
2804 // c+cc are ln(2)/ln(10) to more than 49 bits
2805 const float c_log10 = 0x1.344134p-2f;
2806 const float cc_log10 = 0x1.09f79ep-26f;
2807
2808 // c + cc is ln(2) to more than 49 bits
2809 const float c_log = 0x1.62e42ep-1f;
2810 const float cc_log = 0x1.efa39ep-25f;
2811
2812 SDValue C = DAG.getConstantFP(IsLog10 ? c_log10 : c_log, DL, VT);
2813 SDValue CC = DAG.getConstantFP(IsLog10 ? cc_log10 : cc_log, DL, VT);
2814 // This adds correction terms for which contraction may lead to an increase
2815 // in the error of the approximation, so disable it.
2816 Flags.setAllowContract(false);
2817 R = DAG.getNode(ISD::FMUL, DL, VT, Y, C, Flags);
2818 SDValue NegR = DAG.getNode(ISD::FNEG, DL, VT, R, Flags);
2819 SDValue FMA0 = DAG.getNode(ISD::FMA, DL, VT, Y, C, NegR, Flags);
2820 SDValue FMA1 = DAG.getNode(ISD::FMA, DL, VT, Y, CC, FMA0, Flags);
2821 R = DAG.getNode(ISD::FADD, DL, VT, R, FMA1, Flags);
2822 } else {
2823 // ch+ct is ln(2)/ln(10) to more than 36 bits
2824 const float ch_log10 = 0x1.344000p-2f;
2825 const float ct_log10 = 0x1.3509f6p-18f;
2826
2827 // ch + ct is ln(2) to more than 36 bits
2828 const float ch_log = 0x1.62e000p-1f;
2829 const float ct_log = 0x1.0bfbe8p-15f;
2830
2831 SDValue CH = DAG.getConstantFP(IsLog10 ? ch_log10 : ch_log, DL, VT);
2832 SDValue CT = DAG.getConstantFP(IsLog10 ? ct_log10 : ct_log, DL, VT);
2833
2834 SDValue YAsInt = DAG.getNode(ISD::BITCAST, DL, MVT::i32, Y);
2835 SDValue MaskConst = DAG.getConstant(0xfffff000, DL, MVT::i32);
2836 SDValue YHInt = DAG.getNode(ISD::AND, DL, MVT::i32, YAsInt, MaskConst);
2837 SDValue YH = DAG.getNode(ISD::BITCAST, DL, MVT::f32, YHInt);
2838 SDValue YT = DAG.getNode(ISD::FSUB, DL, VT, Y, YH, Flags);
2839 // This adds correction terms for which contraction may lead to an increase
2840 // in the error of the approximation, so disable it.
2841 Flags.setAllowContract(false);
2842 SDValue YTCT = DAG.getNode(ISD::FMUL, DL, VT, YT, CT, Flags);
2843 SDValue Mad0 = getMad(DAG, DL, VT, YH, CT, YTCT, Flags);
2844 SDValue Mad1 = getMad(DAG, DL, VT, YT, CH, Mad0, Flags);
2845 R = getMad(DAG, DL, VT, YH, CH, Mad1);
2846 }
2847
2848 const bool IsFiniteOnly = Flags.hasNoNaNs() && Flags.hasNoInfs();
2849
2850 // TODO: Check if known finite from source value.
2851 if (!IsFiniteOnly) {
2852 SDValue IsFinite = getIsFinite(DAG, Y, Flags);
2853 R = DAG.getNode(ISD::SELECT, DL, VT, IsFinite, R, Y, Flags);
2854 }
2855
2856 if (IsScaled) {
2857 SDValue Zero = DAG.getConstantFP(0.0f, DL, VT);
2858 SDValue ShiftK =
2859 DAG.getConstantFP(IsLog10 ? 0x1.344136p+3f : 0x1.62e430p+4f, DL, VT);
2860 SDValue Shift =
2861 DAG.getNode(ISD::SELECT, DL, VT, IsScaled, ShiftK, Zero, Flags);
2862 R = DAG.getNode(ISD::FSUB, DL, VT, R, Shift, Flags);
2863 }
2864
2865 return R;
2866}
2867
2871
2872// Do f32 fast math expansion for flog2 or flog10. This is accurate enough for a
2873// promote f16 operation.
2875 SelectionDAG &DAG, bool IsLog10,
2876 SDNodeFlags Flags) const {
2877 EVT VT = Src.getValueType();
2878 unsigned LogOp =
2879 VT == MVT::f32 ? (unsigned)AMDGPUISD::LOG : (unsigned)ISD::FLOG2;
2880
2881 double Log2BaseInverted =
2883
2884 if (VT == MVT::f32) {
2885 auto [ScaledInput, IsScaled] = getScaledLogInput(DAG, SL, Src, Flags);
2886 if (ScaledInput) {
2887 SDValue LogSrc = DAG.getNode(AMDGPUISD::LOG, SL, VT, ScaledInput, Flags);
2888 SDValue ScaledResultOffset =
2889 DAG.getConstantFP(-32.0 * Log2BaseInverted, SL, VT);
2890
2891 SDValue Zero = DAG.getConstantFP(0.0f, SL, VT);
2892
2893 SDValue ResultOffset = DAG.getNode(ISD::SELECT, SL, VT, IsScaled,
2894 ScaledResultOffset, Zero, Flags);
2895
2896 SDValue Log2Inv = DAG.getConstantFP(Log2BaseInverted, SL, VT);
2897
2898 if (Subtarget->hasFastFMAF32())
2899 return DAG.getNode(ISD::FMA, SL, VT, LogSrc, Log2Inv, ResultOffset,
2900 Flags);
2901 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, LogSrc, Log2Inv, Flags);
2902 return DAG.getNode(ISD::FADD, SL, VT, Mul, ResultOffset);
2903 }
2904 }
2905
2906 SDValue Log2Operand = DAG.getNode(LogOp, SL, VT, Src, Flags);
2907 SDValue Log2BaseInvertedOperand = DAG.getConstantFP(Log2BaseInverted, SL, VT);
2908
2909 return DAG.getNode(ISD::FMUL, SL, VT, Log2Operand, Log2BaseInvertedOperand,
2910 Flags);
2911}
2912
2913// This expansion gives a result slightly better than 1ulp.
2915 SelectionDAG &DAG) const {
2916 SDLoc DL(Op);
2917 SDValue X = Op.getOperand(0);
2918
2919 // TODO: Check if reassoc is safe. There is an output change in exp2 and
2920 // exp10, which slightly increases ulp.
2921 SDNodeFlags Flags = Op->getFlags() & ~SDNodeFlags::AllowReassociation;
2922
2923 SDValue DN, F, T;
2924
2925 if (Op.getOpcode() == ISD::FEXP2) {
2926 // dn = rint(x)
2927 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, X, Flags);
2928 // f = x - dn
2929 F = DAG.getNode(ISD::FSUB, DL, MVT::f64, X, DN, Flags);
2930 // t = f*C1 + f*C2
2931 SDValue C1 = DAG.getConstantFP(0x1.62e42fefa39efp-1, DL, MVT::f64);
2932 SDValue C2 = DAG.getConstantFP(0x1.abc9e3b39803fp-56, DL, MVT::f64);
2933 SDValue Mul2 = DAG.getNode(ISD::FMUL, DL, MVT::f64, F, C2, Flags);
2934 T = DAG.getNode(ISD::FMA, DL, MVT::f64, F, C1, Mul2, Flags);
2935 } else if (Op.getOpcode() == ISD::FEXP10) {
2936 // dn = rint(x * C1)
2937 SDValue C1 = DAG.getConstantFP(0x1.a934f0979a371p+1, DL, MVT::f64);
2938 SDValue Mul = DAG.getNode(ISD::FMUL, DL, MVT::f64, X, C1, Flags);
2939 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, Mul, Flags);
2940
2941 // f = FMA(-dn, C2, FMA(-dn, C3, x))
2942 SDValue NegDN = DAG.getNode(ISD::FNEG, DL, MVT::f64, DN, Flags);
2943 SDValue C2 = DAG.getConstantFP(-0x1.9dc1da994fd21p-59, DL, MVT::f64);
2944 SDValue C3 = DAG.getConstantFP(0x1.34413509f79ffp-2, DL, MVT::f64);
2945 SDValue Inner = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C3, X, Flags);
2946 F = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C2, Inner, Flags);
2947
2948 // t = FMA(f, C4, f*C5)
2949 SDValue C4 = DAG.getConstantFP(0x1.26bb1bbb55516p+1, DL, MVT::f64);
2950 SDValue C5 = DAG.getConstantFP(-0x1.f48ad494ea3e9p-53, DL, MVT::f64);
2951 SDValue MulF = DAG.getNode(ISD::FMUL, DL, MVT::f64, F, C5, Flags);
2952 T = DAG.getNode(ISD::FMA, DL, MVT::f64, F, C4, MulF, Flags);
2953 } else { // ISD::FEXP
2954 // dn = rint(x * C1)
2955 SDValue C1 = DAG.getConstantFP(0x1.71547652b82fep+0, DL, MVT::f64);
2956 SDValue Mul = DAG.getNode(ISD::FMUL, DL, MVT::f64, X, C1, Flags);
2957 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, Mul, Flags);
2958
2959 // t = FMA(-dn, C2, FMA(-dn, C3, x))
2960 SDValue NegDN = DAG.getNode(ISD::FNEG, DL, MVT::f64, DN, Flags);
2961 SDValue C2 = DAG.getConstantFP(0x1.abc9e3b39803fp-56, DL, MVT::f64);
2962 SDValue C3 = DAG.getConstantFP(0x1.62e42fefa39efp-1, DL, MVT::f64);
2963 SDValue Inner = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C3, X, Flags);
2964 T = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C2, Inner, Flags);
2965 }
2966
2967 // Polynomial expansion for p
2968 SDValue P = DAG.getConstantFP(0x1.ade156a5dcb37p-26, DL, MVT::f64);
2969 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2970 DAG.getConstantFP(0x1.28af3fca7ab0cp-22, DL, MVT::f64),
2971 Flags);
2972 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2973 DAG.getConstantFP(0x1.71dee623fde64p-19, DL, MVT::f64),
2974 Flags);
2975 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2976 DAG.getConstantFP(0x1.a01997c89e6b0p-16, DL, MVT::f64),
2977 Flags);
2978 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2979 DAG.getConstantFP(0x1.a01a014761f6ep-13, DL, MVT::f64),
2980 Flags);
2981 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2982 DAG.getConstantFP(0x1.6c16c1852b7b0p-10, DL, MVT::f64),
2983 Flags);
2984 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2985 DAG.getConstantFP(0x1.1111111122322p-7, DL, MVT::f64), Flags);
2986 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2987 DAG.getConstantFP(0x1.55555555502a1p-5, DL, MVT::f64), Flags);
2988 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2989 DAG.getConstantFP(0x1.5555555555511p-3, DL, MVT::f64), Flags);
2990 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2991 DAG.getConstantFP(0x1.000000000000bp-1, DL, MVT::f64), Flags);
2992
2993 SDValue One = DAG.getConstantFP(1.0, DL, MVT::f64);
2994
2995 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P, One, Flags);
2996 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P, One, Flags);
2997
2998 // z = ldexp(p, (int)dn)
2999 SDValue DNInt = DAG.getNode(ISD::FP_TO_SINT, DL, MVT::i32, DN);
3000 SDValue Z = DAG.getNode(ISD::FLDEXP, DL, MVT::f64, P, DNInt, Flags);
3001
3002 // Overflow/underflow guards
3003 SDValue CondHi = DAG.getSetCC(
3004 DL, MVT::i1, X, DAG.getConstantFP(1024.0, DL, MVT::f64), ISD::SETULE);
3005
3006 if (!Flags.hasNoInfs()) {
3007 SDValue PInf = DAG.getConstantFP(std::numeric_limits<double>::infinity(),
3008 DL, MVT::f64);
3009 Z = DAG.getSelect(DL, MVT::f64, CondHi, Z, PInf, Flags);
3010 }
3011
3012 SDValue CondLo = DAG.getSetCC(
3013 DL, MVT::i1, X, DAG.getConstantFP(-1075.0, DL, MVT::f64), ISD::SETUGE);
3014 SDValue Zero = DAG.getConstantFP(0.0, DL, MVT::f64);
3015 Z = DAG.getSelect(DL, MVT::f64, CondLo, Z, Zero, Flags);
3016
3017 return Z;
3018}
3019
3021 // v_exp_f32 is good enough for OpenCL, except it doesn't handle denormals.
3022 // If we have to handle denormals, scale up the input and adjust the result.
3023
3024 EVT VT = Op.getValueType();
3025 if (VT == MVT::f64)
3026 return lowerFEXPF64(Op, DAG);
3027
3028 SDLoc SL(Op);
3029 SDValue Src = Op.getOperand(0);
3030 SDNodeFlags Flags = Op->getFlags();
3031
3032 if (VT == MVT::f16) {
3033 // Nothing in half is a denormal when promoted to f32.
3034 assert(!isTypeLegal(MVT::f16));
3035 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src, Flags);
3036 SDValue Log = DAG.getNode(AMDGPUISD::EXP, SL, MVT::f32, Ext, Flags);
3037 return DAG.getNode(ISD::FP_ROUND, SL, VT, Log,
3038 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
3039 }
3040
3041 assert(VT == MVT::f32);
3042
3043 if (!needsDenormHandlingF32(DAG, Src, Flags))
3044 return DAG.getNode(AMDGPUISD::EXP, SL, MVT::f32, Src, Flags);
3045
3046 // bool needs_scaling = x < -0x1.f80000p+6f;
3047 // v_exp_f32(x + (s ? 0x1.0p+6f : 0.0f)) * (s ? 0x1.0p-64f : 1.0f);
3048
3049 // -nextafter(128.0, -1)
3050 SDValue RangeCheckConst = DAG.getConstantFP(-0x1.f80000p+6f, SL, VT);
3051
3052 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3053
3054 SDValue NeedsScaling =
3055 DAG.getSetCC(SL, SetCCVT, Src, RangeCheckConst, ISD::SETOLT);
3056
3057 SDValue SixtyFour = DAG.getConstantFP(0x1.0p+6f, SL, VT);
3058 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
3059
3060 SDValue AddOffset =
3061 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, SixtyFour, Zero);
3062
3063 SDValue AddInput = DAG.getNode(ISD::FADD, SL, VT, Src, AddOffset, Flags);
3064 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, AddInput, Flags);
3065
3066 SDValue TwoExpNeg64 = DAG.getConstantFP(0x1.0p-64f, SL, VT);
3067 SDValue One = DAG.getConstantFP(1.0, SL, VT);
3068 SDValue ResultScale =
3069 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, TwoExpNeg64, One);
3070
3071 return DAG.getNode(ISD::FMUL, SL, VT, Exp2, ResultScale, Flags);
3072}
3073
3075 SelectionDAG &DAG,
3076 SDNodeFlags Flags,
3077 bool IsExp10) const {
3078 // exp(x) -> exp2(M_LOG2E_F * x);
3079 // exp10(x) -> exp2(log2(10) * x);
3080 EVT VT = X.getValueType();
3081 SDValue Const =
3082 DAG.getConstantFP(IsExp10 ? 0x1.a934f0p+1f : numbers::log2e, SL, VT);
3083
3084 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, X, Const, Flags);
3085 return DAG.getNode(VT == MVT::f32 ? (unsigned)AMDGPUISD::EXP
3086 : (unsigned)ISD::FEXP2,
3087 SL, VT, Mul, Flags);
3088}
3089
3091 SelectionDAG &DAG,
3092 SDNodeFlags Flags) const {
3093 EVT VT = X.getValueType();
3094 if (VT != MVT::f32 || !needsDenormHandlingF32(DAG, X, Flags))
3095 return lowerFEXPUnsafeImpl(X, SL, DAG, Flags, /*IsExp10=*/false);
3096
3097 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3098
3099 SDValue Threshold = DAG.getConstantFP(-0x1.5d58a0p+6f, SL, VT);
3100 SDValue NeedsScaling = DAG.getSetCC(SL, SetCCVT, X, Threshold, ISD::SETOLT);
3101
3102 SDValue ScaleOffset = DAG.getConstantFP(0x1.0p+6f, SL, VT);
3103
3104 SDValue ScaledX = DAG.getNode(ISD::FADD, SL, VT, X, ScaleOffset, Flags);
3105
3106 SDValue AdjustedX =
3107 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, ScaledX, X);
3108
3109 const SDValue Log2E = DAG.getConstantFP(numbers::log2e, SL, VT);
3110 SDValue ExpInput = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, Log2E, Flags);
3111
3112 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, ExpInput, Flags);
3113
3114 SDValue ResultScaleFactor = DAG.getConstantFP(0x1.969d48p-93f, SL, VT);
3115 SDValue AdjustedResult =
3116 DAG.getNode(ISD::FMUL, SL, VT, Exp2, ResultScaleFactor, Flags);
3117
3118 return DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, AdjustedResult, Exp2,
3119 Flags);
3120}
3121
3122/// Emit approx-funcs appropriate lowering for exp10. inf/nan should still be
3123/// handled correctly.
3125 SelectionDAG &DAG,
3126 SDNodeFlags Flags) const {
3127 const EVT VT = X.getValueType();
3128
3129 const unsigned Exp2Op = VT == MVT::f32 ? static_cast<unsigned>(AMDGPUISD::EXP)
3130 : static_cast<unsigned>(ISD::FEXP2);
3131
3132 if (VT != MVT::f32 || !needsDenormHandlingF32(DAG, X, Flags)) {
3133 // exp2(x * 0x1.a92000p+1f) * exp2(x * 0x1.4f0978p-11f);
3134 SDValue K0 = DAG.getConstantFP(0x1.a92000p+1f, SL, VT);
3135 SDValue K1 = DAG.getConstantFP(0x1.4f0978p-11f, SL, VT);
3136
3137 SDValue Mul0 = DAG.getNode(ISD::FMUL, SL, VT, X, K0, Flags);
3138 SDValue Exp2_0 = DAG.getNode(Exp2Op, SL, VT, Mul0, Flags);
3139 SDValue Mul1 = DAG.getNode(ISD::FMUL, SL, VT, X, K1, Flags);
3140 SDValue Exp2_1 = DAG.getNode(Exp2Op, SL, VT, Mul1, Flags);
3141 return DAG.getNode(ISD::FMUL, SL, VT, Exp2_0, Exp2_1);
3142 }
3143
3144 // bool s = x < -0x1.2f7030p+5f;
3145 // x += s ? 0x1.0p+5f : 0.0f;
3146 // exp10 = exp2(x * 0x1.a92000p+1f) *
3147 // exp2(x * 0x1.4f0978p-11f) *
3148 // (s ? 0x1.9f623ep-107f : 1.0f);
3149
3150 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3151
3152 SDValue Threshold = DAG.getConstantFP(-0x1.2f7030p+5f, SL, VT);
3153 SDValue NeedsScaling = DAG.getSetCC(SL, SetCCVT, X, Threshold, ISD::SETOLT);
3154
3155 SDValue ScaleOffset = DAG.getConstantFP(0x1.0p+5f, SL, VT);
3156 SDValue ScaledX = DAG.getNode(ISD::FADD, SL, VT, X, ScaleOffset, Flags);
3157 SDValue AdjustedX =
3158 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, ScaledX, X);
3159
3160 SDValue K0 = DAG.getConstantFP(0x1.a92000p+1f, SL, VT);
3161 SDValue K1 = DAG.getConstantFP(0x1.4f0978p-11f, SL, VT);
3162
3163 SDValue Mul0 = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, K0, Flags);
3164 SDValue Exp2_0 = DAG.getNode(Exp2Op, SL, VT, Mul0, Flags);
3165 SDValue Mul1 = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, K1, Flags);
3166 SDValue Exp2_1 = DAG.getNode(Exp2Op, SL, VT, Mul1, Flags);
3167
3168 SDValue MulExps = DAG.getNode(ISD::FMUL, SL, VT, Exp2_0, Exp2_1, Flags);
3169
3170 SDValue ResultScaleFactor = DAG.getConstantFP(0x1.9f623ep-107f, SL, VT);
3171 SDValue AdjustedResult =
3172 DAG.getNode(ISD::FMUL, SL, VT, MulExps, ResultScaleFactor, Flags);
3173
3174 return DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, AdjustedResult, MulExps,
3175 Flags);
3176}
3177
3179 EVT VT = Op.getValueType();
3180
3181 if (VT == MVT::f64)
3182 return lowerFEXPF64(Op, DAG);
3183
3184 SDLoc SL(Op);
3185 SDValue X = Op.getOperand(0);
3186 SDNodeFlags Flags = Op->getFlags();
3187 const bool IsExp10 = Op.getOpcode() == ISD::FEXP10;
3188
3189 // TODO: Interpret allowApproxFunc as ignoring DAZ. This is currently copying
3190 // library behavior. Also, is known-not-daz source sufficient?
3191 if (allowApproxFunc(DAG, Flags)) { // TODO: Does this really require fast?
3192 return IsExp10 ? lowerFEXP10Unsafe(X, SL, DAG, Flags)
3193 : lowerFEXPUnsafe(X, SL, DAG, Flags);
3194 }
3195
3196 if (VT.getScalarType() == MVT::f16) {
3197 if (VT.isVector())
3198 return SDValue();
3199
3200 // Nothing in half is a denormal when promoted to f32.
3201 //
3202 // exp(f16 x) ->
3203 // fptrunc (v_exp_f32 (fmul (fpext x), log2e))
3204 //
3205 // exp10(f16 x) ->
3206 // fptrunc (v_exp_f32 (fmul (fpext x), log2(10)))
3207 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, X, Flags);
3208 SDValue Lowered = lowerFEXPUnsafeImpl(Ext, SL, DAG, Flags, IsExp10);
3209 return DAG.getNode(ISD::FP_ROUND, SL, VT, Lowered,
3210 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
3211 }
3212
3213 assert(VT == MVT::f32);
3214
3215 // Algorithm:
3216 //
3217 // e^x = 2^(x/ln(2)) = 2^(x*(64/ln(2))/64)
3218 //
3219 // x*(64/ln(2)) = n + f, |f| <= 0.5, n is integer
3220 // n = 64*m + j, 0 <= j < 64
3221 //
3222 // e^x = 2^((64*m + j + f)/64)
3223 // = (2^m) * (2^(j/64)) * 2^(f/64)
3224 // = (2^m) * (2^(j/64)) * e^(f*(ln(2)/64))
3225 //
3226 // f = x*(64/ln(2)) - n
3227 // r = f*(ln(2)/64) = x - n*(ln(2)/64)
3228 //
3229 // e^x = (2^m) * (2^(j/64)) * e^r
3230 //
3231 // (2^(j/64)) is precomputed
3232 //
3233 // e^r = 1 + r + (r^2)/2! + (r^3)/3! + (r^4)/4! + (r^5)/5!
3234 // e^r = 1 + q
3235 //
3236 // q = r + (r^2)/2! + (r^3)/3! + (r^4)/4! + (r^5)/5!
3237 //
3238 // e^x = (2^m) * ( (2^(j/64)) + q*(2^(j/64)) )
3239 SDNodeFlags FlagsNoContract = Flags;
3240 FlagsNoContract.setAllowContract(false);
3241
3242 SDValue PH, PL;
3243 if (Subtarget->hasFastFMAF32()) {
3244 const float c_exp = numbers::log2ef;
3245 const float cc_exp = 0x1.4ae0bep-26f; // c+cc are 49 bits
3246 const float c_exp10 = 0x1.a934f0p+1f;
3247 const float cc_exp10 = 0x1.2f346ep-24f;
3248
3249 SDValue C = DAG.getConstantFP(IsExp10 ? c_exp10 : c_exp, SL, VT);
3250 SDValue CC = DAG.getConstantFP(IsExp10 ? cc_exp10 : cc_exp, SL, VT);
3251
3252 PH = DAG.getNode(ISD::FMUL, SL, VT, X, C, Flags);
3253 SDValue NegPH = DAG.getNode(ISD::FNEG, SL, VT, PH, Flags);
3254 SDValue FMA0 = DAG.getNode(ISD::FMA, SL, VT, X, C, NegPH, Flags);
3255 PL = DAG.getNode(ISD::FMA, SL, VT, X, CC, FMA0, Flags);
3256 } else {
3257 const float ch_exp = 0x1.714000p+0f;
3258 const float cl_exp = 0x1.47652ap-12f; // ch + cl are 36 bits
3259
3260 const float ch_exp10 = 0x1.a92000p+1f;
3261 const float cl_exp10 = 0x1.4f0978p-11f;
3262
3263 SDValue CH = DAG.getConstantFP(IsExp10 ? ch_exp10 : ch_exp, SL, VT);
3264 SDValue CL = DAG.getConstantFP(IsExp10 ? cl_exp10 : cl_exp, SL, VT);
3265
3266 SDValue XAsInt = DAG.getNode(ISD::BITCAST, SL, MVT::i32, X);
3267 SDValue MaskConst = DAG.getConstant(0xfffff000, SL, MVT::i32);
3268 SDValue XHAsInt = DAG.getNode(ISD::AND, SL, MVT::i32, XAsInt, MaskConst);
3269 SDValue XH = DAG.getNode(ISD::BITCAST, SL, VT, XHAsInt);
3270 SDValue XL = DAG.getNode(ISD::FSUB, SL, VT, X, XH, Flags);
3271
3272 PH = DAG.getNode(ISD::FMUL, SL, VT, XH, CH, Flags);
3273
3274 SDValue XLCL = DAG.getNode(ISD::FMUL, SL, VT, XL, CL, Flags);
3275 SDValue Mad0 = getMad(DAG, SL, VT, XL, CH, XLCL, Flags);
3276 PL = getMad(DAG, SL, VT, XH, CL, Mad0, Flags);
3277 }
3278
3279 SDValue E = DAG.getNode(ISD::FROUNDEVEN, SL, VT, PH, Flags);
3280
3281 // It is unsafe to contract this fsub into the PH multiply.
3282 SDValue PHSubE = DAG.getNode(ISD::FSUB, SL, VT, PH, E, FlagsNoContract);
3283
3284 SDValue A = DAG.getNode(ISD::FADD, SL, VT, PHSubE, PL, Flags);
3285 SDValue IntE = DAG.getNode(ISD::FP_TO_SINT, SL, MVT::i32, E);
3286 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, A, Flags);
3287
3288 SDValue R = DAG.getNode(ISD::FLDEXP, SL, VT, Exp2, IntE, Flags);
3289
3290 SDValue UnderflowCheckConst =
3291 DAG.getConstantFP(IsExp10 ? -0x1.66d3e8p+5f : -0x1.9d1da0p+6f, SL, VT);
3292
3293 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3294 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
3295 SDValue Underflow =
3296 DAG.getSetCC(SL, SetCCVT, X, UnderflowCheckConst, ISD::SETOLT);
3297
3298 R = DAG.getNode(ISD::SELECT, SL, VT, Underflow, Zero, R);
3299
3300 if (!Flags.hasNoInfs()) {
3301 SDValue OverflowCheckConst =
3302 DAG.getConstantFP(IsExp10 ? 0x1.344136p+5f : 0x1.62e430p+6f, SL, VT);
3303 SDValue Overflow =
3304 DAG.getSetCC(SL, SetCCVT, X, OverflowCheckConst, ISD::SETOGT);
3305 SDValue Inf =
3307 R = DAG.getNode(ISD::SELECT, SL, VT, Overflow, Inf, R);
3308 }
3309
3310 return R;
3311}
3312
3313static bool isCtlzOpc(unsigned Opc) {
3314 return Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_POISON;
3315}
3316
3317static bool isCttzOpc(unsigned Opc) {
3318 return Opc == ISD::CTTZ || Opc == ISD::CTTZ_ZERO_POISON;
3319}
3320
3322 SelectionDAG &DAG) const {
3323 auto SL = SDLoc(Op);
3324 auto Opc = Op.getOpcode();
3325 auto Arg = Op.getOperand(0u);
3326 auto ResultVT = Op.getValueType();
3327
3328 if (ResultVT != MVT::i8 && ResultVT != MVT::i16)
3329 return {};
3330
3332 assert(ResultVT == Arg.getValueType());
3333
3334 const uint64_t NumBits = ResultVT.getFixedSizeInBits();
3335 SDValue NumExtBits = DAG.getConstant(32u - NumBits, SL, MVT::i32);
3336 SDValue NewOp;
3337
3338 if (Opc == ISD::CTLZ_ZERO_POISON) {
3339 NewOp = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, Arg);
3340 NewOp = DAG.getNode(ISD::SHL, SL, MVT::i32, NewOp, NumExtBits);
3341 NewOp = DAG.getNode(Opc, SL, MVT::i32, NewOp);
3342 } else {
3343 NewOp = DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i32, Arg);
3344 NewOp = DAG.getNode(Opc, SL, MVT::i32, NewOp);
3345 NewOp = DAG.getNode(ISD::SUB, SL, MVT::i32, NewOp, NumExtBits);
3346 }
3347
3348 return DAG.getNode(ISD::TRUNCATE, SL, ResultVT, NewOp);
3349}
3350
3352 SDLoc SL(Op);
3353 SDValue Src = Op.getOperand(0);
3354
3355 assert(isCtlzOpc(Op.getOpcode()) || isCttzOpc(Op.getOpcode()));
3356 bool Ctlz = isCtlzOpc(Op.getOpcode());
3357 unsigned NewOpc = Ctlz ? AMDGPUISD::FFBH_U32 : AMDGPUISD::FFBL_B32;
3358
3359 bool ZeroUndef = Op.getOpcode() == ISD::CTLZ_ZERO_POISON ||
3360 Op.getOpcode() == ISD::CTTZ_ZERO_POISON;
3361 bool Is64BitScalar = !Src->isDivergent() && Src.getValueType() == MVT::i64;
3362
3363 if (Src.getValueType() == MVT::i32 || Is64BitScalar) {
3364 // (ctlz hi:lo) -> (umin (ffbh src), 32)
3365 // (cttz hi:lo) -> (umin (ffbl src), 32)
3366 // (ctlz_zero_poison src) -> (ffbh src)
3367 // (cttz_zero_poison src) -> (ffbl src)
3368
3369 // 64-bit scalar version produce 32-bit result
3370 // (ctlz hi:lo) -> (umin (S_FLBIT_I32_B64 src), 64)
3371 // (cttz hi:lo) -> (umin (S_FF1_I32_B64 src), 64)
3372 // (ctlz_zero_poison src) -> (S_FLBIT_I32_B64 src)
3373 // (cttz_zero_poison src) -> (S_FF1_I32_B64 src)
3374 SDValue NewOpr = DAG.getNode(NewOpc, SL, MVT::i32, Src);
3375 if (!ZeroUndef) {
3376 const SDValue ConstVal = DAG.getConstant(
3377 Op.getValueType().getScalarSizeInBits(), SL, MVT::i32);
3378 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, NewOpr, ConstVal);
3379 }
3380 return DAG.getNode(ISD::ZERO_EXTEND, SL, Src.getValueType(), NewOpr);
3381 }
3382
3383 SDValue Lo, Hi;
3384 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3385
3386 SDValue OprLo = DAG.getNode(NewOpc, SL, MVT::i32, Lo);
3387 SDValue OprHi = DAG.getNode(NewOpc, SL, MVT::i32, Hi);
3388
3389 // (ctlz hi:lo) -> (umin3 (ffbh hi), (uaddsat (ffbh lo), 32), 64)
3390 // (cttz hi:lo) -> (umin3 (uaddsat (ffbl hi), 32), (ffbl lo), 64)
3391 // (ctlz_zero_poison hi:lo) -> (umin (ffbh hi), (add (ffbh lo), 32))
3392 // (cttz_zero_poison hi:lo) -> (umin (add (ffbl hi), 32), (ffbl lo))
3393
3394 unsigned AddOpc = ZeroUndef ? ISD::ADD : ISD::UADDSAT;
3395 const SDValue Const32 = DAG.getConstant(32, SL, MVT::i32);
3396 if (Ctlz)
3397 OprLo = DAG.getNode(AddOpc, SL, MVT::i32, OprLo, Const32);
3398 else
3399 OprHi = DAG.getNode(AddOpc, SL, MVT::i32, OprHi, Const32);
3400
3401 SDValue NewOpr;
3402 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, OprLo, OprHi);
3403 if (!ZeroUndef) {
3404 const SDValue Const64 = DAG.getConstant(64, SL, MVT::i32);
3405 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, NewOpr, Const64);
3406 }
3407
3408 return DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i64, NewOpr);
3409}
3410
3412 SDLoc SL(Op);
3413 SDValue Src = Op.getOperand(0);
3414 assert(Src.getValueType() == MVT::i32 && "LowerCTLS only supports i32");
3415 SDValue Ffbh = DAG.getNode(
3416 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
3417 DAG.getTargetConstant(Intrinsic::amdgcn_sffbh, SL, MVT::i32), Src);
3418 SDValue Clamped = DAG.getNode(ISD::UMIN, SL, MVT::i32, Ffbh,
3419 DAG.getConstant(32, SL, MVT::i32));
3420 return DAG.getNode(ISD::ADD, SL, MVT::i32, Clamped,
3421 DAG.getAllOnesConstant(SL, MVT::i32));
3422}
3423
3425 EVT FP16Ty) const {
3426 assert(FP16Ty == MVT::f16 || FP16Ty == MVT::bf16);
3427 SDLoc SL(Op);
3428 SDValue Src = Op.getOperand(0);
3429 SDValue ToF32 = DAG.getNode(Op.getOpcode(), SL, MVT::f32, Src);
3430 SDValue FPRoundFlag = DAG.getIntPtrConstant(0, SL, /*isTarget=*/true);
3431 return DAG.getNode(ISD::FP_ROUND, SL, FP16Ty, ToF32, FPRoundFlag);
3432}
3433
3435 bool Signed) const {
3436 // The regular method converting a 64-bit integer to float roughly consists of
3437 // 2 steps: normalization and rounding. In fact, after normalization, the
3438 // conversion from a 64-bit integer to a float is essentially the same as the
3439 // one from a 32-bit integer. The only difference is that it has more
3440 // trailing bits to be rounded. To leverage the native 32-bit conversion, a
3441 // 64-bit integer could be preprocessed and fit into a 32-bit integer then
3442 // converted into the correct float number. The basic steps for the unsigned
3443 // conversion are illustrated in the following pseudo code:
3444 //
3445 // f32 uitofp(i64 u) {
3446 // i32 hi, lo = split(u);
3447 // // Only count the leading zeros in hi as we have native support of the
3448 // // conversion from i32 to f32. If hi is all 0s, the conversion is
3449 // // reduced to a 32-bit one automatically.
3450 // i32 shamt = clz(hi); // Return 32 if hi is all 0s.
3451 // u <<= shamt;
3452 // hi, lo = split(u);
3453 // hi |= (lo != 0) ? 1 : 0; // Adjust rounding bit in hi based on lo.
3454 // // convert it as a 32-bit integer and scale the result back.
3455 // return uitofp(hi) * 2^(32 - shamt);
3456 // }
3457 //
3458 // The signed one follows the same principle but uses 'ffbh_i32' to count its
3459 // sign bits instead. If 'ffbh_i32' is not available, its absolute value is
3460 // converted instead followed by negation based its sign bit.
3461
3462 SDLoc SL(Op);
3463 SDValue Src = Op.getOperand(0);
3464
3465 SDValue Lo, Hi;
3466 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3467 SDValue Sign;
3468 SDValue ShAmt;
3469 if (Signed && Subtarget->isGCN()) {
3470 // We also need to consider the sign bit in Lo if Hi has just sign bits,
3471 // i.e. Hi is 0 or -1. However, that only needs to take the MSB into
3472 // account. That is, the maximal shift is
3473 // - 32 if Lo and Hi have opposite signs;
3474 // - 33 if Lo and Hi have the same sign.
3475 //
3476 // Or, MaxShAmt = 33 + OppositeSign, where
3477 //
3478 // OppositeSign is defined as ((Lo ^ Hi) >> 31), which is
3479 // - -1 if Lo and Hi have opposite signs; and
3480 // - 0 otherwise.
3481 //
3482 // All in all, ShAmt is calculated as
3483 //
3484 // umin(sffbh(Hi), 33 + (Lo^Hi)>>31) - 1.
3485 //
3486 // or
3487 //
3488 // umin(sffbh(Hi) - 1, 32 + (Lo^Hi)>>31).
3489 //
3490 // to reduce the critical path.
3491 SDValue OppositeSign = DAG.getNode(
3492 ISD::SRA, SL, MVT::i32, DAG.getNode(ISD::XOR, SL, MVT::i32, Lo, Hi),
3493 DAG.getConstant(31, SL, MVT::i32));
3494 SDValue MaxShAmt =
3495 DAG.getNode(ISD::ADD, SL, MVT::i32, DAG.getConstant(32, SL, MVT::i32),
3496 OppositeSign);
3497 // Count the leading sign bits.
3498 ShAmt = DAG.getNode(
3499 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
3500 DAG.getTargetConstant(Intrinsic::amdgcn_sffbh, SL, MVT::i32), Hi);
3501 // Different from unsigned conversion, the shift should be one bit less to
3502 // preserve the sign bit.
3503 ShAmt = DAG.getNode(ISD::SUB, SL, MVT::i32, ShAmt,
3504 DAG.getConstant(1, SL, MVT::i32));
3505 ShAmt = DAG.getNode(ISD::UMIN, SL, MVT::i32, ShAmt, MaxShAmt);
3506 } else {
3507 if (Signed) {
3508 // Without 'ffbh_i32', only leading zeros could be counted. Take the
3509 // absolute value first.
3510 Sign = DAG.getNode(ISD::SRA, SL, MVT::i64, Src,
3511 DAG.getConstant(63, SL, MVT::i64));
3512 SDValue Abs =
3513 DAG.getNode(ISD::XOR, SL, MVT::i64,
3514 DAG.getNode(ISD::ADD, SL, MVT::i64, Src, Sign), Sign);
3515 std::tie(Lo, Hi) = split64BitValue(Abs, DAG);
3516 }
3517 // Count the leading zeros.
3518 ShAmt = DAG.getNode(ISD::CTLZ, SL, MVT::i32, Hi);
3519 // The shift amount for signed integers is [0, 32].
3520 }
3521 // Normalize the given 64-bit integer.
3522 SDValue Norm = DAG.getNode(ISD::SHL, SL, MVT::i64, Src, ShAmt);
3523 // Split it again.
3524 std::tie(Lo, Hi) = split64BitValue(Norm, DAG);
3525 // Calculate the adjust bit for rounding.
3526 // (lo != 0) ? 1 : 0 => (lo >= 1) ? 1 : 0 => umin(1, lo)
3527 SDValue Adjust = DAG.getNode(ISD::UMIN, SL, MVT::i32,
3528 DAG.getConstant(1, SL, MVT::i32), Lo);
3529 // Get the 32-bit normalized integer.
3530 Norm = DAG.getNode(ISD::OR, SL, MVT::i32, Hi, Adjust);
3531 // Convert the normalized 32-bit integer into f32.
3532
3533 bool UseLDEXP = isOperationLegal(ISD::FLDEXP, MVT::f32);
3534 unsigned Opc = Signed && UseLDEXP ? ISD::SINT_TO_FP : ISD::UINT_TO_FP;
3535 SDValue FVal = DAG.getNode(Opc, SL, MVT::f32, Norm);
3536
3537 // Finally, need to scale back the converted floating number as the original
3538 // 64-bit integer is converted as a 32-bit one.
3539 ShAmt = DAG.getNode(ISD::SUB, SL, MVT::i32, DAG.getConstant(32, SL, MVT::i32),
3540 ShAmt);
3541 // On GCN, use LDEXP directly.
3542 if (UseLDEXP)
3543 return DAG.getNode(ISD::FLDEXP, SL, MVT::f32, FVal, ShAmt);
3544
3545 // Otherwise, align 'ShAmt' to the exponent part and add it into the exponent
3546 // part directly to emulate the multiplication of 2^ShAmt. That 8-bit
3547 // exponent is enough to avoid overflowing into the sign bit.
3548 SDValue Exp = DAG.getNode(ISD::SHL, SL, MVT::i32, ShAmt,
3549 DAG.getConstant(23, SL, MVT::i32));
3550 SDValue IVal =
3551 DAG.getNode(ISD::ADD, SL, MVT::i32,
3552 DAG.getNode(ISD::BITCAST, SL, MVT::i32, FVal), Exp);
3553 if (Signed) {
3554 // Set the sign bit.
3555 Sign = DAG.getNode(ISD::SHL, SL, MVT::i32,
3556 DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Sign),
3557 DAG.getConstant(31, SL, MVT::i32));
3558 IVal = DAG.getNode(ISD::OR, SL, MVT::i32, IVal, Sign);
3559 }
3560 return DAG.getNode(ISD::BITCAST, SL, MVT::f32, IVal);
3561}
3562
3564 bool Signed) const {
3565 SDLoc SL(Op);
3566 SDValue Src = Op.getOperand(0);
3567
3568 SDValue Lo, Hi;
3569 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3570
3572 SL, MVT::f64, Hi);
3573
3574 SDValue CvtLo = DAG.getNode(ISD::UINT_TO_FP, SL, MVT::f64, Lo);
3575
3576 SDValue LdExp = DAG.getNode(ISD::FLDEXP, SL, MVT::f64, CvtHi,
3577 DAG.getConstant(32, SL, MVT::i32));
3578 // TODO: Should this propagate fast-math-flags?
3579 return DAG.getNode(ISD::FADD, SL, MVT::f64, LdExp, CvtLo);
3580}
3581
3583 SelectionDAG &DAG) const {
3584 // TODO: Factor out code common with LowerSINT_TO_FP.
3585 EVT DestVT = Op.getValueType();
3586 SDValue Src = Op.getOperand(0);
3587 EVT SrcVT = Src.getValueType();
3588
3589 if (SrcVT == MVT::i16) {
3590 if (DestVT == MVT::f16)
3591 return Op;
3592 SDLoc DL(Op);
3593
3594 // Promote src to i32
3595 SDValue Ext = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, Src);
3596 return DAG.getNode(ISD::UINT_TO_FP, DL, DestVT, Ext);
3597 }
3598
3599 if (DestVT == MVT::bf16 || DestVT == MVT::f16)
3600 return LowerINT_TO_FP16(Op, DAG, DestVT);
3601
3602 if (SrcVT != MVT::i64)
3603 return Op;
3604
3605 if (DestVT == MVT::f32)
3606 return LowerINT_TO_FP32(Op, DAG, false);
3607
3608 assert(DestVT == MVT::f64);
3609 return LowerINT_TO_FP64(Op, DAG, false);
3610}
3611
3613 SelectionDAG &DAG) const {
3614 EVT DestVT = Op.getValueType();
3615
3616 SDValue Src = Op.getOperand(0);
3617 EVT SrcVT = Src.getValueType();
3618
3619 if (SrcVT == MVT::i16) {
3620 if (DestVT == MVT::f16)
3621 return Op;
3622
3623 SDLoc DL(Op);
3624 // Promote src to i32
3625 SDValue Ext = DAG.getNode(ISD::SIGN_EXTEND, DL, MVT::i32, Src);
3626 return DAG.getNode(ISD::SINT_TO_FP, DL, DestVT, Ext);
3627 }
3628
3629 if (DestVT == MVT::bf16 || DestVT == MVT::f16)
3630 return LowerINT_TO_FP16(Op, DAG, DestVT);
3631
3632 if (SrcVT != MVT::i64)
3633 return Op;
3634
3635 // TODO: Factor out code common with LowerUINT_TO_FP.
3636
3637 if (DestVT == MVT::f32)
3638 return LowerINT_TO_FP32(Op, DAG, true);
3639
3640 assert(DestVT == MVT::f64);
3641 return LowerINT_TO_FP64(Op, DAG, true);
3642}
3643
3645 bool Signed) const {
3646 SDLoc SL(Op);
3647
3648 SDValue Src = Op.getOperand(0);
3649 EVT SrcVT = Src.getValueType();
3650
3651 assert(SrcVT == MVT::f32 || SrcVT == MVT::f64);
3652
3653 // The basic idea of converting a floating point number into a pair of 32-bit
3654 // integers is illustrated as follows:
3655 //
3656 // tf := trunc(val);
3657 // hif := floor(tf * 2^-32);
3658 // lof := tf - hif * 2^32; // lof is always positive due to floor.
3659 // hi := fptoi(hif);
3660 // lo := fptoi(lof);
3661 //
3662 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, SrcVT, Src);
3663 SDValue Sign;
3664 if (Signed && SrcVT == MVT::f32) {
3665 // However, a 32-bit floating point number has only 23 bits mantissa and
3666 // it's not enough to hold all the significant bits of `lof` if val is
3667 // negative. To avoid the loss of precision, We need to take the absolute
3668 // value after truncating and flip the result back based on the original
3669 // signedness.
3670 Sign = DAG.getNode(ISD::SRA, SL, MVT::i32,
3671 DAG.getNode(ISD::BITCAST, SL, MVT::i32, Trunc),
3672 DAG.getConstant(31, SL, MVT::i32));
3673 Trunc = DAG.getNode(ISD::FABS, SL, SrcVT, Trunc);
3674 }
3675
3676 SDValue K0, K1;
3677 if (SrcVT == MVT::f64) {
3678 K0 = DAG.getConstantFP(
3679 llvm::bit_cast<double>(UINT64_C(/*2^-32*/ 0x3df0000000000000)), SL,
3680 SrcVT);
3681 K1 = DAG.getConstantFP(
3682 llvm::bit_cast<double>(UINT64_C(/*-2^32*/ 0xc1f0000000000000)), SL,
3683 SrcVT);
3684 } else {
3685 K0 = DAG.getConstantFP(
3686 llvm::bit_cast<float>(UINT32_C(/*2^-32*/ 0x2f800000)), SL, SrcVT);
3687 K1 = DAG.getConstantFP(
3688 llvm::bit_cast<float>(UINT32_C(/*-2^32*/ 0xcf800000)), SL, SrcVT);
3689 }
3690 // TODO: Should this propagate fast-math-flags?
3691 SDValue Mul = DAG.getNode(ISD::FMUL, SL, SrcVT, Trunc, K0);
3692
3693 SDValue FloorMul = DAG.getNode(ISD::FFLOOR, SL, SrcVT, Mul);
3694
3695 SDValue Fma = DAG.getNode(ISD::FMA, SL, SrcVT, FloorMul, K1, Trunc);
3696
3697 SDValue Hi = DAG.getNode((Signed && SrcVT == MVT::f64) ? ISD::FP_TO_SINT
3699 SL, MVT::i32, FloorMul);
3700 SDValue Lo = DAG.getNode(ISD::FP_TO_UINT, SL, MVT::i32, Fma);
3701
3702 SDValue Result = DAG.getNode(ISD::BITCAST, SL, MVT::i64,
3703 DAG.getBuildVector(MVT::v2i32, SL, {Lo, Hi}));
3704
3705 if (Signed && SrcVT == MVT::f32) {
3706 assert(Sign);
3707 // Flip the result based on the signedness, which is either all 0s or 1s.
3708 Sign = DAG.getNode(ISD::BITCAST, SL, MVT::i64,
3709 DAG.getBuildVector(MVT::v2i32, SL, {Sign, Sign}));
3710 // r := xor(r, sign) - sign;
3711 Result =
3712 DAG.getNode(ISD::SUB, SL, MVT::i64,
3713 DAG.getNode(ISD::XOR, SL, MVT::i64, Result, Sign), Sign);
3714 }
3715
3716 return Result;
3717}
3718
3720 SDLoc DL(Op);
3721 SDValue N0 = Op.getOperand(0);
3722
3723 // Convert to target node to get known bits
3724 if (N0.getValueType() == MVT::f32)
3725 return DAG.getNode(AMDGPUISD::FP_TO_FP16, DL, Op.getValueType(), N0);
3726
3727 if (Op->getFlags().hasApproximateFuncs()) {
3728 // There is a generic expand for FP_TO_FP16 with unsafe fast math.
3729 return SDValue();
3730 }
3731
3732 return LowerF64ToF16Safe(N0, DL, DAG);
3733}
3734
3735// return node in i32
3737 SelectionDAG &DAG) const {
3738 assert(Src.getSimpleValueType() == MVT::f64);
3739
3740 // f64 -> f16 conversion using round-to-nearest-even rounding mode.
3741 // TODO: We can generate better code for True16.
3742 const unsigned ExpMask = 0x7ff;
3743 const unsigned ExpBiasf64 = 1023;
3744 const unsigned ExpBiasf16 = 15;
3745 SDValue Zero = DAG.getConstant(0, DL, MVT::i32);
3746 SDValue One = DAG.getConstant(1, DL, MVT::i32);
3747 SDValue U = DAG.getNode(ISD::BITCAST, DL, MVT::i64, Src);
3748 SDValue UH = DAG.getNode(ISD::SRL, DL, MVT::i64, U,
3749 DAG.getConstant(32, DL, MVT::i64));
3750 UH = DAG.getZExtOrTrunc(UH, DL, MVT::i32);
3751 U = DAG.getZExtOrTrunc(U, DL, MVT::i32);
3752 SDValue E = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3753 DAG.getConstant(20, DL, MVT::i64));
3754 E = DAG.getNode(ISD::AND, DL, MVT::i32, E,
3755 DAG.getConstant(ExpMask, DL, MVT::i32));
3756 // Subtract the fp64 exponent bias (1023) to get the real exponent and
3757 // add the f16 bias (15) to get the biased exponent for the f16 format.
3758 E = DAG.getNode(ISD::ADD, DL, MVT::i32, E,
3759 DAG.getConstant(-ExpBiasf64 + ExpBiasf16, DL, MVT::i32));
3760
3761 SDValue M = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3762 DAG.getConstant(8, DL, MVT::i32));
3763 M = DAG.getNode(ISD::AND, DL, MVT::i32, M,
3764 DAG.getConstant(0xffe, DL, MVT::i32));
3765
3766 SDValue MaskedSig = DAG.getNode(ISD::AND, DL, MVT::i32, UH,
3767 DAG.getConstant(0x1ff, DL, MVT::i32));
3768 MaskedSig = DAG.getNode(ISD::OR, DL, MVT::i32, MaskedSig, U);
3769
3770 SDValue Lo40Set = DAG.getSelectCC(DL, MaskedSig, Zero, Zero, One, ISD::SETEQ);
3771 M = DAG.getNode(ISD::OR, DL, MVT::i32, M, Lo40Set);
3772
3773 // (M != 0 ? 0x0200 : 0) | 0x7c00;
3774 SDValue I = DAG.getNode(ISD::OR, DL, MVT::i32,
3775 DAG.getSelectCC(DL, M, Zero, DAG.getConstant(0x0200, DL, MVT::i32),
3776 Zero, ISD::SETNE), DAG.getConstant(0x7c00, DL, MVT::i32));
3777
3778 // N = M | (E << 12);
3779 SDValue N = DAG.getNode(ISD::OR, DL, MVT::i32, M,
3780 DAG.getNode(ISD::SHL, DL, MVT::i32, E,
3781 DAG.getConstant(12, DL, MVT::i32)));
3782
3783 // B = clamp(1-E, 0, 13);
3784 SDValue OneSubExp = DAG.getNode(ISD::SUB, DL, MVT::i32,
3785 One, E);
3786 SDValue B = DAG.getNode(ISD::SMAX, DL, MVT::i32, OneSubExp, Zero);
3787 B = DAG.getNode(ISD::SMIN, DL, MVT::i32, B,
3788 DAG.getConstant(13, DL, MVT::i32));
3789
3790 SDValue SigSetHigh = DAG.getNode(ISD::OR, DL, MVT::i32, M,
3791 DAG.getConstant(0x1000, DL, MVT::i32));
3792
3793 SDValue D = DAG.getNode(ISD::SRL, DL, MVT::i32, SigSetHigh, B);
3794 SDValue D0 = DAG.getNode(ISD::SHL, DL, MVT::i32, D, B);
3795 SDValue D1 = DAG.getSelectCC(DL, D0, SigSetHigh, One, Zero, ISD::SETNE);
3796 D = DAG.getNode(ISD::OR, DL, MVT::i32, D, D1);
3797
3798 SDValue V = DAG.getSelectCC(DL, E, One, D, N, ISD::SETLT);
3799 SDValue VLow3 = DAG.getNode(ISD::AND, DL, MVT::i32, V,
3800 DAG.getConstant(0x7, DL, MVT::i32));
3801 V = DAG.getNode(ISD::SRL, DL, MVT::i32, V,
3802 DAG.getConstant(2, DL, MVT::i32));
3803 SDValue V0 = DAG.getSelectCC(DL, VLow3, DAG.getConstant(3, DL, MVT::i32),
3804 One, Zero, ISD::SETEQ);
3805 SDValue V1 = DAG.getSelectCC(DL, VLow3, DAG.getConstant(5, DL, MVT::i32),
3806 One, Zero, ISD::SETGT);
3807 V1 = DAG.getNode(ISD::OR, DL, MVT::i32, V0, V1);
3808 V = DAG.getNode(ISD::ADD, DL, MVT::i32, V, V1);
3809
3810 V = DAG.getSelectCC(DL, E, DAG.getConstant(30, DL, MVT::i32),
3811 DAG.getConstant(0x7c00, DL, MVT::i32), V, ISD::SETGT);
3812 V = DAG.getSelectCC(DL, E, DAG.getConstant(1039, DL, MVT::i32),
3813 I, V, ISD::SETEQ);
3814
3815 // Extract the sign bit.
3816 SDValue Sign = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3817 DAG.getConstant(16, DL, MVT::i32));
3818 Sign = DAG.getNode(ISD::AND, DL, MVT::i32, Sign,
3819 DAG.getConstant(0x8000, DL, MVT::i32));
3820
3821 return DAG.getNode(ISD::OR, DL, MVT::i32, Sign, V);
3822}
3823
3825 SelectionDAG &DAG) const {
3826 SDValue Src = Op.getOperand(0);
3827 unsigned OpOpcode = Op.getOpcode();
3828 EVT SrcVT = Src.getValueType();
3829 EVT DestVT = Op.getValueType();
3830
3831 // Will be selected natively
3832 if (SrcVT == MVT::f16 && DestVT == MVT::i16)
3833 return Op;
3834
3835 if (SrcVT == MVT::bf16 || (SrcVT == MVT::f16 && DestVT == MVT::i32)) {
3836 SDLoc DL(Op);
3837 SDValue PromotedSrc = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
3838 return DAG.getNode(Op.getOpcode(), DL, DestVT, PromotedSrc);
3839 }
3840
3841 // Promote i16 to i32
3842 if (DestVT == MVT::i16 && (SrcVT == MVT::f32 || SrcVT == MVT::f64)) {
3843 SDLoc DL(Op);
3844
3845 SDValue FpToInt32 = DAG.getNode(OpOpcode, DL, MVT::i32, Src);
3846 return DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, FpToInt32);
3847 }
3848
3849 if (DestVT != MVT::i64)
3850 return Op;
3851
3852 if (SrcVT == MVT::f16 ||
3853 (SrcVT == MVT::f32 && Src.getOpcode() == ISD::FP16_TO_FP)) {
3854 SDLoc DL(Op);
3855
3856 SDValue FpToInt32 = DAG.getNode(OpOpcode, DL, MVT::i32, Src);
3857 unsigned Ext =
3859 return DAG.getNode(Ext, DL, MVT::i64, FpToInt32);
3860 }
3861
3862 if (SrcVT == MVT::f32 || SrcVT == MVT::f64)
3863 return LowerFP_TO_INT64(Op, DAG, OpOpcode == ISD::FP_TO_SINT);
3864
3865 return SDValue();
3866}
3867
3869 SelectionDAG &DAG) const {
3870 SDValue Src = Op.getOperand(0);
3871 unsigned OpOpcode = Op.getOpcode();
3872 EVT SrcVT = Src.getValueType();
3873 EVT DstVT = Op.getValueType();
3874 SDValue SatVTOp = Op.getNode()->getOperand(1);
3875 EVT SatVT = cast<VTSDNode>(SatVTOp)->getVT();
3876 SDLoc DL(Op);
3877
3878 uint64_t DstWidth = DstVT.getScalarSizeInBits();
3879 uint64_t SatWidth = SatVT.getScalarSizeInBits();
3880 assert(SatWidth <= DstWidth && "Saturation width cannot exceed result width");
3881
3882 // Scalar cases will be selected natively to v_cvt_/s_cvt_ instructions.
3883 // v2f32 -> v2i16 will be selected natively to v_cvt_pk_[iu]16_f32.
3884 if (SatWidth == DstWidth) {
3885 if ((DstVT == MVT::i32 && (SrcVT == MVT::f32 || SrcVT == MVT::f64)) ||
3886 (DstVT == MVT::i16 && (SrcVT == MVT::f16 || SrcVT == MVT::f32)) ||
3887 (DstVT == MVT::v2i16 && SrcVT == MVT::v2f32))
3888 return Op;
3889 }
3890
3891 // Vectors can only be selected natively.
3892 if (DstVT.isVector())
3893 return SDValue();
3894
3895 // Perform all saturation at selected width (i16 or i32) and truncate
3896 if (SatWidth < DstWidth && SatWidth <= 32) {
3897 // For f16 conversion with sub-i16 saturation perform saturation
3898 // at i16, if available in the target. This removes the need for extra f16
3899 // to f32 conversion. For all the others use i32.
3900 MVT ResultVT =
3901 Subtarget->has16BitInsts() && SrcVT == MVT::f16 && SatWidth < 16
3902 ? MVT::i16
3903 : MVT::i32;
3904
3905 const SDValue ResultVTOp = DAG.getValueType(ResultVT);
3906 const uint64_t ResultWidth = ResultVT.getScalarSizeInBits();
3907
3908 // First, convert input float into selected integer (i16 or i32)
3909 SDValue FpToInt = DAG.getNode(OpOpcode, DL, ResultVT, Src, ResultVTOp);
3910 SDValue IntSatVal;
3911
3912 // Then, clamp at the saturation width using either i16 or i32 instructions
3913 if (OpOpcode == ISD::FP_TO_SINT_SAT) {
3914 SDValue MinConst = DAG.getConstant(
3915 APInt::getSignedMaxValue(SatWidth).sext(ResultWidth), DL, ResultVT);
3916 SDValue MaxConst = DAG.getConstant(
3917 APInt::getSignedMinValue(SatWidth).sext(ResultWidth), DL, ResultVT);
3918 SDValue MinVal = DAG.getNode(ISD::SMIN, DL, ResultVT, FpToInt, MinConst);
3919 IntSatVal = DAG.getNode(ISD::SMAX, DL, ResultVT, MinVal, MaxConst);
3920 } else {
3921 SDValue MinConst = DAG.getConstant(
3922 APInt::getMaxValue(SatWidth).zext(ResultWidth), DL, ResultVT);
3923 IntSatVal = DAG.getNode(ISD::UMIN, DL, ResultVT, FpToInt, MinConst);
3924 }
3925
3926 // Finally, after saturating at i16 or i32 fit into the destination type
3927 return DAG.getExtOrTrunc(OpOpcode == ISD::FP_TO_SINT_SAT, IntSatVal, DL,
3928 DstVT);
3929 }
3930
3931 // SatWidth == DstWidth or SatWidth > 32
3932
3933 // Saturate at i32 for i64 dst and f16/bf16 src (will invoke f16 promotion
3934 // below)
3935 if (DstVT == MVT::i64 &&
3936 (SrcVT == MVT::f16 || SrcVT == MVT::bf16 ||
3937 (SrcVT == MVT::f32 && Src.getOpcode() == ISD::FP16_TO_FP))) {
3938 const SDValue Int32VTOp = DAG.getValueType(MVT::i32);
3939 return DAG.getNode(OpOpcode, DL, DstVT, Src, Int32VTOp);
3940 }
3941
3942 // Promote f16/bf16 src to f32 for i32 conversion
3943 if (DstVT == MVT::i32 && (SrcVT == MVT::f16 || SrcVT == MVT::bf16)) {
3944 SDValue PromotedSrc = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
3945 return DAG.getNode(Op.getOpcode(), DL, DstVT, PromotedSrc, SatVTOp);
3946 }
3947
3948 // For DstWidth < 16, promote i1 and i8 dst to i16 (if legal) with sub-i16
3949 // saturation. For DstWidth == 16, promote i16 dst to i32 with sub-i32
3950 // saturation; this covers i16.f32 and i16.f64
3951 if (DstWidth < 32) {
3952 // Note: this triggers SatWidth < DstWidth above to generate saturated
3953 // truncate by requesting MVT::i16/i32 destination with SatWidth < 16/32.
3954 MVT PromoteVT =
3955 (DstWidth < 16 && Subtarget->has16BitInsts()) ? MVT::i16 : MVT::i32;
3956 SDValue FpToInt = DAG.getNode(OpOpcode, DL, PromoteVT, Src, SatVTOp);
3957 return DAG.getNode(ISD::TRUNCATE, DL, DstVT, FpToInt);
3958 }
3959
3960 // TODO: can we implement i64 dst for f32/f64?
3961
3962 return SDValue();
3963}
3964
3966 SelectionDAG &DAG) const {
3967 EVT ExtraVT = cast<VTSDNode>(Op.getOperand(1))->getVT();
3968 MVT VT = Op.getSimpleValueType();
3969 MVT ScalarVT = VT.getScalarType();
3970
3971 assert(VT.isVector());
3972
3973 SDValue Src = Op.getOperand(0);
3974 SDLoc DL(Op);
3975
3976 // TODO: Don't scalarize on Evergreen?
3977 unsigned NElts = VT.getVectorNumElements();
3979 DAG.ExtractVectorElements(Src, Args, 0, NElts);
3980
3981 SDValue VTOp = DAG.getValueType(ExtraVT.getScalarType());
3982 for (unsigned I = 0; I < NElts; ++I)
3983 Args[I] = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, ScalarVT, Args[I], VTOp);
3984
3985 return DAG.getBuildVector(VT, DL, Args);
3986}
3987
3988//===----------------------------------------------------------------------===//
3989// Custom DAG optimizations
3990//===----------------------------------------------------------------------===//
3991
3992static bool isU24(SDValue Op, SelectionDAG &DAG) {
3993 return AMDGPUTargetLowering::numBitsUnsigned(Op, DAG) <= 24;
3994}
3995
3996static bool isI24(SDValue Op, SelectionDAG &DAG) {
3997 EVT VT = Op.getValueType();
3998 return VT.getSizeInBits() >= 24 && // Types less than 24-bit should be treated
3999 // as unsigned 24-bit values.
4001}
4002
4005 SelectionDAG &DAG = DCI.DAG;
4006 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
4007 bool IsIntrin = Node24->getOpcode() == ISD::INTRINSIC_WO_CHAIN;
4008
4009 SDValue LHS = IsIntrin ? Node24->getOperand(1) : Node24->getOperand(0);
4010 SDValue RHS = IsIntrin ? Node24->getOperand(2) : Node24->getOperand(1);
4011 unsigned NewOpcode = Node24->getOpcode();
4012 if (IsIntrin) {
4013 unsigned IID = Node24->getConstantOperandVal(0);
4014 switch (IID) {
4015 case Intrinsic::amdgcn_mul_i24:
4016 NewOpcode = AMDGPUISD::MUL_I24;
4017 break;
4018 case Intrinsic::amdgcn_mul_u24:
4019 NewOpcode = AMDGPUISD::MUL_U24;
4020 break;
4021 case Intrinsic::amdgcn_mulhi_i24:
4022 NewOpcode = AMDGPUISD::MULHI_I24;
4023 break;
4024 case Intrinsic::amdgcn_mulhi_u24:
4025 NewOpcode = AMDGPUISD::MULHI_U24;
4026 break;
4027 default:
4028 llvm_unreachable("Expected 24-bit mul intrinsic");
4029 }
4030 }
4031
4032 APInt Demanded = APInt::getLowBitsSet(LHS.getValueSizeInBits(), 24);
4033
4034 // First try to simplify using SimplifyMultipleUseDemandedBits which allows
4035 // the operands to have other uses, but will only perform simplifications that
4036 // involve bypassing some nodes for this user.
4037 SDValue DemandedLHS = TLI.SimplifyMultipleUseDemandedBits(LHS, Demanded, DAG);
4038 SDValue DemandedRHS = TLI.SimplifyMultipleUseDemandedBits(RHS, Demanded, DAG);
4039 if (DemandedLHS || DemandedRHS)
4040 return DAG.getNode(NewOpcode, SDLoc(Node24), Node24->getVTList(),
4041 DemandedLHS ? DemandedLHS : LHS,
4042 DemandedRHS ? DemandedRHS : RHS);
4043
4044 // Now try SimplifyDemandedBits which can simplify the nodes used by our
4045 // operands if this node is the only user.
4046 if (TLI.SimplifyDemandedBits(LHS, Demanded, DCI))
4047 return SDValue(Node24, 0);
4048 if (TLI.SimplifyDemandedBits(RHS, Demanded, DCI))
4049 return SDValue(Node24, 0);
4050
4051 return SDValue();
4052}
4053
4054template <typename IntTy>
4056 uint32_t Width, const SDLoc &DL) {
4057 if (Width + Offset < 32) {
4058 uint32_t Shl = static_cast<uint32_t>(Src0) << (32 - Offset - Width);
4059 IntTy Result = static_cast<IntTy>(Shl) >> (32 - Width);
4060 if constexpr (std::is_signed_v<IntTy>) {
4061 return DAG.getSignedConstant(Result, DL, MVT::i32);
4062 } else {
4063 return DAG.getConstant(Result, DL, MVT::i32);
4064 }
4065 }
4066
4067 return DAG.getConstant(Src0 >> Offset, DL, MVT::i32);
4068}
4069
4070static bool hasVolatileUser(SDNode *Val) {
4071 for (SDNode *U : Val->users()) {
4072 if (MemSDNode *M = dyn_cast<MemSDNode>(U)) {
4073 if (M->isVolatile())
4074 return true;
4075 }
4076 }
4077
4078 return false;
4079}
4080
4082 // i32 vectors are the canonical memory type.
4083 if (VT.getScalarType() == MVT::i32 || isTypeLegal(VT))
4084 return false;
4085
4086 if (!VT.isByteSized())
4087 return false;
4088
4089 unsigned Size = VT.getStoreSize();
4090
4091 if ((Size == 1 || Size == 2 || Size == 4) && !VT.isVector())
4092 return false;
4093
4094 if (Size == 3 || (Size > 4 && (Size % 4 != 0)))
4095 return false;
4096
4097 return true;
4098}
4099
4100// Replace load of an illegal type with a bitcast from a load of a friendlier
4101// type.
4103 DAGCombinerInfo &DCI) const {
4104 if (!DCI.isBeforeLegalize())
4105 return SDValue();
4106
4108 if (!LN->isSimple() || !ISD::isNormalLoad(LN) || hasVolatileUser(LN))
4109 return SDValue();
4110
4111 SDLoc SL(N);
4112 SelectionDAG &DAG = DCI.DAG;
4113 EVT VT = LN->getMemoryVT();
4114
4115 unsigned Size = VT.getStoreSize();
4116 Align Alignment = LN->getAlign();
4117 if (Alignment < Size && isTypeLegal(VT)) {
4118 unsigned IsFast;
4119 unsigned AS = LN->getAddressSpace();
4120
4121 // Expand unaligned loads earlier than legalization. Due to visitation order
4122 // problems during legalization, the emitted instructions to pack and unpack
4123 // the bytes again are not eliminated in the case of an unaligned copy.
4125 VT, AS, Alignment, LN->getMemOperand()->getFlags(), &IsFast)) {
4126 if (VT.isVector())
4127 return SplitVectorLoad(SDValue(LN, 0), DAG);
4128
4129 SDValue Ops[2];
4130 std::tie(Ops[0], Ops[1]) = expandUnalignedLoad(LN, DAG);
4131
4132 return DAG.getMergeValues(Ops, SDLoc(N));
4133 }
4134
4135 if (!IsFast)
4136 return SDValue();
4137 }
4138
4139 if (!shouldCombineMemoryType(VT))
4140 return SDValue();
4141
4142 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
4143
4144 SDValue NewLoad
4145 = DAG.getLoad(NewVT, SL, LN->getChain(),
4146 LN->getBasePtr(), LN->getMemOperand());
4147
4148 SDValue BC = DAG.getNode(ISD::BITCAST, SL, VT, NewLoad);
4149 DCI.CombineTo(N, BC, NewLoad.getValue(1));
4150 return SDValue(N, 0);
4151}
4152
4153// Replace store of an illegal type with a store of a bitcast to a friendlier
4154// type.
4156 DAGCombinerInfo &DCI) const {
4157 if (!DCI.isBeforeLegalize())
4158 return SDValue();
4159
4161 if (!SN->isSimple() || !ISD::isNormalStore(SN))
4162 return SDValue();
4163
4164 EVT VT = SN->getMemoryVT();
4165 unsigned Size = VT.getStoreSize();
4166
4167 SDLoc SL(N);
4168 SelectionDAG &DAG = DCI.DAG;
4169 Align Alignment = SN->getAlign();
4170 if (Alignment < Size && isTypeLegal(VT)) {
4171 unsigned IsFast;
4172 unsigned AS = SN->getAddressSpace();
4173
4174 // Expand unaligned stores earlier than legalization. Due to visitation
4175 // order problems during legalization, the emitted instructions to pack and
4176 // unpack the bytes again are not eliminated in the case of an unaligned
4177 // copy.
4179 VT, AS, Alignment, SN->getMemOperand()->getFlags(), &IsFast)) {
4180 if (VT.isVector())
4181 return SplitVectorStore(SDValue(SN, 0), DAG);
4182
4183 return expandUnalignedStore(SN, DAG);
4184 }
4185
4186 if (!IsFast)
4187 return SDValue();
4188 }
4189
4190 if (!shouldCombineMemoryType(VT))
4191 return SDValue();
4192
4193 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
4194 SDValue Val = SN->getValue();
4195
4196 // DCI.AddToWorklist(Val.getNode());
4197
4198 bool OtherUses = !Val.hasOneUse();
4199 SDValue CastVal = DAG.getBitcast(NewVT, Val);
4200 if (OtherUses) {
4201 SDValue CastBack = DAG.getBitcast(VT, CastVal);
4202 DAG.ReplaceAllUsesOfValueWith(Val, CastBack);
4203 }
4204
4205 return DAG.getStore(SN->getChain(), SL, CastVal,
4206 SN->getBasePtr(), SN->getMemOperand());
4207}
4208
4209// FIXME: This should go in generic DAG combiner with an isTruncateFree check,
4210// but isTruncateFree is inaccurate for i16 now because of SALU vs. VALU
4211// issues.
4213 DAGCombinerInfo &DCI) const {
4214 SelectionDAG &DAG = DCI.DAG;
4215 SDValue N0 = N->getOperand(0);
4216
4217 // (vt2 (assertzext (truncate vt0:x), vt1)) ->
4218 // (vt2 (truncate (assertzext vt0:x, vt1)))
4219 if (N0.getOpcode() == ISD::TRUNCATE) {
4220 SDValue N1 = N->getOperand(1);
4221 EVT ExtVT = cast<VTSDNode>(N1)->getVT();
4222 SDLoc SL(N);
4223
4224 SDValue Src = N0.getOperand(0);
4225 EVT SrcVT = Src.getValueType();
4226 if (SrcVT.bitsGE(ExtVT)) {
4227 SDValue NewInReg = DAG.getNode(N->getOpcode(), SL, SrcVT, Src, N1);
4228 return DAG.getNode(ISD::TRUNCATE, SL, N->getValueType(0), NewInReg);
4229 }
4230 }
4231
4232 return SDValue();
4233}
4234
4236 SDNode *N, DAGCombinerInfo &DCI) const {
4237 unsigned IID = N->getConstantOperandVal(0);
4238 switch (IID) {
4239 case Intrinsic::amdgcn_mul_i24:
4240 case Intrinsic::amdgcn_mul_u24:
4241 case Intrinsic::amdgcn_mulhi_i24:
4242 case Intrinsic::amdgcn_mulhi_u24:
4243 return simplifyMul24(N, DCI);
4244 case Intrinsic::amdgcn_fract:
4245 case Intrinsic::amdgcn_rsq:
4246 case Intrinsic::amdgcn_rcp_legacy:
4247 case Intrinsic::amdgcn_rsq_legacy:
4248 case Intrinsic::amdgcn_rsq_clamp:
4249 case Intrinsic::amdgcn_tanh:
4250 case Intrinsic::amdgcn_prng_b32: {
4251 // FIXME: This is probably wrong. If src is an sNaN, it won't be quieted
4252 SDValue Src = N->getOperand(1);
4253 return Src.isUndef() ? Src : SDValue();
4254 }
4255 case Intrinsic::amdgcn_frexp_exp: {
4256 // frexp_exp (fneg x) -> frexp_exp x
4257 // frexp_exp (fabs x) -> frexp_exp x
4258 // frexp_exp (fneg (fabs x)) -> frexp_exp x
4259 SDValue Src = N->getOperand(1);
4260 SDValue PeekSign = peekFPSignOps(Src);
4261 if (PeekSign == Src)
4262 return SDValue();
4263 return SDValue(DCI.DAG.UpdateNodeOperands(N, N->getOperand(0), PeekSign),
4264 0);
4265 }
4266 default:
4267 return SDValue();
4268 }
4269}
4270
4271/// Split the 64-bit value \p LHS into two 32-bit components, and perform the
4272/// binary operation \p Opc to it with the corresponding constant operands.
4274 DAGCombinerInfo &DCI, const SDLoc &SL,
4275 unsigned Opc, SDValue LHS,
4276 uint32_t ValLo, uint32_t ValHi) const {
4277 SelectionDAG &DAG = DCI.DAG;
4278 SDValue Lo, Hi;
4279 std::tie(Lo, Hi) = split64BitValue(LHS, DAG);
4280
4281 SDValue LoRHS = DAG.getConstant(ValLo, SL, MVT::i32);
4282 SDValue HiRHS = DAG.getConstant(ValHi, SL, MVT::i32);
4283
4284 SDValue LoAnd = DAG.getNode(Opc, SL, MVT::i32, Lo, LoRHS);
4285 SDValue HiAnd = DAG.getNode(Opc, SL, MVT::i32, Hi, HiRHS);
4286
4287 // Re-visit the ands. It's possible we eliminated one of them and it could
4288 // simplify the vector.
4289 DCI.AddToWorklist(Lo.getNode());
4290 DCI.AddToWorklist(Hi.getNode());
4291
4292 SDValue Vec = DAG.getBuildVector(MVT::v2i32, SL, {LoAnd, HiAnd});
4293 return DAG.getNode(ISD::BITCAST, SL, MVT::i64, Vec);
4294}
4295
4297 DAGCombinerInfo &DCI) const {
4298 EVT VT = N->getValueType(0);
4299 SDValue LHS = N->getOperand(0);
4300 SDValue RHS = N->getOperand(1);
4302 SDLoc SL(N);
4303 SelectionDAG &DAG = DCI.DAG;
4304
4305 unsigned RHSVal;
4306 if (CRHS) {
4307 RHSVal = CRHS->getZExtValue();
4308 if (!RHSVal)
4309 return LHS;
4310
4311 switch (LHS->getOpcode()) {
4312 default:
4313 break;
4314 case ISD::ZERO_EXTEND:
4315 case ISD::SIGN_EXTEND:
4316 case ISD::ANY_EXTEND: {
4317 SDValue X = LHS->getOperand(0);
4318
4319 if (VT == MVT::i32 && RHSVal == 16 && X.getValueType() == MVT::i16 &&
4320 isOperationLegal(ISD::BUILD_VECTOR, MVT::v2i16)) {
4321 // Prefer build_vector as the canonical form if packed types are legal.
4322 // (shl ([asz]ext i16:x), 16 -> build_vector 0, x
4323 SDValue Vec = DAG.getBuildVector(
4324 MVT::v2i16, SL,
4325 {DAG.getConstant(0, SL, MVT::i16), LHS->getOperand(0)});
4326 return DAG.getNode(ISD::BITCAST, SL, MVT::i32, Vec);
4327 }
4328
4329 // shl (ext x) => zext (shl x), if shift does not overflow int
4330 if (VT != MVT::i64)
4331 break;
4333 unsigned LZ = Known.countMinLeadingZeros();
4334 if (LZ < RHSVal)
4335 break;
4336 EVT XVT = X.getValueType();
4337 SDValue Shl = DAG.getNode(ISD::SHL, SL, XVT, X, SDValue(CRHS, 0));
4338 return DAG.getZExtOrTrunc(Shl, SL, VT);
4339 }
4340 }
4341 }
4342
4343 if (VT.getScalarType() != MVT::i64)
4344 return SDValue();
4345
4346 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4347 // common case, splitting this into a move and a 32-bit shift is faster and
4348 // the same code size.
4349 KnownBits Known = DAG.computeKnownBits(RHS);
4350
4351 EVT ElementType = VT.getScalarType();
4352 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4353 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4354
4355 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4356 return SDValue();
4357 SDValue ShiftAmt;
4358
4359 if (CRHS) {
4360 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4361 TargetType);
4362 } else {
4363 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4364 const SDValue ShiftMask =
4365 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4366 // This AND instruction will clamp out of bounds shift values.
4367 // It will also be removed during later instruction selection.
4368 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4369 }
4370
4371 SDValue Lo = DAG.getNode(ISD::TRUNCATE, SL, TargetType, LHS);
4372 SDValue NewShift =
4373 DAG.getNode(ISD::SHL, SL, TargetType, Lo, ShiftAmt, N->getFlags());
4374
4375 const SDValue Zero = DAG.getConstant(0, SL, TargetScalarType);
4376 SDValue Vec;
4377
4378 if (VT.isVector()) {
4379 EVT ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4380 unsigned NElts = TargetType.getVectorNumElements();
4382 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2, Zero);
4383
4384 DAG.ExtractVectorElements(NewShift, HiOps, 0, NElts);
4385 for (unsigned I = 0; I != NElts; ++I)
4386 HiAndLoOps[2 * I + 1] = HiOps[I];
4387 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4388 } else {
4389 EVT ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4390 Vec = DAG.getBuildVector(ConcatType, SL, {Zero, NewShift});
4391 }
4392 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4393}
4394
4396 DAGCombinerInfo &DCI) const {
4397 SDValue RHS = N->getOperand(1);
4399 EVT VT = N->getValueType(0);
4400 SDValue LHS = N->getOperand(0);
4401 SelectionDAG &DAG = DCI.DAG;
4402 SDLoc SL(N);
4403
4404 if (VT.getScalarType() != MVT::i64)
4405 return SDValue();
4406
4407 // For C >= 32
4408 // i64 (sra x, C) -> (build_pair (sra hi_32(x), C - 32), sra hi_32(x), 31))
4409
4410 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4411 // common case, splitting this into a move and a 32-bit shift is faster and
4412 // the same code size.
4413 KnownBits Known = DAG.computeKnownBits(RHS);
4414
4415 EVT ElementType = VT.getScalarType();
4416 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4417 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4418
4419 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4420 return SDValue();
4421
4422 SDValue ShiftFullAmt =
4423 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4424 SDValue ShiftAmt;
4425 if (CRHS) {
4426 unsigned RHSVal = CRHS->getZExtValue();
4427 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4428 TargetType);
4429 } else if (Known.getMinValue().getZExtValue() ==
4430 (ElementType.getSizeInBits() - 1)) {
4431 ShiftAmt = ShiftFullAmt;
4432 } else {
4433 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4434 const SDValue ShiftMask =
4435 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4436 // This AND instruction will clamp out of bounds shift values.
4437 // It will also be removed during later instruction selection.
4438 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4439 }
4440
4441 EVT ConcatType;
4442 SDValue Hi;
4443 SDLoc LHSSL(LHS);
4444 // Bitcast LHS into ConcatType so hi-half of source can be extracted into Hi
4445 if (VT.isVector()) {
4446 unsigned NElts = TargetType.getVectorNumElements();
4447 ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4448 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4449 SmallVector<SDValue, 8> HiOps(NElts);
4450 SmallVector<SDValue, 16> HiAndLoOps;
4451
4452 DAG.ExtractVectorElements(SplitLHS, HiAndLoOps, 0, NElts * 2);
4453 for (unsigned I = 0; I != NElts; ++I) {
4454 HiOps[I] = HiAndLoOps[2 * I + 1];
4455 }
4456 Hi = DAG.getNode(ISD::BUILD_VECTOR, LHSSL, TargetType, HiOps);
4457 } else {
4458 const SDValue One = DAG.getConstant(1, LHSSL, TargetScalarType);
4459 ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4460 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4461 Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, LHSSL, TargetType, SplitLHS, One);
4462 }
4463
4464 KnownBits KnownLHS = DAG.computeKnownBits(LHS);
4465 SDValue NewShift, HiShift;
4466 if (KnownLHS.isNegative()) {
4467 HiShift = DAG.getAllOnesConstant(SL, TargetType);
4468 NewShift =
4469 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4470 } else if (CRHS &&
4471 CRHS->getZExtValue() == (ElementType.getSizeInBits() - 1)) {
4472 NewShift = HiShift =
4473 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4474 } else {
4475 Hi = DAG.getFreeze(Hi);
4476 HiShift = DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftFullAmt);
4477 NewShift =
4478 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4479 }
4480
4481 SDValue Vec;
4482 if (VT.isVector()) {
4483 unsigned NElts = TargetType.getVectorNumElements();
4486 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2);
4487
4488 DAG.ExtractVectorElements(HiShift, HiOps, 0, NElts);
4489 DAG.ExtractVectorElements(NewShift, LoOps, 0, NElts);
4490 for (unsigned I = 0; I != NElts; ++I) {
4491 HiAndLoOps[2 * I + 1] = HiOps[I];
4492 HiAndLoOps[2 * I] = LoOps[I];
4493 }
4494 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4495 } else {
4496 Vec = DAG.getBuildVector(ConcatType, SL, {NewShift, HiShift});
4497 }
4498 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4499}
4500
4502 DAGCombinerInfo &DCI) const {
4503 SDValue RHS = N->getOperand(1);
4505 EVT VT = N->getValueType(0);
4506 SDValue LHS = N->getOperand(0);
4507 SelectionDAG &DAG = DCI.DAG;
4508 SDLoc SL(N);
4509 unsigned RHSVal;
4510
4511 if (CRHS) {
4512 RHSVal = CRHS->getZExtValue();
4513
4514 // fold (srl (and x, c1 << c2), c2) -> (and (srl(x, c2), c1)
4515 // this improves the ability to match BFE patterns in isel.
4516 if (LHS.getOpcode() == ISD::AND) {
4517 if (auto *Mask = dyn_cast<ConstantSDNode>(LHS.getOperand(1))) {
4518 unsigned MaskIdx, MaskLen;
4519 if (Mask->getAPIntValue().isShiftedMask(MaskIdx, MaskLen) &&
4520 MaskIdx == RHSVal) {
4521 return DAG.getNode(ISD::AND, SL, VT,
4522 DAG.getNode(ISD::SRL, SL, VT, LHS.getOperand(0),
4523 N->getOperand(1)),
4524 DAG.getNode(ISD::SRL, SL, VT, LHS.getOperand(1),
4525 N->getOperand(1)));
4526 }
4527 }
4528 }
4529 }
4530
4531 if (VT.getScalarType() != MVT::i64)
4532 return SDValue();
4533
4534 // for C >= 32
4535 // i64 (srl x, C) -> (build_pair (srl hi_32(x), C - 32), 0)
4536
4537 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4538 // common case, splitting this into a move and a 32-bit shift is faster and
4539 // the same code size.
4540 KnownBits Known = DAG.computeKnownBits(RHS);
4541
4542 EVT ElementType = VT.getScalarType();
4543 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4544 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4545
4546 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4547 return SDValue();
4548
4549 SDValue ShiftAmt;
4550 if (CRHS) {
4551 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4552 TargetType);
4553 } else {
4554 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4555 const SDValue ShiftMask =
4556 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4557 // This AND instruction will clamp out of bounds shift values.
4558 // It will also be removed during later instruction selection.
4559 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4560 }
4561
4562 const SDValue Zero = DAG.getConstant(0, SL, TargetScalarType);
4563 EVT ConcatType;
4564 SDValue Hi;
4565 SDLoc LHSSL(LHS);
4566 // Bitcast LHS into ConcatType so hi-half of source can be extracted into Hi
4567 if (VT.isVector()) {
4568 unsigned NElts = TargetType.getVectorNumElements();
4569 ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4570 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4571 SmallVector<SDValue, 8> HiOps(NElts);
4572 SmallVector<SDValue, 16> HiAndLoOps;
4573
4574 DAG.ExtractVectorElements(SplitLHS, HiAndLoOps, /*Start=*/0, NElts * 2);
4575 for (unsigned I = 0; I != NElts; ++I)
4576 HiOps[I] = HiAndLoOps[2 * I + 1];
4577 Hi = DAG.getNode(ISD::BUILD_VECTOR, LHSSL, TargetType, HiOps);
4578 } else {
4579 const SDValue One = DAG.getConstant(1, LHSSL, TargetScalarType);
4580 ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4581 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4582 Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, LHSSL, TargetType, SplitLHS, One);
4583 }
4584
4585 SDValue NewShift =
4586 DAG.getNode(ISD::SRL, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4587
4588 SDValue Vec;
4589 if (VT.isVector()) {
4590 unsigned NElts = TargetType.getVectorNumElements();
4592 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2, Zero);
4593
4594 DAG.ExtractVectorElements(NewShift, LoOps, 0, NElts);
4595 for (unsigned I = 0; I != NElts; ++I)
4596 HiAndLoOps[2 * I] = LoOps[I];
4597 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4598 } else {
4599 Vec = DAG.getBuildVector(ConcatType, SL, {NewShift, Zero});
4600 }
4601 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4602}
4603
4605 SDNode *N, DAGCombinerInfo &DCI) const {
4606 SDLoc SL(N);
4607 SelectionDAG &DAG = DCI.DAG;
4608 EVT VT = N->getValueType(0);
4609 SDValue Src = N->getOperand(0);
4610
4611 // vt1 (truncate (bitcast (build_vector vt0:x, ...))) -> vt1 (bitcast vt0:x)
4612 if (Src.getOpcode() == ISD::BITCAST && !VT.isVector()) {
4613 SDValue Vec = Src.getOperand(0);
4614 if (Vec.getOpcode() == ISD::BUILD_VECTOR) {
4615 SDValue Elt0 = Vec.getOperand(0);
4616 EVT EltVT = Elt0.getValueType();
4617 if (VT.getFixedSizeInBits() <= EltVT.getFixedSizeInBits()) {
4618 if (EltVT.isFloatingPoint()) {
4619 Elt0 = DAG.getNode(ISD::BITCAST, SL,
4620 EltVT.changeTypeToInteger(), Elt0);
4621 }
4622
4623 return DAG.getNode(ISD::TRUNCATE, SL, VT, Elt0);
4624 }
4625 }
4626 }
4627
4628 // Equivalent of above for accessing the high element of a vector as an
4629 // integer operation.
4630 // trunc (srl (bitcast (build_vector x, y))), 16 -> trunc (bitcast y)
4631 if (Src.getOpcode() == ISD::SRL && !VT.isVector()) {
4632 if (auto *K = isConstOrConstSplat(Src.getOperand(1))) {
4633 SDValue BV = stripBitcast(Src.getOperand(0));
4634 if (BV.getOpcode() == ISD::BUILD_VECTOR) {
4635 EVT SrcEltVT = BV.getOperand(0).getValueType();
4636 unsigned SrcEltSize = SrcEltVT.getSizeInBits();
4637 unsigned BitIndex = K->getZExtValue();
4638 unsigned PartIndex = BitIndex / SrcEltSize;
4639
4640 if (PartIndex * SrcEltSize == BitIndex &&
4641 PartIndex < BV.getNumOperands()) {
4642 if (SrcEltVT.getSizeInBits() == VT.getSizeInBits()) {
4643 SDValue SrcElt =
4644 DAG.getNode(ISD::BITCAST, SL, SrcEltVT.changeTypeToInteger(),
4645 BV.getOperand(PartIndex));
4646 return DAG.getNode(ISD::TRUNCATE, SL, VT, SrcElt);
4647 }
4648 }
4649 }
4650 }
4651 }
4652
4653 // Partially shrink 64-bit shifts to 32-bit if reduced to 16-bit.
4654 //
4655 // i16 (trunc (srl i64:x, K)), K <= 16 ->
4656 // i16 (trunc (srl (i32 (trunc x), K)))
4657 if (VT.getScalarSizeInBits() < 32) {
4658 EVT SrcVT = Src.getValueType();
4659 if (SrcVT.getScalarSizeInBits() > 32 &&
4660 (Src.getOpcode() == ISD::SRL ||
4661 Src.getOpcode() == ISD::SRA ||
4662 Src.getOpcode() == ISD::SHL)) {
4663 SDValue Amt = Src.getOperand(1);
4664 KnownBits Known = DAG.computeKnownBits(Amt);
4665
4666 // - For left shifts, do the transform as long as the shift
4667 // amount is still legal for i32, so when ShiftAmt < 32 (<= 31)
4668 // - For right shift, do it if ShiftAmt <= (32 - Size) to avoid
4669 // losing information stored in the high bits when truncating.
4670 const unsigned MaxCstSize =
4671 (Src.getOpcode() == ISD::SHL) ? 31 : (32 - VT.getScalarSizeInBits());
4672 if (Known.getMaxValue().ule(MaxCstSize)) {
4673 EVT MidVT = VT.isVector() ?
4674 EVT::getVectorVT(*DAG.getContext(), MVT::i32,
4675 VT.getVectorNumElements()) : MVT::i32;
4676
4677 EVT NewShiftVT = getShiftAmountTy(MidVT, DAG.getDataLayout());
4678 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, MidVT,
4679 Src.getOperand(0));
4680 DCI.AddToWorklist(Trunc.getNode());
4681
4682 if (Amt.getValueType() != NewShiftVT) {
4683 Amt = DAG.getZExtOrTrunc(Amt, SL, NewShiftVT);
4684 DCI.AddToWorklist(Amt.getNode());
4685 }
4686
4687 SDValue ShrunkShift = DAG.getNode(Src.getOpcode(), SL, MidVT,
4688 Trunc, Amt);
4689 return DAG.getNode(ISD::TRUNCATE, SL, VT, ShrunkShift);
4690 }
4691 }
4692 }
4693
4694 return SDValue();
4695}
4696
4697// We need to specifically handle i64 mul here to avoid unnecessary conversion
4698// instructions. If we only match on the legalized i64 mul expansion,
4699// SimplifyDemandedBits will be unable to remove them because there will be
4700// multiple uses due to the separate mul + mulh[su].
4701static SDValue getMul24(SelectionDAG &DAG, const SDLoc &SL,
4702 SDValue N0, SDValue N1, unsigned Size, bool Signed) {
4703 if (Size <= 32) {
4704 unsigned MulOpc = Signed ? AMDGPUISD::MUL_I24 : AMDGPUISD::MUL_U24;
4705 return DAG.getNode(MulOpc, SL, MVT::i32, N0, N1);
4706 }
4707
4708 unsigned MulLoOpc = Signed ? AMDGPUISD::MUL_I24 : AMDGPUISD::MUL_U24;
4709 unsigned MulHiOpc = Signed ? AMDGPUISD::MULHI_I24 : AMDGPUISD::MULHI_U24;
4710
4711 SDValue MulLo = DAG.getNode(MulLoOpc, SL, MVT::i32, N0, N1);
4712 SDValue MulHi = DAG.getNode(MulHiOpc, SL, MVT::i32, N0, N1);
4713
4714 return DAG.getNode(ISD::BUILD_PAIR, SL, MVT::i64, MulLo, MulHi);
4715}
4716
4717/// If \p V is an add of a constant 1, returns the other operand. Otherwise
4718/// return SDValue().
4719static SDValue getAddOneOp(const SDNode *V) {
4720 if (V->getOpcode() != ISD::ADD)
4721 return SDValue();
4722
4723 return isOneConstant(V->getOperand(1)) ? V->getOperand(0) : SDValue();
4724}
4725
4727 DAGCombinerInfo &DCI) const {
4728 assert(N->getOpcode() == ISD::MUL);
4729 EVT VT = N->getValueType(0);
4730
4731 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4732 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4733 // unnecessarily). isDivergent() is used as an approximation of whether the
4734 // value is in an SGPR.
4735 if (!N->isDivergent())
4736 return SDValue();
4737
4738 unsigned Size = VT.getSizeInBits();
4739 if (VT.isVector() || Size > 64)
4740 return SDValue();
4741
4742 SelectionDAG &DAG = DCI.DAG;
4743 SDLoc DL(N);
4744
4745 SDValue N0 = N->getOperand(0);
4746 SDValue N1 = N->getOperand(1);
4747
4748 // Undo InstCombine canonicalize X * (Y + 1) -> X * Y + X to enable mad
4749 // matching.
4750
4751 // mul x, (add y, 1) -> add (mul x, y), x
4752 auto IsFoldableAdd = [](SDValue V) -> SDValue {
4753 SDValue AddOp = getAddOneOp(V.getNode());
4754 if (!AddOp)
4755 return SDValue();
4756
4757 if (V.hasOneUse() || all_of(V->users(), [](const SDNode *U) -> bool {
4758 return U->getOpcode() == ISD::MUL;
4759 }))
4760 return AddOp;
4761
4762 return SDValue();
4763 };
4764
4765 // FIXME: The selection pattern is not properly checking for commuted
4766 // operands, so we have to place the mul in the LHS
4767 if (SDValue MulOper = IsFoldableAdd(N0)) {
4768 SDValue MulVal = DAG.getNode(N->getOpcode(), DL, VT, N1, MulOper);
4769 return DAG.getNode(ISD::ADD, DL, VT, MulVal, N1);
4770 }
4771
4772 if (SDValue MulOper = IsFoldableAdd(N1)) {
4773 SDValue MulVal = DAG.getNode(N->getOpcode(), DL, VT, N0, MulOper);
4774 return DAG.getNode(ISD::ADD, DL, VT, MulVal, N0);
4775 }
4776
4777 // There are i16 integer mul/mad.
4778 if (isTypeLegal(MVT::i16) && VT.getScalarType().bitsLE(MVT::i16))
4779 return SDValue();
4780
4781 // SimplifyDemandedBits has the annoying habit of turning useful zero_extends
4782 // in the source into any_extends if the result of the mul is truncated. Since
4783 // we can assume the high bits are whatever we want, use the underlying value
4784 // to avoid the unknown high bits from interfering.
4785 if (N0.getOpcode() == ISD::ANY_EXTEND)
4786 N0 = N0.getOperand(0);
4787
4788 if (N1.getOpcode() == ISD::ANY_EXTEND)
4789 N1 = N1.getOperand(0);
4790
4791 SDValue Mul;
4792
4793 if (Subtarget->hasMulU24() && isU24(N0, DAG) && isU24(N1, DAG)) {
4794 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4795 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4796 Mul = getMul24(DAG, DL, N0, N1, Size, false);
4797 } else if (Subtarget->hasMulI24() && isI24(N0, DAG) && isI24(N1, DAG)) {
4798 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4799 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4800 Mul = getMul24(DAG, DL, N0, N1, Size, true);
4801 } else {
4802 return SDValue();
4803 }
4804
4805 // We need to use sext even for MUL_U24, because MUL_U24 is used
4806 // for signed multiply of 8 and 16-bit types.
4807 return DAG.getSExtOrTrunc(Mul, DL, VT);
4808}
4809
4810SDValue
4812 DAGCombinerInfo &DCI) const {
4813 if (N->getValueType(0) != MVT::i32)
4814 return SDValue();
4815
4816 SelectionDAG &DAG = DCI.DAG;
4817 SDLoc DL(N);
4818
4819 bool Signed = N->getOpcode() == ISD::SMUL_LOHI;
4820 SDValue N0 = N->getOperand(0);
4821 SDValue N1 = N->getOperand(1);
4822
4823 // SimplifyDemandedBits has the annoying habit of turning useful zero_extends
4824 // in the source into any_extends if the result of the mul is truncated. Since
4825 // we can assume the high bits are whatever we want, use the underlying value
4826 // to avoid the unknown high bits from interfering.
4827 if (N0.getOpcode() == ISD::ANY_EXTEND)
4828 N0 = N0.getOperand(0);
4829 if (N1.getOpcode() == ISD::ANY_EXTEND)
4830 N1 = N1.getOperand(0);
4831
4832 // Try to use two fast 24-bit multiplies (one for each half of the result)
4833 // instead of one slow extending multiply.
4834 unsigned LoOpcode = 0;
4835 unsigned HiOpcode = 0;
4836 if (Signed) {
4837 if (Subtarget->hasMulI24() && isI24(N0, DAG) && isI24(N1, DAG)) {
4838 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4839 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4840 LoOpcode = AMDGPUISD::MUL_I24;
4841 HiOpcode = AMDGPUISD::MULHI_I24;
4842 }
4843 } else {
4844 if (Subtarget->hasMulU24() && isU24(N0, DAG) && isU24(N1, DAG)) {
4845 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4846 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4847 LoOpcode = AMDGPUISD::MUL_U24;
4848 HiOpcode = AMDGPUISD::MULHI_U24;
4849 }
4850 }
4851 if (!LoOpcode)
4852 return SDValue();
4853
4854 SDValue Lo = DAG.getNode(LoOpcode, DL, MVT::i32, N0, N1);
4855 SDValue Hi = DAG.getNode(HiOpcode, DL, MVT::i32, N0, N1);
4856 DCI.CombineTo(N, Lo, Hi);
4857 return SDValue(N, 0);
4858}
4859
4861 DAGCombinerInfo &DCI) const {
4862 EVT VT = N->getValueType(0);
4863
4864 if (!Subtarget->hasMulI24() || VT.isVector())
4865 return SDValue();
4866
4867 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4868 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4869 // unnecessarily). isDivergent() is used as an approximation of whether the
4870 // value is in an SGPR.
4871 // This doesn't apply if no s_mul_hi is available (since we'll end up with a
4872 // valu op anyway)
4873 if (Subtarget->hasSMulHi() && !N->isDivergent())
4874 return SDValue();
4875
4876 SelectionDAG &DAG = DCI.DAG;
4877 SDLoc DL(N);
4878
4879 SDValue N0 = N->getOperand(0);
4880 SDValue N1 = N->getOperand(1);
4881
4882 if (!isI24(N0, DAG) || !isI24(N1, DAG))
4883 return SDValue();
4884
4885 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4886 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4887
4888 SDValue Mulhi = DAG.getNode(AMDGPUISD::MULHI_I24, DL, MVT::i32, N0, N1);
4889 DCI.AddToWorklist(Mulhi.getNode());
4890 return DAG.getSExtOrTrunc(Mulhi, DL, VT);
4891}
4892
4894 DAGCombinerInfo &DCI) const {
4895 EVT VT = N->getValueType(0);
4896
4897 if (VT.isVector() || VT.getSizeInBits() > 32 || !Subtarget->hasMulU24())
4898 return SDValue();
4899
4900 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4901 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4902 // unnecessarily). isDivergent() is used as an approximation of whether the
4903 // value is in an SGPR.
4904 // This doesn't apply if no s_mul_hi is available (since we'll end up with a
4905 // valu op anyway)
4906 if (!N->isDivergent() && Subtarget->hasSMulHi())
4907 return SDValue();
4908
4909 SelectionDAG &DAG = DCI.DAG;
4910 SDLoc DL(N);
4911
4912 SDValue N0 = N->getOperand(0);
4913 SDValue N1 = N->getOperand(1);
4914
4915 if (!isU24(N0, DAG) || !isU24(N1, DAG))
4916 return SDValue();
4917
4918 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4919 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4920
4921 SDValue Mulhi = DAG.getNode(AMDGPUISD::MULHI_U24, DL, MVT::i32, N0, N1);
4922 DCI.AddToWorklist(Mulhi.getNode());
4923 return DAG.getZExtOrTrunc(Mulhi, DL, VT);
4924}
4925
4926SDValue AMDGPUTargetLowering::getFFBX_U32(SelectionDAG &DAG,
4927 SDValue Op,
4928 const SDLoc &DL,
4929 unsigned Opc) const {
4930 EVT VT = Op.getValueType();
4931 if (VT.bitsGT(MVT::i32))
4932 return SDValue();
4933
4934 if (VT != MVT::i32)
4935 Op = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, Op);
4936
4937 SDValue FFBX = DAG.getNode(Opc, DL, MVT::i32, Op);
4938 if (VT != MVT::i32)
4939 FFBX = DAG.getNode(ISD::TRUNCATE, DL, VT, FFBX);
4940
4941 return FFBX;
4942}
4943
4944// The native instructions return -1 on 0 input. Optimize out a select that
4945// produces -1 on 0.
4946//
4947// TODO: If zero is not undef, we could also do this if the output is compared
4948// against the bitwidth.
4949//
4950// TODO: Should probably combine against FFBH_U32 instead of ctlz directly.
4952 SDValue LHS, SDValue RHS,
4953 DAGCombinerInfo &DCI) const {
4954 if (!isNullConstant(Cond.getOperand(1)))
4955 return SDValue();
4956
4957 SelectionDAG &DAG = DCI.DAG;
4958 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(Cond.getOperand(2))->get();
4959 SDValue CmpLHS = Cond.getOperand(0);
4960
4961 // select (setcc x, 0, eq), -1, (ctlz_zero_poison x) -> ffbh_u32 x
4962 // select (setcc x, 0, eq), -1, (cttz_zero_poison x) -> ffbl_u32 x
4963 if (CCOpcode == ISD::SETEQ &&
4964 (isCtlzOpc(RHS.getOpcode()) || isCttzOpc(RHS.getOpcode())) &&
4965 RHS.getOperand(0) == CmpLHS && isAllOnesConstant(LHS)) {
4966 unsigned Opc =
4967 isCttzOpc(RHS.getOpcode()) ? AMDGPUISD::FFBL_B32 : AMDGPUISD::FFBH_U32;
4968 return getFFBX_U32(DAG, CmpLHS, SL, Opc);
4969 }
4970
4971 // select (setcc x, 0, ne), (ctlz_zero_poison x), -1 -> ffbh_u32 x
4972 // select (setcc x, 0, ne), (cttz_zero_poison x), -1 -> ffbl_u32 x
4973 if (CCOpcode == ISD::SETNE &&
4974 (isCtlzOpc(LHS.getOpcode()) || isCttzOpc(LHS.getOpcode())) &&
4975 LHS.getOperand(0) == CmpLHS && isAllOnesConstant(RHS)) {
4976 unsigned Opc =
4977 isCttzOpc(LHS.getOpcode()) ? AMDGPUISD::FFBL_B32 : AMDGPUISD::FFBH_U32;
4978
4979 return getFFBX_U32(DAG, CmpLHS, SL, Opc);
4980 }
4981
4982 return SDValue();
4983}
4984
4986 unsigned Op,
4987 const SDLoc &SL,
4988 SDValue Cond,
4989 SDValue N1,
4990 SDValue N2) {
4991 SelectionDAG &DAG = DCI.DAG;
4992 EVT VT = N1.getValueType();
4993
4994 SDValue NewSelect = DAG.getNode(ISD::SELECT, SL, VT, Cond,
4995 N1.getOperand(0), N2.getOperand(0));
4996 DCI.AddToWorklist(NewSelect.getNode());
4997 return DAG.getNode(Op, SL, VT, NewSelect);
4998}
4999
5000// Pull a free FP operation out of a select so it may fold into uses.
5001//
5002// select c, (fneg x), (fneg y) -> fneg (select c, x, y)
5003// select c, (fneg x), k -> fneg (select c, x, (fneg k))
5004//
5005// select c, (fabs x), (fabs y) -> fabs (select c, x, y)
5006// select c, (fabs x), +k -> fabs (select c, x, k)
5007SDValue
5009 SDValue N) const {
5010 SelectionDAG &DAG = DCI.DAG;
5011 SDValue Cond = N.getOperand(0);
5012 SDValue LHS = N.getOperand(1);
5013 SDValue RHS = N.getOperand(2);
5014
5015 EVT VT = N.getValueType();
5016 if ((LHS.getOpcode() == ISD::FABS && RHS.getOpcode() == ISD::FABS) ||
5017 (LHS.getOpcode() == ISD::FNEG && RHS.getOpcode() == ISD::FNEG)) {
5019 return SDValue();
5020
5021 return distributeOpThroughSelect(DCI, LHS.getOpcode(),
5022 SDLoc(N), Cond, LHS, RHS);
5023 }
5024
5025 bool Inv = false;
5026 if (RHS.getOpcode() == ISD::FABS || RHS.getOpcode() == ISD::FNEG) {
5027 std::swap(LHS, RHS);
5028 Inv = true;
5029 }
5030
5031 // TODO: Support vector constants.
5033 if ((LHS.getOpcode() == ISD::FNEG || LHS.getOpcode() == ISD::FABS) && CRHS &&
5034 !selectSupportsSourceMods(N.getNode())) {
5035 SDLoc SL(N);
5036 // If one side is an fneg/fabs and the other is a constant, we can push the
5037 // fneg/fabs down. If it's an fabs, the constant needs to be non-negative.
5038 SDValue NewLHS = LHS.getOperand(0);
5039 SDValue NewRHS = RHS;
5040
5041 // Careful: if the neg can be folded up, don't try to pull it back down.
5042 bool ShouldFoldNeg = true;
5043
5044 if (NewLHS.hasOneUse()) {
5045 unsigned Opc = NewLHS.getOpcode();
5046 if (LHS.getOpcode() == ISD::FNEG && fnegFoldsIntoOp(NewLHS.getNode()))
5047 ShouldFoldNeg = false;
5048 if (LHS.getOpcode() == ISD::FABS && Opc == ISD::FMUL)
5049 ShouldFoldNeg = false;
5050 }
5051
5052 if (ShouldFoldNeg) {
5053 if (LHS.getOpcode() == ISD::FABS && CRHS->isNegative())
5054 return SDValue();
5055
5056 // We're going to be forced to use a source modifier anyway, there's no
5057 // point to pulling the negate out unless we can get a size reduction by
5058 // negating the constant.
5059 //
5060 // TODO: Generalize to use getCheaperNegatedExpression which doesn't know
5061 // about cheaper constants.
5062 if (NewLHS.getOpcode() == ISD::FABS &&
5064 return SDValue();
5065
5067 return SDValue();
5068
5069 if (LHS.getOpcode() == ISD::FNEG)
5070 NewRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5071
5072 if (Inv)
5073 std::swap(NewLHS, NewRHS);
5074
5075 SDValue NewSelect = DAG.getNode(ISD::SELECT, SL, VT,
5076 Cond, NewLHS, NewRHS);
5077 DCI.AddToWorklist(NewSelect.getNode());
5078 return DAG.getNode(LHS.getOpcode(), SL, VT, NewSelect);
5079 }
5080 }
5081
5082 return SDValue();
5083}
5084
5086 DAGCombinerInfo &DCI) const {
5087 if (SDValue Folded = foldFreeOpFromSelect(DCI, SDValue(N, 0)))
5088 return Folded;
5089
5090 SDValue Cond = N->getOperand(0);
5091 if (Cond.getOpcode() != ISD::SETCC)
5092 return SDValue();
5093
5094 EVT VT = N->getValueType(0);
5095 SDValue LHS = Cond.getOperand(0);
5096 SDValue RHS = Cond.getOperand(1);
5097 SDValue CC = Cond.getOperand(2);
5098
5099 SDValue True = N->getOperand(1);
5100 SDValue False = N->getOperand(2);
5101
5102 if (Cond.hasOneUse()) { // TODO: Look for multiple select uses.
5103 SelectionDAG &DAG = DCI.DAG;
5104 if (DAG.isConstantValueOfAnyType(True) &&
5105 !DAG.isConstantValueOfAnyType(False)) {
5106 // Swap cmp + select pair to move constant to false input.
5107 // This will allow using VOPC cndmasks more often.
5108 // select (setcc x, y), k, x -> select (setccinv x, y), x, k
5109
5110 SDLoc SL(N);
5111 ISD::CondCode NewCC =
5112 getSetCCInverse(cast<CondCodeSDNode>(CC)->get(), LHS.getValueType());
5113
5114 SDValue NewCond = DAG.getSetCC(SL, Cond.getValueType(), LHS, RHS, NewCC);
5115 return DAG.getNode(ISD::SELECT, SL, VT, NewCond, False, True);
5116 }
5117
5118 if (VT == MVT::f32 && Subtarget->hasFminFmaxLegacy()) {
5119 SDValue MinMax = combineFMinMaxLegacy(SDLoc(N), VT, LHS, RHS, True, False,
5120 CC, N->getFlags(), DCI);
5121 // Revisit this node so we can catch min3/max3/med3 patterns.
5122 //DCI.AddToWorklist(MinMax.getNode());
5123 return MinMax;
5124 }
5125 }
5126
5127 // There's no reason to not do this if the condition has other uses.
5128 return performCtlz_CttzCombine(SDLoc(N), Cond, True, False, DCI);
5129}
5130
5131static bool isInv2Pi(const APFloat &APF) {
5132 static const APFloat KF16(APFloat::IEEEhalf(), APInt(16, 0x3118));
5133 static const APFloat KF32(APFloat::IEEEsingle(), APInt(32, 0x3e22f983));
5134 static const APFloat KF64(APFloat::IEEEdouble(), APInt(64, 0x3fc45f306dc9c882));
5135
5136 return APF.bitwiseIsEqual(KF16) ||
5137 APF.bitwiseIsEqual(KF32) ||
5138 APF.bitwiseIsEqual(KF64);
5139}
5140
5141// 0 and 1.0 / (0.5 * pi) do not have inline immmediates, so there is an
5142// additional cost to negate them.
5145 if (C->isZero())
5146 return C->isNegative() ? NegatibleCost::Cheaper : NegatibleCost::Expensive;
5147
5148 if (Subtarget->hasInv2PiInlineImm() && isInv2Pi(C->getValueAPF()))
5149 return C->isNegative() ? NegatibleCost::Cheaper : NegatibleCost::Expensive;
5150
5152}
5153
5159
5165
5166static unsigned inverseMinMax(unsigned Opc) {
5167 switch (Opc) {
5168 case ISD::FMAXNUM:
5169 return ISD::FMINNUM;
5170 case ISD::FMINNUM:
5171 return ISD::FMAXNUM;
5172 case ISD::FMAXNUM_IEEE:
5173 return ISD::FMINNUM_IEEE;
5174 case ISD::FMINNUM_IEEE:
5175 return ISD::FMAXNUM_IEEE;
5176 case ISD::FMAXIMUM:
5177 return ISD::FMINIMUM;
5178 case ISD::FMINIMUM:
5179 return ISD::FMAXIMUM;
5180 case ISD::FMAXIMUMNUM:
5181 return ISD::FMINIMUMNUM;
5182 case ISD::FMINIMUMNUM:
5183 return ISD::FMAXIMUMNUM;
5184 case AMDGPUISD::FMAX_LEGACY:
5185 return AMDGPUISD::FMIN_LEGACY;
5186 case AMDGPUISD::FMIN_LEGACY:
5187 return AMDGPUISD::FMAX_LEGACY;
5188 default:
5189 llvm_unreachable("invalid min/max opcode");
5190 }
5191}
5192
5193/// \return true if it's profitable to try to push an fneg into its source
5194/// instruction.
5196 // If the input has multiple uses and we can either fold the negate down, or
5197 // the other uses cannot, give up. This both prevents unprofitable
5198 // transformations and infinite loops: we won't repeatedly try to fold around
5199 // a negate that has no 'good' form.
5200 if (N0.hasOneUse()) {
5201 // This may be able to fold into the source, but at a code size cost. Don't
5202 // fold if the fold into the user is free.
5203 if (allUsesHaveSourceMods(N, 0))
5204 return false;
5205 } else {
5206 if (fnegFoldsIntoOp(N0.getNode()) &&
5208 return false;
5209 }
5210
5211 return true;
5212}
5213
5215 DAGCombinerInfo &DCI) const {
5216 SelectionDAG &DAG = DCI.DAG;
5217 SDValue N0 = N->getOperand(0);
5218 EVT VT = N->getValueType(0);
5219
5220 unsigned Opc = N0.getOpcode();
5221
5222 if (!shouldFoldFNegIntoSrc(N, N0))
5223 return SDValue();
5224
5225 SDLoc SL(N);
5226 switch (Opc) {
5227 case ISD::FADD: {
5228 if (!N0->getFlags().hasNoSignedZeros() && !N->getFlags().hasNoSignedZeros())
5229 return SDValue();
5230
5231 // (fneg (fadd x, y)) -> (fadd (fneg x), (fneg y))
5232 SDValue LHS = N0.getOperand(0);
5233 SDValue RHS = N0.getOperand(1);
5234
5235 if (LHS.getOpcode() != ISD::FNEG)
5236 LHS = DAG.getNode(ISD::FNEG, SL, VT, LHS);
5237 else
5238 LHS = LHS.getOperand(0);
5239
5240 if (RHS.getOpcode() != ISD::FNEG)
5241 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5242 else
5243 RHS = RHS.getOperand(0);
5244
5245 SDValue Res = DAG.getNode(ISD::FADD, SL, VT, LHS, RHS, N0->getFlags());
5246 if (Res.getOpcode() != ISD::FADD)
5247 return SDValue(); // Op got folded away.
5248 if (!N0.hasOneUse())
5249 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5250 return Res;
5251 }
5252 case ISD::FMUL:
5253 case AMDGPUISD::FMUL_LEGACY: {
5254 // (fneg (fmul x, y)) -> (fmul x, (fneg y))
5255 // (fneg (fmul_legacy x, y)) -> (fmul_legacy x, (fneg y))
5256 SDValue LHS = N0.getOperand(0);
5257 SDValue RHS = N0.getOperand(1);
5258
5259 if (LHS.getOpcode() == ISD::FNEG)
5260 LHS = LHS.getOperand(0);
5261 else if (RHS.getOpcode() == ISD::FNEG)
5262 RHS = RHS.getOperand(0);
5263 else
5264 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5265
5266 SDValue Res = DAG.getNode(Opc, SL, VT, LHS, RHS, N0->getFlags());
5267 if (Res.getOpcode() != Opc)
5268 return SDValue(); // Op got folded away.
5269 if (!N0.hasOneUse())
5270 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5271 return Res;
5272 }
5273 case ISD::FMA:
5274 case ISD::FMAD: {
5275 // TODO: handle llvm.amdgcn.fma.legacy
5276 if (!N0->getFlags().hasNoSignedZeros() && !N->getFlags().hasNoSignedZeros())
5277 return SDValue();
5278
5279 // (fneg (fma x, y, z)) -> (fma x, (fneg y), (fneg z))
5280 SDValue LHS = N0.getOperand(0);
5281 SDValue MHS = N0.getOperand(1);
5282 SDValue RHS = N0.getOperand(2);
5283
5284 if (LHS.getOpcode() == ISD::FNEG)
5285 LHS = LHS.getOperand(0);
5286 else if (MHS.getOpcode() == ISD::FNEG)
5287 MHS = MHS.getOperand(0);
5288 else
5289 MHS = DAG.getNode(ISD::FNEG, SL, VT, MHS);
5290
5291 if (RHS.getOpcode() != ISD::FNEG)
5292 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5293 else
5294 RHS = RHS.getOperand(0);
5295
5296 SDValue Res = DAG.getNode(Opc, SL, VT, LHS, MHS, RHS);
5297 if (Res.getOpcode() != Opc)
5298 return SDValue(); // Op got folded away.
5299 if (!N0.hasOneUse())
5300 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5301 return Res;
5302 }
5303 case ISD::FMAXNUM:
5304 case ISD::FMINNUM:
5305 case ISD::FMAXNUM_IEEE:
5306 case ISD::FMINNUM_IEEE:
5307 case ISD::FMINIMUM:
5308 case ISD::FMAXIMUM:
5309 case ISD::FMINIMUMNUM:
5310 case ISD::FMAXIMUMNUM:
5311 case AMDGPUISD::FMAX_LEGACY:
5312 case AMDGPUISD::FMIN_LEGACY: {
5313 // fneg (fmaxnum x, y) -> fminnum (fneg x), (fneg y)
5314 // fneg (fminnum x, y) -> fmaxnum (fneg x), (fneg y)
5315 // fneg (fmax_legacy x, y) -> fmin_legacy (fneg x), (fneg y)
5316 // fneg (fmin_legacy x, y) -> fmax_legacy (fneg x), (fneg y)
5317
5318 SDValue LHS = N0.getOperand(0);
5319 SDValue RHS = N0.getOperand(1);
5320
5321 // 0 doesn't have a negated inline immediate.
5322 // TODO: This constant check should be generalized to other operations.
5324 return SDValue();
5325
5326 // Swapping min<->max flips which operand a signed zero tie selects.
5327 if ((Opc == AMDGPUISD::FMIN_LEGACY || Opc == AMDGPUISD::FMAX_LEGACY) &&
5328 !canIgnoreLegacyMinMaxTies(DAG, N0->getFlags(), LHS, RHS))
5329 return SDValue();
5330
5331 SDValue NegLHS = DAG.getNode(ISD::FNEG, SL, VT, LHS);
5332 SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5333 unsigned Opposite = inverseMinMax(Opc);
5334
5335 SDValue Res = DAG.getNode(Opposite, SL, VT, NegLHS, NegRHS, N0->getFlags());
5336 if (Res.getOpcode() != Opposite)
5337 return SDValue(); // Op got folded away.
5338 if (!N0.hasOneUse())
5339 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5340 return Res;
5341 }
5342 case AMDGPUISD::FMED3: {
5343 // med3 sorts a NaN input as smaller than everything regardless of its sign,
5344 // so negating all operands does not sign-flip the median when an input may
5345 // be NaN.
5346 if (!N0->getFlags().hasNoNaNs())
5347 return SDValue();
5348
5349 SDValue Ops[3];
5350 for (unsigned I = 0; I < 3; ++I)
5351 Ops[I] = DAG.getNode(ISD::FNEG, SL, VT, N0->getOperand(I), N0->getFlags());
5352
5353 SDValue Res = DAG.getNode(AMDGPUISD::FMED3, SL, VT, Ops, N0->getFlags());
5354 if (Res.getOpcode() != AMDGPUISD::FMED3)
5355 return SDValue(); // Op got folded away.
5356
5357 if (!N0.hasOneUse()) {
5358 SDValue Neg = DAG.getNode(ISD::FNEG, SL, VT, Res);
5359 DAG.ReplaceAllUsesWith(N0, Neg);
5360
5361 for (SDNode *U : Neg->users())
5362 DCI.AddToWorklist(U);
5363 }
5364
5365 return Res;
5366 }
5367 case ISD::FP_EXTEND:
5368 case ISD::FTRUNC:
5369 case ISD::FRINT:
5370 case ISD::FNEARBYINT: // XXX - Should fround be handled?
5371 case ISD::FROUNDEVEN:
5372 case ISD::FSIN:
5373 case ISD::FCANONICALIZE:
5374 case AMDGPUISD::RCP:
5375 case AMDGPUISD::RCP_LEGACY:
5376 case AMDGPUISD::RCP_IFLAG:
5377 case AMDGPUISD::SIN_HW: {
5378 SDValue CvtSrc = N0.getOperand(0);
5379 if (CvtSrc.getOpcode() == ISD::FNEG) {
5380 // (fneg (fp_extend (fneg x))) -> (fp_extend x)
5381 // (fneg (rcp (fneg x))) -> (rcp x)
5382 return DAG.getNode(Opc, SL, VT, CvtSrc.getOperand(0));
5383 }
5384
5385 if (!N0.hasOneUse())
5386 return SDValue();
5387
5388 // (fneg (fp_extend x)) -> (fp_extend (fneg x))
5389 // (fneg (rcp x)) -> (rcp (fneg x))
5390 SDValue Neg = DAG.getNode(ISD::FNEG, SL, CvtSrc.getValueType(), CvtSrc);
5391 return DAG.getNode(Opc, SL, VT, Neg, N0->getFlags());
5392 }
5393 case ISD::FP_ROUND: {
5394 SDValue CvtSrc = N0.getOperand(0);
5395
5396 if (CvtSrc.getOpcode() == ISD::FNEG) {
5397 // (fneg (fp_round (fneg x))) -> (fp_round x)
5398 return DAG.getNode(ISD::FP_ROUND, SL, VT,
5399 CvtSrc.getOperand(0), N0.getOperand(1));
5400 }
5401
5402 if (!N0.hasOneUse())
5403 return SDValue();
5404
5405 // (fneg (fp_round x)) -> (fp_round (fneg x))
5406 SDValue Neg = DAG.getNode(ISD::FNEG, SL, CvtSrc.getValueType(), CvtSrc);
5407 return DAG.getNode(ISD::FP_ROUND, SL, VT, Neg, N0.getOperand(1));
5408 }
5409 case ISD::FP16_TO_FP: {
5410 // v_cvt_f32_f16 supports source modifiers on pre-VI targets without legal
5411 // f16, but legalization of f16 fneg ends up pulling it out of the source.
5412 // Put the fneg back as a legal source operation that can be matched later.
5413 SDLoc SL(N);
5414
5415 SDValue Src = N0.getOperand(0);
5416 EVT SrcVT = Src.getValueType();
5417
5418 // fneg (fp16_to_fp x) -> fp16_to_fp (xor x, 0x8000)
5419 SDValue IntFNeg = DAG.getNode(ISD::XOR, SL, SrcVT, Src,
5420 DAG.getConstant(0x8000, SL, SrcVT));
5421 return DAG.getNode(ISD::FP16_TO_FP, SL, N->getValueType(0), IntFNeg);
5422 }
5423 case ISD::SELECT: {
5424 // fneg (select c, a, b) -> select c, (fneg a), (fneg b)
5425 // TODO: Invert conditions of foldFreeOpFromSelect
5426 return SDValue();
5427 }
5428 case ISD::BITCAST: {
5429 SDLoc SL(N);
5430 SDValue BCSrc = N0.getOperand(0);
5431 if (BCSrc.getOpcode() == ISD::BUILD_VECTOR) {
5432 SDValue HighBits = BCSrc.getOperand(BCSrc.getNumOperands() - 1);
5433 if (VT != MVT::f64 || HighBits.getValueType().getSizeInBits() != 32 ||
5434 !fnegFoldsIntoOp(HighBits.getNode()))
5435 return SDValue();
5436
5437 // f64 fneg only really needs to operate on the high half of of the
5438 // register, so try to force it to an f32 operation to help make use of
5439 // source modifiers.
5440 //
5441 //
5442 // fneg (f64 (bitcast (build_vector x, y))) ->
5443 // f64 (bitcast (build_vector (bitcast i32:x to f32),
5444 // (fneg (bitcast i32:y to f32)))
5445
5446 SDValue CastHi = DAG.getNode(ISD::BITCAST, SL, MVT::f32, HighBits);
5447 SDValue NegHi = DAG.getNode(ISD::FNEG, SL, MVT::f32, CastHi);
5448 SDValue CastBack =
5449 DAG.getNode(ISD::BITCAST, SL, HighBits.getValueType(), NegHi);
5450
5452 Ops.back() = CastBack;
5453 DCI.AddToWorklist(NegHi.getNode());
5454 SDValue Build =
5455 DAG.getNode(ISD::BUILD_VECTOR, SL, BCSrc.getValueType(), Ops);
5456 SDValue Result = DAG.getNode(ISD::BITCAST, SL, VT, Build);
5457
5458 if (!N0.hasOneUse())
5459 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Result));
5460 return Result;
5461 }
5462
5463 if (BCSrc.getOpcode() == ISD::SELECT && VT == MVT::f32 &&
5464 BCSrc.hasOneUse()) {
5465 // fneg (bitcast (f32 (select cond, i32:lhs, i32:rhs))) ->
5466 // select cond, (bitcast i32:lhs to f32), (bitcast i32:rhs to f32)
5467
5468 // TODO: Cast back result for multiple uses is beneficial in some cases.
5469
5470 SDValue LHS =
5471 DAG.getNode(ISD::BITCAST, SL, MVT::f32, BCSrc.getOperand(1));
5472 SDValue RHS =
5473 DAG.getNode(ISD::BITCAST, SL, MVT::f32, BCSrc.getOperand(2));
5474
5475 SDValue NegLHS = DAG.getNode(ISD::FNEG, SL, MVT::f32, LHS);
5476 SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, MVT::f32, RHS);
5477
5478 return DAG.getNode(ISD::SELECT, SL, MVT::f32, BCSrc.getOperand(0), NegLHS,
5479 NegRHS);
5480 }
5481
5482 return SDValue();
5483 }
5484 default:
5485 return SDValue();
5486 }
5487}
5488
5490 DAGCombinerInfo &DCI) const {
5491 SelectionDAG &DAG = DCI.DAG;
5492 SDValue N0 = N->getOperand(0);
5493
5494 if (!N0.hasOneUse())
5495 return SDValue();
5496
5497 switch (N0.getOpcode()) {
5498 case ISD::FP16_TO_FP: {
5499 assert(!isTypeLegal(MVT::f16) && "should only see if f16 is illegal");
5500 SDLoc SL(N);
5501 SDValue Src = N0.getOperand(0);
5502 EVT SrcVT = Src.getValueType();
5503
5504 // fabs (fp16_to_fp x) -> fp16_to_fp (and x, 0x7fff)
5505 SDValue IntFAbs = DAG.getNode(ISD::AND, SL, SrcVT, Src,
5506 DAG.getConstant(0x7fff, SL, SrcVT));
5507 return DAG.getNode(ISD::FP16_TO_FP, SL, N->getValueType(0), IntFAbs);
5508 }
5509 case ISD::FP_ROUND: {
5510 SDLoc SL(N);
5511 SDValue CvtSrc = N0.getOperand(0);
5512
5513 // fabs (fp_round x) -> fp_round (fabs x)
5514 SDValue Abs = DAG.getNode(ISD::FABS, SL, CvtSrc.getValueType(), CvtSrc,
5515 N->getFlags());
5516 return DAG.getNode(ISD::FP_ROUND, SL, N->getValueType(0), Abs,
5517 N0.getOperand(1), N0->getFlags());
5518 }
5519 default:
5520 return SDValue();
5521 }
5522}
5523
5525 DAGCombinerInfo &DCI) const {
5526 const auto *CFP = dyn_cast<ConstantFPSDNode>(N->getOperand(0));
5527 if (!CFP)
5528 return SDValue();
5529
5530 std::optional<APFloat> Result = AMDGPU::evaluateRcp(CFP->getValueAPF());
5531 if (!Result)
5532 return SDValue();
5533
5534 return DCI.DAG.getConstantFP(*Result, SDLoc(N), N->getValueType(0));
5535}
5536
5538 if (!Subtarget->isGCN())
5539 return false;
5540
5543 auto &ST = DAG.getSubtarget<GCNSubtarget>();
5544 const auto *TII = ST.getInstrInfo();
5545
5546 if (!ST.hasVMovB64Inst() || (!SDConstant && !SDFPConstant))
5547 return false;
5548
5549 if (ST.has64BitLiterals())
5550 return true;
5551
5552 if (SDConstant) {
5553 const APInt &APVal = SDConstant->getAPIntValue();
5554 return isUInt<32>(APVal.getZExtValue()) || TII->isInlineConstant(APVal);
5555 }
5556
5557 APInt Val = SDFPConstant->getValueAPF().bitcastToAPInt();
5558 return isUInt<32>(Val.getZExtValue()) || TII->isInlineConstant(Val);
5559}
5560
5562 DAGCombinerInfo &DCI) const {
5563 SelectionDAG &DAG = DCI.DAG;
5564 SDLoc DL(N);
5565
5566 switch(N->getOpcode()) {
5567 default:
5568 break;
5569 case ISD::BITCAST: {
5570 EVT DestVT = N->getValueType(0);
5571
5572 // Push casts through vector builds. This helps avoid emitting a large
5573 // number of copies when materializing floating point vector constants.
5574 //
5575 // vNt1 bitcast (vNt0 (build_vector t0:x, t0:y)) =>
5576 // vnt1 = build_vector (t1 (bitcast t0:x)), (t1 (bitcast t0:y))
5577 if (DestVT.isVector()) {
5578 SDValue Src = N->getOperand(0);
5579 if (Src.getOpcode() == ISD::BUILD_VECTOR &&
5582 EVT SrcVT = Src.getValueType();
5583 unsigned NElts = DestVT.getVectorNumElements();
5584
5585 if (SrcVT.getVectorNumElements() == NElts) {
5586 EVT DestEltVT = DestVT.getVectorElementType();
5587
5588 SmallVector<SDValue, 8> CastedElts;
5589 SDLoc SL(N);
5590 for (unsigned I = 0, E = SrcVT.getVectorNumElements(); I != E; ++I) {
5591 SDValue Elt = Src.getOperand(I);
5592 CastedElts.push_back(DAG.getNode(ISD::BITCAST, DL, DestEltVT, Elt));
5593 }
5594
5595 return DAG.getBuildVector(DestVT, SL, CastedElts);
5596 }
5597 }
5598 }
5599
5600 if (DestVT.getSizeInBits() != 64 || !DestVT.isVector())
5601 break;
5602
5603 // Fold bitcasts of constants.
5604 //
5605 // v2i32 (bitcast i64:k) -> build_vector lo_32(k), hi_32(k)
5606 // TODO: Generalize and move to DAGCombiner
5607 SDValue Src = N->getOperand(0);
5609 SDLoc SL(N);
5610 if (isInt64ImmLegal(C, DAG))
5611 break;
5612 uint64_t CVal = C->getZExtValue();
5613 SDValue BV = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32,
5614 DAG.getConstant(Lo_32(CVal), SL, MVT::i32),
5615 DAG.getConstant(Hi_32(CVal), SL, MVT::i32));
5616 return DAG.getNode(ISD::BITCAST, SL, DestVT, BV);
5617 }
5618
5620 const APInt &Val = C->getValueAPF().bitcastToAPInt();
5621 SDLoc SL(N);
5622 if (isInt64ImmLegal(C, DAG))
5623 break;
5624 uint64_t CVal = Val.getZExtValue();
5625 SDValue Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32,
5626 DAG.getConstant(Lo_32(CVal), SL, MVT::i32),
5627 DAG.getConstant(Hi_32(CVal), SL, MVT::i32));
5628
5629 return DAG.getNode(ISD::BITCAST, SL, DestVT, Vec);
5630 }
5631
5632 break;
5633 }
5634 case ISD::SHL:
5635 case ISD::SRA:
5636 case ISD::SRL: {
5637 // Range metadata can be invalidated when loads are converted to legal types
5638 // (e.g. v2i64 -> v4i32).
5639 // Try to convert vector shl/sra/srl before type legalization so that range
5640 // metadata can be utilized.
5641 if (!(N->getValueType(0).isVector() &&
5644 break;
5645 if (N->getOpcode() == ISD::SHL)
5646 return performShlCombine(N, DCI);
5647 if (N->getOpcode() == ISD::SRA)
5648 return performSraCombine(N, DCI);
5649 return performSrlCombine(N, DCI);
5650 }
5651 case ISD::TRUNCATE:
5652 return performTruncateCombine(N, DCI);
5653 case ISD::MUL:
5654 return performMulCombine(N, DCI);
5655 case AMDGPUISD::MUL_U24:
5656 case AMDGPUISD::MUL_I24: {
5657 if (SDValue Simplified = simplifyMul24(N, DCI))
5658 return Simplified;
5659 break;
5660 }
5661 case AMDGPUISD::MULHI_I24:
5662 case AMDGPUISD::MULHI_U24:
5663 return simplifyMul24(N, DCI);
5664 case ISD::SMUL_LOHI:
5665 case ISD::UMUL_LOHI:
5666 return performMulLoHiCombine(N, DCI);
5667 case ISD::MULHS:
5668 return performMulhsCombine(N, DCI);
5669 case ISD::MULHU:
5670 return performMulhuCombine(N, DCI);
5671 case ISD::SELECT:
5672 return performSelectCombine(N, DCI);
5673 case ISD::FNEG:
5674 return performFNegCombine(N, DCI);
5675 case ISD::FABS:
5676 return performFAbsCombine(N, DCI);
5677 case AMDGPUISD::BFE_I32:
5678 case AMDGPUISD::BFE_U32: {
5679 assert(N->getValueType(0) == MVT::i32 &&
5680 "BFE_I32/BFE_U32 is a 32-bit operation");
5681 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(N->getOperand(2));
5682 if (!Width)
5683 break;
5684
5685 uint32_t WidthVal = Width->getZExtValue() & 0x1f;
5686 if (WidthVal == 0)
5687 return DAG.getConstant(0, DL, MVT::i32);
5688
5690 if (!Offset)
5691 break;
5692
5693 SDValue BitsFrom = N->getOperand(0);
5694 uint32_t OffsetVal = Offset->getZExtValue() & 0x1f;
5695
5696 bool Signed = N->getOpcode() == AMDGPUISD::BFE_I32;
5697
5698 if (OffsetVal == 0) {
5699 // This is already sign / zero extended, so try to fold away extra BFEs.
5700 EVT SmallVT = EVT::getIntegerVT(*DAG.getContext(), WidthVal);
5701 if (Signed) {
5702 if (DAG.ComputeNumSignBits(BitsFrom) >= 32 - WidthVal + 1)
5703 return BitsFrom;
5704
5705 // This is a sign_extend_inreg. Replace it to take advantage of existing
5706 // DAG Combines. If not eliminated, we will match back to BFE during
5707 // selection.
5708
5709 // TODO: The sext_inreg of extended types ends, although we can could
5710 // handle them in a single BFE.
5711 return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i32, BitsFrom,
5712 DAG.getValueType(SmallVT));
5713 }
5714
5715 if (DAG.MaskedValueIsZero(BitsFrom,
5716 APInt::getHighBitsSet(32, 32 - WidthVal)))
5717 return BitsFrom;
5718
5719 return DAG.getZeroExtendInReg(BitsFrom, DL, SmallVT);
5720 }
5721
5722 if (ConstantSDNode *CVal = dyn_cast<ConstantSDNode>(BitsFrom)) {
5723 if (Signed) {
5724 return constantFoldBFE<int32_t>(DAG,
5725 CVal->getSExtValue(),
5726 OffsetVal,
5727 WidthVal,
5728 DL);
5729 }
5730
5731 return constantFoldBFE<uint32_t>(DAG,
5732 CVal->getZExtValue(),
5733 OffsetVal,
5734 WidthVal,
5735 DL);
5736 }
5737
5738 if ((OffsetVal + WidthVal) >= 32 &&
5739 !(OffsetVal == 16 && WidthVal == 16 && Subtarget->hasSDWA())) {
5740 SDValue ShiftVal = DAG.getConstant(OffsetVal, DL, MVT::i32);
5741 return DAG.getNode(Signed ? ISD::SRA : ISD::SRL, DL, MVT::i32,
5742 BitsFrom, ShiftVal);
5743 }
5744
5745 if (BitsFrom.hasOneUse()) {
5746 APInt Demanded = APInt::getBitsSet(32,
5747 OffsetVal,
5748 OffsetVal + WidthVal);
5749
5752 !DCI.isBeforeLegalizeOps());
5753 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
5754 if (TLI.ShrinkDemandedConstant(BitsFrom, Demanded, TLO) ||
5755 TLI.SimplifyDemandedBits(BitsFrom, Demanded, Known, TLO)) {
5756 DCI.CommitTargetLoweringOpt(TLO);
5757 }
5758 }
5759
5760 break;
5761 }
5762 case ISD::LOAD:
5763 return performLoadCombine(N, DCI);
5764 case ISD::STORE:
5765 return performStoreCombine(N, DCI);
5766 case AMDGPUISD::RCP:
5767 case AMDGPUISD::RCP_IFLAG:
5768 return performRcpCombine(N, DCI);
5769 case ISD::AssertZext:
5770 case ISD::AssertSext:
5771 return performAssertSZExtCombine(N, DCI);
5773 return performIntrinsicWOChainCombine(N, DCI);
5774 case AMDGPUISD::FMAD_FTZ: {
5775 SDValue N0 = N->getOperand(0);
5776 SDValue N1 = N->getOperand(1);
5777 SDValue N2 = N->getOperand(2);
5778 EVT VT = N->getValueType(0);
5779
5780 // FMAD_FTZ is a FMAD + flush denormals to zero.
5781 // We flush the inputs, the intermediate step, and the output.
5785 if (N0CFP && N1CFP && N2CFP) {
5786 const auto FTZ = [](const APFloat &V) {
5787 if (V.isDenormal()) {
5788 APFloat Zero(V.getSemantics(), 0);
5789 return V.isNegative() ? -Zero : Zero;
5790 }
5791 return V;
5792 };
5793
5794 APFloat V0 = FTZ(N0CFP->getValueAPF());
5795 APFloat V1 = FTZ(N1CFP->getValueAPF());
5796 APFloat V2 = FTZ(N2CFP->getValueAPF());
5798 V0 = FTZ(V0);
5800 return DAG.getConstantFP(FTZ(V0), DL, VT);
5801 }
5802 break;
5803 }
5804 }
5805 return SDValue();
5806}
5807
5809 SDValue Op, const APInt &OriginalDemandedBits,
5810 const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO,
5811 unsigned Depth) const {
5812 switch (Op.getOpcode()) {
5814 switch (Op.getConstantOperandVal(0)) {
5815 case Intrinsic::amdgcn_readfirstlane:
5816 case Intrinsic::amdgcn_readlane:
5817 case Intrinsic::amdgcn_wwm: {
5818 if (SimplifyDemandedBits(Op.getOperand(1), OriginalDemandedBits,
5819 OriginalDemandedElts, Known, TLO, Depth + 1))
5820 return true;
5821 break;
5822 }
5823 case Intrinsic::amdgcn_set_inactive:
5824 case Intrinsic::amdgcn_set_inactive_chain_arg: {
5825 // The result is operand 1 in active lanes and operand 2 in inactive
5826 // lanes, so the known bits are the intersection of both operands.
5827 KnownBits KnownValue, KnownInactive;
5828 if (SimplifyDemandedBits(Op.getOperand(1), OriginalDemandedBits,
5829 OriginalDemandedElts, KnownValue, TLO,
5830 Depth + 1))
5831 return true;
5832 if (SimplifyDemandedBits(Op.getOperand(2), OriginalDemandedBits,
5833 OriginalDemandedElts, KnownInactive, TLO,
5834 Depth + 1))
5835 return true;
5836 Known = KnownValue.intersectWith(KnownInactive);
5837 break;
5838 }
5839 default:
5840 break;
5841 }
5842 break;
5843 }
5844 default:
5845 break;
5846 }
5847
5848 return false;
5849}
5850
5851//===----------------------------------------------------------------------===//
5852// Helper functions
5853//===----------------------------------------------------------------------===//
5854
5856 const TargetRegisterClass *RC,
5857 Register Reg, EVT VT,
5858 const SDLoc &SL,
5859 bool RawReg) const {
5861 MachineRegisterInfo &MRI = MF.getRegInfo();
5862 Register VReg;
5863
5864 if (!MRI.isLiveIn(Reg)) {
5865 VReg = MRI.createVirtualRegister(RC);
5866 MRI.addLiveIn(Reg, VReg);
5867 } else {
5868 VReg = MRI.getLiveInVirtReg(Reg);
5869 }
5870
5871 if (RawReg)
5872 return DAG.getRegister(VReg, VT);
5873
5874 return DAG.getCopyFromReg(DAG.getEntryNode(), SL, VReg, VT);
5875}
5876
5877// This may be called multiple times, and nothing prevents creating multiple
5878// objects at the same offset. See if we already defined this object.
5880 int64_t Offset) {
5881 for (int I = MFI.getObjectIndexBegin(); I < 0; ++I) {
5882 if (MFI.getObjectOffset(I) == Offset) {
5883 assert(MFI.getObjectSize(I) == Size);
5884 return I;
5885 }
5886 }
5887
5888 return MFI.CreateFixedObject(Size, Offset, true);
5889}
5890
5892 EVT VT,
5893 const SDLoc &SL,
5894 int64_t Offset) const {
5896 MachineFrameInfo &MFI = MF.getFrameInfo();
5897 int FI = getOrCreateFixedStackObject(MFI, VT.getStoreSize(), Offset);
5898
5899 auto SrcPtrInfo = MachinePointerInfo::getStack(MF, Offset);
5900 SDValue Ptr = DAG.getFrameIndex(FI, MVT::i32);
5901
5902 return DAG.getLoad(VT, SL, DAG.getEntryNode(), Ptr, SrcPtrInfo, Align(4),
5905}
5906
5908 const SDLoc &SL,
5909 SDValue Chain,
5910 SDValue ArgVal,
5911 int64_t Offset) const {
5915
5916 SDValue Ptr = DAG.getConstant(Offset, SL, MVT::i32);
5917 // Stores to the argument stack area are relative to the stack pointer.
5918 SDValue SP =
5919 DAG.getCopyFromReg(Chain, SL, Info->getStackPtrOffsetReg(), MVT::i32);
5920 Ptr = DAG.getNode(ISD::ADD, SL, MVT::i32, SP, Ptr);
5921 SDValue Store = DAG.getStore(Chain, SL, ArgVal, Ptr, DstInfo, Align(4),
5923 return Store;
5924}
5925
5927 const TargetRegisterClass *RC,
5928 EVT VT, const SDLoc &SL,
5929 const ArgDescriptor &Arg) const {
5930 assert(Arg && "Attempting to load missing argument");
5931
5932 SDValue V = Arg.isRegister() ?
5933 CreateLiveInRegister(DAG, RC, Arg.getRegister(), VT, SL) :
5934 loadStackInputValue(DAG, VT, SL, Arg.getStackOffset());
5935
5936 if (!Arg.isMasked())
5937 return V;
5938
5939 unsigned Mask = Arg.getMask();
5940 unsigned Shift = llvm::countr_zero<unsigned>(Mask);
5941 V = DAG.getNode(ISD::SRL, SL, VT, V,
5942 DAG.getShiftAmountConstant(Shift, VT, SL));
5943 return DAG.getNode(ISD::AND, SL, VT, V,
5944 DAG.getConstant(Mask >> Shift, SL, VT));
5945}
5946
5948 uint64_t ExplicitKernArgSize, const ImplicitParameter Param) const {
5949 unsigned ExplicitArgOffset = Subtarget->getExplicitKernelArgOffset();
5950 const Align Alignment = Subtarget->getAlignmentForImplicitArgPtr();
5951 uint64_t ArgOffset =
5952 alignTo(ExplicitKernArgSize, Alignment) + ExplicitArgOffset;
5953 switch (Param) {
5954 case FIRST_IMPLICIT:
5955 return ArgOffset;
5956 case PRIVATE_BASE:
5958 case SHARED_BASE:
5959 return ArgOffset + AMDGPU::ImplicitArg::SHARED_BASE_OFFSET;
5960 case QUEUE_PTR:
5961 return ArgOffset + AMDGPU::ImplicitArg::QUEUE_PTR_OFFSET;
5962 }
5963 llvm_unreachable("unexpected implicit parameter type");
5964}
5965
5972
5974 SelectionDAG &DAG, int Enabled,
5975 int &RefinementSteps,
5976 bool &UseOneConstNR,
5977 bool Reciprocal) const {
5978 EVT VT = Operand.getValueType();
5979
5980 if (VT == MVT::f32) {
5981 RefinementSteps = 0;
5982 return DAG.getNode(AMDGPUISD::RSQ, SDLoc(Operand), VT, Operand);
5983 }
5984
5985 // TODO: There is also f64 rsq instruction, but the documentation is less
5986 // clear on its precision.
5987
5988 return SDValue();
5989}
5990
5992 SelectionDAG &DAG, int Enabled,
5993 int &RefinementSteps) const {
5994 EVT VT = Operand.getValueType();
5995
5996 if (VT == MVT::f32) {
5997 // Reciprocal, < 1 ulp error.
5998 //
5999 // This reciprocal approximation converges to < 0.5 ulp error with one
6000 // newton rhapson performed with two fused multiple adds (FMAs).
6001
6002 RefinementSteps = 0;
6003 return DAG.getNode(AMDGPUISD::RCP, SDLoc(Operand), VT, Operand);
6004 }
6005
6006 // TODO: There is also f64 rcp instruction, but the documentation is less
6007 // clear on its precision.
6008
6009 return SDValue();
6010}
6011
6012static unsigned workitemIntrinsicDim(unsigned ID) {
6013 switch (ID) {
6014 case Intrinsic::amdgcn_workitem_id_x:
6015 return 0;
6016 case Intrinsic::amdgcn_workitem_id_y:
6017 return 1;
6018 case Intrinsic::amdgcn_workitem_id_z:
6019 return 2;
6020 default:
6021 llvm_unreachable("not a workitem intrinsic");
6022 }
6023}
6024
6026 const SDValue Op, KnownBits &Known,
6027 const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth) const {
6028
6029 Known.resetAll(); // Don't know anything.
6030
6031 unsigned Opc = Op.getOpcode();
6032
6033 switch (Opc) {
6034 default:
6035 break;
6036 case AMDGPUISD::CARRY:
6037 case AMDGPUISD::BORROW: {
6038 Known.Zero = APInt::getHighBitsSet(32, 31);
6039 break;
6040 }
6041
6042 case AMDGPUISD::BFE_I32:
6043 case AMDGPUISD::BFE_U32: {
6044 ConstantSDNode *CWidth = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6045 if (!CWidth)
6046 return;
6047
6048 uint32_t Width = CWidth->getZExtValue() & 0x1f;
6049
6050 if (Opc == AMDGPUISD::BFE_U32)
6051 Known.Zero = APInt::getHighBitsSet(32, 32 - Width);
6052
6053 break;
6054 }
6055 case AMDGPUISD::FP_TO_FP16: {
6056 unsigned BitWidth = Known.getBitWidth();
6057
6058 // High bits are zero.
6060 break;
6061 }
6062 case AMDGPUISD::MUL_U24:
6063 case AMDGPUISD::MUL_I24: {
6064 KnownBits LHSKnown = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6065 KnownBits RHSKnown = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6066 unsigned BitWidth = Op.getScalarValueSizeInBits();
6067
6068 // Sign/Zero extend from 24 bits.
6069 if (Opc == AMDGPUISD::MUL_I24) {
6070 LHSKnown = LHSKnown.trunc(24).sext(BitWidth);
6071 RHSKnown = RHSKnown.trunc(24).sext(BitWidth);
6072 } else {
6073 LHSKnown = LHSKnown.trunc(24).zext(BitWidth);
6074 RHSKnown = RHSKnown.trunc(24).zext(BitWidth);
6075 }
6076
6077 // TODO: SelfMultiply can be poison, but not undef.
6078 bool SelfMultiply = Op.getOperand(0) == Op.getOperand(1);
6079 if (SelfMultiply)
6080 SelfMultiply &= DAG.isGuaranteedNotToBeUndefOrPoison(
6081 Op.getOperand(0), DemandedElts, UndefPoisonKind::UndefOrPoison,
6082 Depth + 1);
6083
6084 Known = KnownBits::mul(LHSKnown, RHSKnown, SelfMultiply);
6085 break;
6086 }
6087 case AMDGPUISD::PERM: {
6088 ConstantSDNode *CMask = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6089 if (!CMask)
6090 return;
6091
6092 KnownBits LHSKnown = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6093 KnownBits RHSKnown = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6094 unsigned Sel = CMask->getZExtValue();
6095
6096 for (unsigned I = 0; I < 32; I += 8) {
6097 unsigned SelBits = Sel & 0xff;
6098 if (SelBits < 4) {
6099 SelBits *= 8;
6100 Known.One |= ((RHSKnown.One.getZExtValue() >> SelBits) & 0xff) << I;
6101 Known.Zero |= ((RHSKnown.Zero.getZExtValue() >> SelBits) & 0xff) << I;
6102 } else if (SelBits < 7) {
6103 SelBits = (SelBits & 3) * 8;
6104 Known.One |= ((LHSKnown.One.getZExtValue() >> SelBits) & 0xff) << I;
6105 Known.Zero |= ((LHSKnown.Zero.getZExtValue() >> SelBits) & 0xff) << I;
6106 } else if (SelBits == 0x0c) {
6107 Known.Zero |= 0xFFull << I;
6108 } else if (SelBits > 0x0c) {
6109 Known.One |= 0xFFull << I;
6110 }
6111 Sel >>= 8;
6112 }
6113 break;
6114 }
6115 case AMDGPUISD::BUFFER_LOAD_UBYTE: {
6116 Known.Zero.setHighBits(24);
6117 break;
6118 }
6119 case AMDGPUISD::BUFFER_LOAD_USHORT: {
6120 Known.Zero.setHighBits(16);
6121 break;
6122 }
6123 case AMDGPUISD::LDS: {
6124 auto *GA = cast<GlobalAddressSDNode>(Op.getOperand(0).getNode());
6125 Align Alignment = GA->getGlobal()->getPointerAlignment(DAG.getDataLayout());
6126
6127 Known.Zero.setHighBits(16);
6128 Known.Zero.setLowBits(Log2(Alignment));
6129 break;
6130 }
6131 case AMDGPUISD::SMIN3:
6132 case AMDGPUISD::SMAX3:
6133 case AMDGPUISD::SMED3:
6134 case AMDGPUISD::UMIN3:
6135 case AMDGPUISD::UMAX3:
6136 case AMDGPUISD::UMED3: {
6137 KnownBits Known2 = DAG.computeKnownBits(Op.getOperand(2), Depth + 1);
6138 if (Known2.isUnknown())
6139 break;
6140
6141 KnownBits Known1 = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6142 if (Known1.isUnknown())
6143 break;
6144
6145 KnownBits Known0 = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6146 if (Known0.isUnknown())
6147 break;
6148
6149 // TODO: Handle LeadZero/LeadOne from UMIN/UMAX handling.
6150 Known.Zero = Known0.Zero & Known1.Zero & Known2.Zero;
6151 Known.One = Known0.One & Known1.One & Known2.One;
6152 break;
6153 }
6155 unsigned IID = Op.getConstantOperandVal(0);
6156 switch (IID) {
6157 case Intrinsic::amdgcn_workitem_id_x:
6158 case Intrinsic::amdgcn_workitem_id_y:
6159 case Intrinsic::amdgcn_workitem_id_z: {
6160 unsigned MaxValue = Subtarget->getMaxWorkitemID(
6162 Known.Zero.setHighBits(llvm::countl_zero(MaxValue));
6163 break;
6164 }
6165 default:
6166 break;
6167 }
6168 }
6169 }
6170}
6171
6173 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
6174 unsigned Depth) const {
6175 switch (Op.getOpcode()) {
6176 case AMDGPUISD::BFE_I32: {
6177 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6178 if (!Width)
6179 return 1;
6180
6181 unsigned SignBits = 32 - (Width->getZExtValue() & 0x1f) + 1;
6182 if (!isNullConstant(Op.getOperand(1)))
6183 return SignBits;
6184
6185 // TODO: Could probably figure something out with non-0 offsets.
6186 unsigned Op0SignBits = DAG.ComputeNumSignBits(Op.getOperand(0), Depth + 1);
6187 return std::max(SignBits, Op0SignBits);
6188 }
6189
6190 case AMDGPUISD::BFE_U32: {
6191 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6192 return Width ? 32 - (Width->getZExtValue() & 0x1f) : 1;
6193 }
6194
6195 case AMDGPUISD::CARRY:
6196 case AMDGPUISD::BORROW:
6197 return 31;
6198 case AMDGPUISD::BUFFER_LOAD_BYTE:
6199 return 25;
6200 case AMDGPUISD::BUFFER_LOAD_SHORT:
6201 return 17;
6202 case AMDGPUISD::BUFFER_LOAD_UBYTE:
6203 return 24;
6204 case AMDGPUISD::BUFFER_LOAD_USHORT:
6205 return 16;
6206 case AMDGPUISD::FP_TO_FP16:
6207 return 16;
6208 case AMDGPUISD::SMIN3:
6209 case AMDGPUISD::SMAX3:
6210 case AMDGPUISD::SMED3:
6211 case AMDGPUISD::UMIN3:
6212 case AMDGPUISD::UMAX3:
6213 case AMDGPUISD::UMED3: {
6214 unsigned Tmp2 = DAG.ComputeNumSignBits(Op.getOperand(2), Depth + 1);
6215 if (Tmp2 == 1)
6216 return 1; // Early out.
6217
6218 unsigned Tmp1 = DAG.ComputeNumSignBits(Op.getOperand(1), Depth + 1);
6219 if (Tmp1 == 1)
6220 return 1; // Early out.
6221
6222 unsigned Tmp0 = DAG.ComputeNumSignBits(Op.getOperand(0), Depth + 1);
6223 if (Tmp0 == 1)
6224 return 1; // Early out.
6225
6226 return std::min({Tmp0, Tmp1, Tmp2});
6227 }
6228 default:
6229 return 1;
6230 }
6231}
6232
6234 GISelValueTracking &Analysis, Register R, const APInt &DemandedElts,
6235 const MachineRegisterInfo &MRI, unsigned Depth) const {
6236 const MachineInstr *MI = MRI.getVRegDef(R);
6237 if (!MI)
6238 return 1;
6239
6240 // TODO: Check range metadata on MMO.
6241 switch (MI->getOpcode()) {
6242 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE:
6243 return 25;
6244 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT:
6245 return 17;
6246 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
6247 return 24;
6248 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
6249 return 16;
6250 case AMDGPU::G_AMDGPU_SMED3:
6251 case AMDGPU::G_AMDGPU_UMED3: {
6252 auto [Dst, Src0, Src1, Src2] = MI->getFirst4Regs();
6253 unsigned Tmp2 = Analysis.computeNumSignBits(Src2, DemandedElts, Depth + 1);
6254 if (Tmp2 == 1)
6255 return 1;
6256 unsigned Tmp1 = Analysis.computeNumSignBits(Src1, DemandedElts, Depth + 1);
6257 if (Tmp1 == 1)
6258 return 1;
6259 unsigned Tmp0 = Analysis.computeNumSignBits(Src0, DemandedElts, Depth + 1);
6260 if (Tmp0 == 1)
6261 return 1;
6262 return std::min({Tmp0, Tmp1, Tmp2});
6263 }
6264 default:
6265 return 1;
6266 }
6267}
6268
6270 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
6271 UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const {
6272 unsigned Opcode = Op.getOpcode();
6273 switch (Opcode) {
6274 case AMDGPUISD::BFE_I32:
6275 case AMDGPUISD::BFE_U32:
6276 return false;
6277 }
6279 Op, DemandedElts, DAG, Kind, ConsiderFlags, Depth);
6280}
6281
6283 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN,
6284 unsigned Depth) const {
6285 unsigned Opcode = Op.getOpcode();
6286 switch (Opcode) {
6287 case AMDGPUISD::FMIN_LEGACY:
6288 case AMDGPUISD::FMAX_LEGACY:
6289 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6290 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6291 case AMDGPUISD::FMUL_LEGACY:
6292 case AMDGPUISD::CVT_PKRTZ_F16_F32: {
6293 if (SNaN)
6294 return true;
6295 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6296 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6297 }
6298 case AMDGPUISD::FMED3:
6299 case AMDGPUISD::FMIN3:
6300 case AMDGPUISD::FMAX3:
6301 case AMDGPUISD::FMINIMUM3:
6302 case AMDGPUISD::FMAXIMUM3:
6303 case AMDGPUISD::FMAD_FTZ: {
6304 if (SNaN)
6305 return true;
6306 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6307 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6308 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1);
6309 }
6310 case AMDGPUISD::CVT_F32_UBYTE0:
6311 case AMDGPUISD::CVT_F32_UBYTE1:
6312 case AMDGPUISD::CVT_F32_UBYTE2:
6313 case AMDGPUISD::CVT_F32_UBYTE3:
6314 return true;
6315
6316 case AMDGPUISD::RCP:
6317 case AMDGPUISD::RSQ:
6318 case AMDGPUISD::RCP_LEGACY:
6319 case AMDGPUISD::RSQ_CLAMP: {
6320 if (SNaN)
6321 return true;
6322
6323 // TODO: Need is known positive check.
6324 return false;
6325 }
6326 case ISD::FLDEXP:
6327 case AMDGPUISD::FRACT: {
6328 if (SNaN)
6329 return true;
6330 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1);
6331 }
6332 case AMDGPUISD::DIV_SCALE:
6333 case AMDGPUISD::DIV_FMAS:
6334 case AMDGPUISD::DIV_FIXUP:
6335 // TODO: Refine on operands.
6336 return SNaN;
6337 case AMDGPUISD::SIN_HW:
6338 case AMDGPUISD::COS_HW: {
6339 // TODO: Need check for infinity
6340 return SNaN;
6341 }
6343 unsigned IntrinsicID = Op.getConstantOperandVal(0);
6344 // TODO: Handle more intrinsics
6345 switch (IntrinsicID) {
6346 case Intrinsic::amdgcn_cubeid:
6347 case Intrinsic::amdgcn_cvt_off_f32_i4:
6348 return true;
6349
6350 case Intrinsic::amdgcn_frexp_mant: {
6351 if (SNaN)
6352 return true;
6353 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6354 }
6355 case Intrinsic::amdgcn_cvt_pkrtz: {
6356 if (SNaN)
6357 return true;
6358 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6359 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1);
6360 }
6361 case Intrinsic::amdgcn_rcp:
6362 case Intrinsic::amdgcn_rsq:
6363 case Intrinsic::amdgcn_rcp_legacy:
6364 case Intrinsic::amdgcn_rsq_legacy:
6365 case Intrinsic::amdgcn_rsq_clamp:
6366 case Intrinsic::amdgcn_tanh: {
6367 if (SNaN)
6368 return true;
6369
6370 // TODO: Need is known positive check.
6371 return false;
6372 }
6373 case Intrinsic::amdgcn_trig_preop:
6374 case Intrinsic::amdgcn_fdot2:
6375 // TODO: Refine on operand
6376 return SNaN;
6377 case Intrinsic::amdgcn_fma_legacy:
6378 if (SNaN)
6379 return true;
6380 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6381 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1) &&
6382 DAG.isKnownNeverNaN(Op.getOperand(3), SNaN, Depth + 1);
6383 default:
6384 return false;
6385 }
6386 }
6387 default:
6388 return false;
6389 }
6390}
6391
6393 Register N0, Register N1) const {
6394 return MRI.hasOneNonDBGUse(N0); // FIXME: handle regbanks
6395}
return SDValue()
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static LLVM_READONLY bool hasSourceMods(const MachineInstr &MI)
static bool isInv2Pi(const APFloat &APF)
static LLVM_READONLY bool opMustUseVOP3Encoding(const MachineInstr &MI, const MachineRegisterInfo &MRI)
returns true if the operation will definitely need to use a 64-bit encoding, and thus will use a VOP3...
static unsigned inverseMinMax(unsigned Opc)
unsigned Imm
static SDValue extractF64Exponent(SDValue Hi, const SDLoc &SL, SelectionDAG &DAG)
static unsigned workitemIntrinsicDim(unsigned ID)
static int getOrCreateFixedStackObject(MachineFrameInfo &MFI, unsigned Size, int64_t Offset)
static SDValue constantFoldBFE(SelectionDAG &DAG, IntTy Src0, uint32_t Offset, uint32_t Width, const SDLoc &DL)
static SDValue getMad(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue X, SDValue Y, SDValue C, SDNodeFlags Flags=SDNodeFlags())
static SDValue getAddOneOp(const SDNode *V)
If V is an add of a constant 1, returns the other operand.
static bool canIgnoreLegacyMinMaxTies(const SelectionDAG &DAG, SDNodeFlags Flags, SDValue LHS, SDValue RHS)
static LLVM_READONLY bool selectSupportsSourceMods(const SDNode *N)
Return true if v_cndmask_b32 will support fabs/fneg source modifiers for the type for ISD::SELECT.
static cl::opt< bool > AMDGPUBypassSlowDiv("amdgpu-bypass-slow-div", cl::desc("Skip 64-bit divide for dynamic 32-bit values"), cl::init(true))
static SDValue getMul24(SelectionDAG &DAG, const SDLoc &SL, SDValue N0, SDValue N1, unsigned Size, bool Signed)
static bool fnegFoldsIntoOp(const SDNode *N)
static bool isI24(SDValue Op, SelectionDAG &DAG)
static bool isCttzOpc(unsigned Opc)
static bool isU24(SDValue Op, SelectionDAG &DAG)
static bool valueIsKnownNeverF32Denorm(SDValue Src)
Return true if it's known that Src can never be an f32 denormal value.
static SDValue distributeOpThroughSelect(TargetLowering::DAGCombinerInfo &DCI, unsigned Op, const SDLoc &SL, SDValue Cond, SDValue N1, SDValue N2)
static SDValue peekFNeg(SDValue Val)
static SDValue simplifyMul24(SDNode *Node24, TargetLowering::DAGCombinerInfo &DCI)
static bool isCtlzOpc(unsigned Opc)
static LLVM_READNONE bool fnegFoldsIntoOpcode(unsigned Opc)
static bool hasVolatileUser(SDNode *Val)
Interface definition of the TargetLowering class that is common to all AMD GPUs.
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
block Block Frequency Analysis
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_READNONE
Definition Compiler.h:323
#define LLVM_READONLY
Definition Compiler.h:330
Provides analysis for querying information about KnownBits during GISel passes.
const HexagonInstrInfo * TII
static MaybeAlign getAlign(Value *Ptr)
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
#define T
#define P(N)
const SmallVectorImpl< MachineOperand > & Cond
#define CH(x, y, z)
Definition SHA256.cpp:34
Func MI getDebugLoc()))
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
BinaryOperator * Mul
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
unsigned allocateLDSGlobal(const DataLayout &DL, const GlobalVariable &GV)
void recordNumNamedBarriers(uint32_t GVAddr, unsigned BarCnt)
static std::optional< uint32_t > getLDSAbsoluteAddress(const GlobalValue &GV)
static unsigned numBitsSigned(SDValue Op, SelectionDAG &DAG)
unsigned ComputeNumSignBitsForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
This method can be implemented by targets that want to expose additional information about sign bits ...
SDValue performMulhuCombine(SDNode *N, DAGCombinerInfo &DCI) const
EVT getTypeForExtReturn(LLVMContext &Context, EVT VT, ISD::NodeType ExtendKind) const override
Return the type that should be used to zero or sign extend a zeroext/signext integer return value.
SDValue SplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Split a vector load into 2 loads of half the vector.
SDValue LowerCONCAT_VECTORS(SDValue Op, SelectionDAG &DAG) const
SDValue performLoadCombine(SDNode *N, DAGCombinerInfo &DCI) const
void analyzeFormalArgumentsCompute(CCState &State, const SmallVectorImpl< ISD::InputArg > &Ins) const
The SelectionDAGBuilder will automatically promote function arguments with illegal types.
SDValue LowerF64ToF16Safe(SDValue Src, const SDLoc &DL, SelectionDAG &DAG) const
SDValue LowerFROUND(SDValue Op, SelectionDAG &DAG) const
SDValue storeStackInputValue(SelectionDAG &DAG, const SDLoc &SL, SDValue Chain, SDValue ArgVal, int64_t Offset) const
bool storeOfVectorConstantIsCheap(bool IsZero, EVT MemVT, unsigned NumElem, unsigned AS) const override
Return true if it is expected to be cheaper to do a store of vector constant with the given size and ...
SDValue LowerEXTRACT_SUBVECTOR(SDValue Op, SelectionDAG &DAG) const
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
bool shouldCombineMemoryType(EVT VT) const
SDValue splitBinaryBitConstantOpImpl(DAGCombinerInfo &DCI, const SDLoc &SL, unsigned Opc, SDValue LHS, uint32_t ValLo, uint32_t ValHi) const
Split the 64-bit value LHS into two 32-bit components, and perform the binary operation Opc to it wit...
SDValue lowerUnhandledCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals, StringRef Reason) const
virtual SDValue LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI, SDValue Op, SelectionDAG &DAG) const
SDValue performAssertSZExtCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isTruncateFree(EVT Src, EVT Dest) const override
bool aggressivelyPreferBuildVectorSources(EVT VecVT) const override
SDValue LowerFCEIL(SDValue Op, SelectionDAG &DAG) const
TargetLowering::NegatibleCost getConstantNegateCost(const ConstantFPSDNode *C) const
SDValue LowerFLOGUnsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, bool IsLog10, SDNodeFlags Flags) const
SDValue combineFMinMaxLegacy(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const
Flags must be the select flags, not the compare (SELECT_CC flags come from the fcmp and say nothing a...
SDValue performMulhsCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue lowerFEXPUnsafeImpl(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags, bool IsExp10) const
bool isSDNodeAlwaysUniform(const SDNode *N) const override
bool isDesirableToCommuteWithShift(const SDNode *N, CombineLevel Level) const override
Return true if it is profitable to move this shift by a constant amount through its operand,...
SDValue performShlCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isCheapToSpeculateCtlz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic ctlz.
SDValue LowerSDIVREM(SDValue Op, SelectionDAG &DAG) const
bool isFNegFree(EVT VT) const override
Return true if an fneg operation is free to the point where it is never worthwhile to replace it with...
SDValue LowerFLOG10(SDValue Op, SelectionDAG &DAG) const
SDValue LowerINT_TO_FP64(SDValue Op, SelectionDAG &DAG, bool Signed) const
unsigned computeNumSignBitsForTargetInstr(GISelValueTracking &Analysis, Register R, const APInt &DemandedElts, const MachineRegisterInfo &MRI, unsigned Depth=0) const override
This method can be implemented by targets that want to expose additional information about sign bits ...
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
SDValue LowerFP_TO_FP16(SDValue Op, SelectionDAG &DAG) const
SDValue addTokenForArgument(SDValue Chain, SelectionDAG &DAG, MachineFrameInfo &MFI, int ClobberedFI) const
bool isConstantCheaperToNegate(SDValue N) const
bool isReassocProfitable(MachineRegisterInfo &MRI, Register N0, Register N1) const override
bool isKnownNeverNaNForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN=false, unsigned Depth=0) const override
If SNaN is false,.
static bool needsDenormHandlingF32(const SelectionDAG &DAG, SDValue Src, SDNodeFlags Flags)
uint32_t getImplicitParameterOffset(const MachineFunction &MF, const ImplicitParameter Param) const
Helper function that returns the byte offset of the given type of implicit parameter.
SDValue lowerFEXPF64(SDValue Op, SelectionDAG &DAG) const
SDValue LowerFFLOOR(SDValue Op, SelectionDAG &DAG) const
SDValue performSelectCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue performFNegCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const
bool isConstantCostlierToNegate(SDValue N) const
SDValue loadInputValue(SelectionDAG &DAG, const TargetRegisterClass *RC, EVT VT, const SDLoc &SL, const ArgDescriptor &Arg) const
SDValue lowerFEXP10Unsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags) const
Emit approx-funcs appropriate lowering for exp10.
bool shouldReduceLoadWidth(SDNode *Load, ISD::LoadExtType ExtType, EVT ExtVT, std::optional< unsigned > ByteOffset) const override
Return true if it is profitable to reduce a load to a smaller type.
SDValue LowerUINT_TO_FP(SDValue Op, SelectionDAG &DAG) const
bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const override
Return true if Op can create undef or poison from non-undef & non-poison operands.
bool isCheapToSpeculateCttz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic cttz.
SDValue performCtlz_CttzCombine(const SDLoc &SL, SDValue Cond, SDValue LHS, SDValue RHS, DAGCombinerInfo &DCI) const
SDValue performSraCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isSelectSupported(SelectSupportKind) const override
bool isZExtFree(Type *Src, Type *Dest) const override
Return true if any actual instruction that defines a value of type FromTy implicitly zero-extends the...
SDValue lowerFEXP2(SDValue Op, SelectionDAG &DAG) const
SDValue LowerCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
SDValue performSrlCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue lowerFEXP(SDValue Op, SelectionDAG &DAG) const
SDValue getIsLtSmallestNormal(SelectionDAG &DAG, SDValue Op, SDNodeFlags Flags) const
SDValue getIsFinite(SelectionDAG &DAG, SDValue Op, SDNodeFlags Flags) const
bool isLoadBitCastBeneficial(EVT, EVT, const SelectionDAG &DAG, const MachineMemOperand &MMO) const final
Return true if the following transform is beneficial: fold (conv (load x)) -> (load (conv*)x) On arch...
std::pair< SDValue, SDValue > splitVector(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HighVT, SelectionDAG &DAG) const
Split a vector value into two parts of types LoVT and HiVT.
AMDGPUTargetLowering(const TargetMachine &TM, const TargetSubtargetInfo &STI, const AMDGPUSubtarget &AMDGPUSTI)
SDValue LowerFLOGCommon(SDValue Op, SelectionDAG &DAG) const
SDValue foldFreeOpFromSelect(TargetLowering::DAGCombinerInfo &DCI, SDValue N) const
SDValue LowerINT_TO_FP32(SDValue Op, SelectionDAG &DAG, bool Signed) const
bool isFAbsFree(EVT VT) const override
Return true if an fabs operation is free to the point where it is never worthwhile to replace it with...
bool isInt64ImmLegal(SDNode *Val, SelectionDAG &DAG) const
Check whether value Val can be supported by v_mov_b64, for the current target.
SDValue loadStackInputValue(SelectionDAG &DAG, EVT VT, const SDLoc &SL, int64_t Offset) const
Similar to CreateLiveInRegister, except value maybe loaded from a stack slot rather than passed in a ...
SDValue LowerFLOG2(SDValue Op, SelectionDAG &DAG) const
static EVT getEquivalentMemType(LLVMContext &Context, EVT VT)
SDValue LowerCTLS(SDValue Op, SelectionDAG &DAG) const
Split a vector store into multiple scalar stores.
SDValue getSqrtEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps, bool &UseOneConstNR, bool Reciprocal) const override
Hooks for building estimates in place of slower divisions and square roots.
SDValue performTruncateCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerSINT_TO_FP(SDValue Op, SelectionDAG &DAG) const
static SDValue stripBitcast(SDValue Val)
SDValue LowerBlockAddress(SDValue Op, SelectionDAG &DAG) const
SDValue CreateLiveInRegister(SelectionDAG &DAG, const TargetRegisterClass *RC, Register Reg, EVT VT, const SDLoc &SL, bool RawReg=false) const
Helper function that adds Reg to the LiveIn list of the DAG's MachineFunction.
SDValue SplitVectorStore(SDValue Op, SelectionDAG &DAG) const
Split a vector store into 2 stores of half the vector.
SDValue LowerCTLZ_CTTZ(SDValue Op, SelectionDAG &DAG) const
SDValue getNegatedExpression(SDValue Op, SelectionDAG &DAG, bool LegalOperations, bool ForCodeSize, NegatibleCost &Cost, unsigned Depth) const override
Return the newly negated expression if the cost is not expensive and set the cost in Cost to indicate...
std::pair< SDValue, SDValue > split64BitValue(SDValue Op, SelectionDAG &DAG) const
Return 64-bit value Op as two 32-bit integers.
SDValue performMulCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue combineFMinMaxLegacyImpl(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const
SDValue getRecipEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps) const override
Return a reciprocal estimate value for the input operand.
SDValue LowerFNEARBYINT(SDValue Op, SelectionDAG &DAG) const
SDValue LowerSIGN_EXTEND_INREG(SDValue Op, SelectionDAG &DAG) const
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
std::pair< SDValue, SDValue > getScaledLogInput(SelectionDAG &DAG, const SDLoc SL, SDValue Op, SDNodeFlags Flags) const
If denormal handling is required return the scaled input to FLOG2, and the check for denormal range.
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
Selects the correct CCAssignFn for a given CallingConvention value.
bool SimplifyDemandedBitsForTargetNode(SDValue Op, const APInt &OriginalDemandedBits, const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth) const override
Attempt to simplify any target nodes based on the demanded bits/elts, returning true on success.
static bool allUsesHaveSourceMods(const SDNode *N, unsigned CostThreshold=4)
SDValue LowerFROUNDEVEN(SDValue Op, SelectionDAG &DAG) const
bool isFPImmLegal(const APFloat &Imm, EVT VT, bool ForCodeSize) const override
Returns true if the target can instruction select the specified FP immediate natively.
static unsigned numBitsUnsigned(SDValue Op, SelectionDAG &DAG)
SDValue lowerFEXPUnsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags) const
SDValue LowerFTRUNC(SDValue Op, SelectionDAG &DAG) const
SDValue LowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const
static bool allowApproxFunc(const SelectionDAG &DAG, SDNodeFlags Flags)
bool ShouldShrinkFPConstant(EVT VT) const override
If true, then instruction selection should seek to shrink the FP constant of the specified type to a ...
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
SDValue performStoreCombine(SDNode *N, DAGCombinerInfo &DCI) const
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
SDValue performRcpCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue getLoHalf64(SDValue Op, SelectionDAG &DAG) const
SDValue lowerCTLZResults(SDValue Op, SelectionDAG &DAG) const
SDValue LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG) const
SDValue performFAbsCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT64(SDValue Op, SelectionDAG &DAG, bool Signed) const
static bool shouldFoldFNegIntoSrc(SDNode *FNeg, SDValue FNegSrc)
bool isNarrowingProfitable(SDNode *N, EVT SrcVT, EVT DestVT) const override
Return true if it's profitable to narrow operations of type SrcVT to DestVT.
SDValue LowerFRINT(SDValue Op, SelectionDAG &DAG) const
SDValue performIntrinsicWOChainCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerUDIVREM(SDValue Op, SelectionDAG &DAG) const
SDValue performMulLoHiCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
void LowerUDIVREM64(SDValue Op, SelectionDAG &DAG, SmallVectorImpl< SDValue > &Results) const
SDValue WidenOrSplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Widen a suitably aligned v3 load.
SDValue LowerDIVREMToFloat(SDValue Op, SelectionDAG &DAG, bool sign) const
std::pair< EVT, EVT > getSplitDestVTs(const EVT &VT, SelectionDAG &DAG) const
Split a vector type into two parts.
SDValue getHiHalf64(SDValue Op, SelectionDAG &DAG) const
SDValue LowerINT_TO_FP16(SDValue Op, SelectionDAG &DAG, EVT FP16Ty) const
unsigned getVectorIdxWidth(const DataLayout &) const override
Returns the type to be used for the index operand vector operations.
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
bool bitwiseIsEqual(const APFloat &RHS) const
Definition APFloat.h:1548
opStatus add(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1285
opStatus multiply(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1303
static APFloat getSmallestNormalized(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) normalized finite number in the given semantics.
Definition APFloat.h:1262
APInt bitcastToAPInt() const
Definition APFloat.h:1475
static APFloat getInf(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Infinity.
Definition APFloat.h:1202
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1561
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
Definition APInt.h:203
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:255
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
Definition APInt.h:206
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:216
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:303
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:293
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
const BlockAddress * getBlockAddress() const
CCState - This class holds information needed while lowering arguments and return values.
static CCValAssign getCustomMem(unsigned ValNo, MVT ValVT, int64_t Offset, MVT LocVT, LocInfo HTP)
const APFloat & getValueAPF() const
bool isNegative() const
Return true if the value is negative.
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Diagnostic information for unsupported feature in backend.
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
Definition Function.cpp:360
iterator_range< arg_iterator > args()
Definition Function.h:877
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
Machine Value Type.
static auto integer_fixedlen_vector_valuetypes()
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
static auto integer_valuetypes()
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
int getObjectIndexBegin() const
Return the minimum frame object index.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Representation of each machine instruction.
A description of a memory reference used in the backend.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOInvariant
The memory access always returns the same value (or traps).
Flags getFlags() const
Return the raw flags of the source value,.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool isLiveIn(Register Reg) const
LLVM_ABI Register getLiveInVirtReg(MCRegister PReg) const
getLiveInVirtReg - If PReg is a live-in physical register, return the corresponding live-in virtual r...
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
This is an abstract virtual class for memory operations.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
Align getAlign() const
bool isSimple() const
Returns true if the memory operation is neither atomic or volatile.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
bool isInvariant() const
EVT getMemoryVT() const
Return the type of the in-memory value.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
const DebugLoc & getDebugLoc() const
Represents one node in the SelectionDAG.
ArrayRef< SDUse > ops() const
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
SDNodeFlags getFlags() const
SDVTList getVTList() const
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
iterator_range< user_iterator > users()
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
unsigned getOpcode() const
unsigned getNumOperands() const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
SIModeRegisterDefaults getMode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI bool isKnownNeverLogicalZero(SDValue Op, const APInt &DemandedElts, unsigned Depth=0) const
Test whether the given floating point SDValue (or all elements of it, if it is a vector) is known to ...
SDValue getExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT, unsigned Opcode)
Convert Op, which must be of integer type, to the integer type VT, by either any/sign/zero-extending ...
LLVM_ABI unsigned ComputeMaxSignificantBits(SDValue Op, unsigned Depth=0) const
Get the upper bound on bit size for this Value Op as a signed integer.
const SDValue & getRoot() const
Return the root tag of the SelectionDAG.
const TargetSubtargetInfo & getSubtarget() const
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI void ExtractVectorElements(SDValue Op, SmallVectorImpl< SDValue > &Args, unsigned Start=0, unsigned Count=0, EVT EltVT=EVT())
Append the extracted elements from Start to Count out of the vector Op in Args.
LLVM_ABI SDValue getFreeze(SDValue V)
Return a freeze using the SDLoc of the value operand.
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
const TargetLowering & getTargetLoweringInfo() const
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
SDValue getSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS, SDValue RHS, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build Select's if you just have operands and don't want to check...
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
bool isConstantValueOfAnyType(SDValue N) const
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode Cond, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build SelectCC's if you just have an ISD::CondCode instead of an...
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(SDValue Op, UndefPoisonKind Kind=UndefPoisonKind::UndefOrPoison, unsigned Depth=0) const
Return true if this function can prove that Op is never poison and, Kind can be used to track poison ...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI bool isKnownNeverNaN(SDValue Op, const APInt &DemandedElts, bool SNaN=false, unsigned Depth=0) const
Test whether the given SDValue (or all elements of it, if it is a vector) is known to never be NaN in...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
SDValue getTargetBlockAddress(const BlockAddress *BA, EVT VT, int64_t Offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
const SDValue & setRoot(SDValue N)
Set the current root tag of the SelectionDAG.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
This class is used to represent ISD::STORE nodes.
const SDValue & getBasePtr() const
const SDValue & getValue() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
void setMaxDivRemBitWidthSupported(unsigned SizeInBits)
Set the size in bits of the maximum div/rem the backend supports.
bool PredictableSelectIsExpensive
Tells the code generator that select is more expensive than a branch if the branch is usually predict...
virtual bool shouldReduceLoadWidth(SDNode *Load, ISD::LoadExtType ExtTy, EVT NewVT, std::optional< unsigned > ByteOffset=std::nullopt) const
Return true if it is profitable to reduce a load to a smaller type.
unsigned MaxStoresPerMemcpyOptSize
Likewise for functions with the OptSize attribute.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
unsigned MaxGluedStoresPerMemcpy
Specify max number of store instructions to glue in inlined memcpy.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
void addBypassSlowDiv(unsigned int SlowBitWidth, unsigned int FastBitWidth)
Tells the code generator which bitwidths to bypass.
void setMaxLargeFPConvertBitWidthSupported(unsigned SizeInBits)
Set the size in bits of the maximum fp to/from int conversion the backend supports.
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
SelectSupportKind
Enum that describes what type of support for selects the target has.
virtual bool allowsMisalignedMemoryAccesses(EVT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *=nullptr) const
Determine if the target supports unaligned memory accesses.
unsigned MaxStoresPerMemsetOptSize
Likewise for functions with the OptSize attribute.
EVT getShiftAmountTy(EVT LHSTy, const DataLayout &DL) const
Returns the type for the shift amount of a shift opcode.
unsigned MaxStoresPerMemmove
Specify maximum number of store instructions per memmove call.
virtual EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const
Return the ValueType of the result of SETCC operations.
unsigned MaxStoresPerMemmoveOptSize
Likewise for functions with the OptSize attribute.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
void setSupportsUnalignedAtomics(bool UnalignedSupported)
Sets whether unaligned atomic operations are supported.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
unsigned MaxStoresPerMemset
Specify maximum number of store instructions per memset call.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
void setMinCmpXchgSizeInBits(unsigned SizeInBits)
Sets the minimum cmpxchg or ll/sc size supported by the backend.
void AddPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
If Opc/OrigVT is specified as being promoted, the promotion code defaults to trying a larger integer/...
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
unsigned GatherAllAliasesMaxDepth
Depth that GatherAllAliases should continue looking for chain dependencies when trying to find a more...
NegatibleCost
Enum that specifies when a float negation is beneficial.
bool allowsMemoryAccessForAlignment(LLVMContext &Context, const DataLayout &DL, EVT VT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const
This function returns true if the memory access is aligned or if the target allows this specific unal...
unsigned MaxStoresPerMemcpy
Specify maximum number of store instructions per memcpy call.
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
void setJumpIsExpensive(bool isExpensive=true)
Tells the code generator not to expand logic operations on comparison predicates into separate sequen...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
SDValue scalarizeVectorStore(StoreSDNode *ST, SelectionDAG &DAG) const
SDValue SimplifyMultipleUseDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, SelectionDAG &DAG, unsigned Depth=0) const
More limited version of SimplifyDemandedBits that can be used to "lookthrough" ops that don't contrib...
SDValue expandUnalignedStore(StoreSDNode *ST, SelectionDAG &DAG) const
Expands an unaligned store to 2 half-size stores for integer values, and possibly more for vectors.
bool ShrinkDemandedConstant(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, TargetLoweringOpt &TLO) const
Check to see if the specified operand of the specified instruction is a constant integer.
std::pair< SDValue, SDValue > expandUnalignedLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Expands an unaligned load to 2 half-size loads for an integer, and possibly more for vectors.
virtual SDValue getNegatedExpression(SDValue Op, SelectionDAG &DAG, bool LegalOps, bool OptForSize, NegatibleCost &Cost, unsigned Depth=0) const
Return the newly negated expression if the cost is not expensive and set the cost in Cost to indicate...
std::pair< SDValue, SDValue > scalarizeVectorLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Turn load of vector type into a load of the individual elements.
bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Op.
TargetLowering(const TargetLowering &)=delete
virtual bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const
Return true if Op can create undef or poison from non-undef & non-poison operands.
Primary interface to the complete machine description for the target machine.
TargetSubtargetInfo - Generic base class for all target subtargets.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
LLVM Value Representation.
Definition Value.h:75
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
bool isIntrinsicAlwaysUniform(unsigned IntrID)
TargetExtType * isNamedBarrier(const GlobalVariable &GV)
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
bool isUniformMMO(const MachineMemOperand *MMO)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
Definition CallingConv.h:47
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:41
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:829
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:275
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:602
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:789
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ ADDC
Carry-setting nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:294
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
Definition ISDOpcodes.h:524
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:863
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:520
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:890
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:586
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:749
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:280
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:254
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:798
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BRIND
BRIND - Indirect branch.
@ BR_JT
BR_JT - Jumptable branch.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:543
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:550
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:806
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:247
@ CTLS
Count leading redundant sign bits.
Definition ISDOpcodes.h:802
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:706
@ STRICT_FP16_TO_FP
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:651
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:616
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ EntryToken
EntryToken - This is the marker used to indicate the start of a region.
Definition ISDOpcodes.h:48
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:578
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:224
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:821
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:898
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:729
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:815
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:328
@ INLINEASM_BR
INLINEASM_BR - Branching version of inline asm. Used by asm-goto.
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:205
@ ADDE
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:304
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:567
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:53
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:797
@ FFREXP
FFREXP - frexp, extract fractional and exponent component of a floating-point value.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ INLINEASM
INLINEASM - Represents an inline asm block.
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:955
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:62
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:536
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:213
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:558
bool isNormalStore(const SDNode *N)
Returns true if the specified node is a non-truncating and unindexed store.
LLVM_ABI CondCode getSetCCInverse(CondCode Operation, EVT Type)
Return the operation corresponding to !(X op Y), where 'op' is a valid SetCC operation.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
initializer< Ty > init(const Ty &Val)
constexpr double ln2
constexpr double ln10
constexpr float log2ef
Definition MathExtras.h:52
constexpr double log2e
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
InstructionCost Cost
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
LLVM_ABI void ComputeValueVTs(const TargetLowering &TLI, const DataLayout &DL, Type *Ty, SmallVectorImpl< EVT > &ValueVTs, SmallVectorImpl< EVT > *MemVTs=nullptr, SmallVectorImpl< TypeSize > *Offsets=nullptr, TypeSize StartingOffset=TypeSize::getZero())
ComputeValueVTs - Given an LLVM IR type, compute a sequence of EVTs that represent all the individual...
Definition Analysis.cpp:119
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
SDValue peekFPSignOps(SDValue Val)
Strip fabs/fneg/fcopysign from a value to get the underlying source.
LLVM_ABI ConstantFPSDNode * isConstOrConstSplatFP(SDValue N, bool AllowUndefs=false)
Returns the SDNode if it is a constant splat BuildVector or constant float.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
CombineLevel
Definition DAGCombine.h:15
@ AfterLegalizeDAG
Definition DAGCombine.h:19
@ BeforeLegalizeTypes
Definition DAGCombine.h:16
@ AfterLegalizeTypes
Definition DAGCombine.h:17
To bit_cast(const From &from) noexcept
Definition bit.h:90
@ Mul
Product of integers.
@ Add
Sum of integers.
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
UndefPoisonKind
Enumeration to track whether we are interested in Undef, Poison, or both.
Definition UndefPoison.h:20
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:201
static cl::opt< unsigned > CostThreshold("dfa-cost-threshold", cl::desc("Maximum cost accepted for the transformation"), cl::Hidden, cl::init(50))
APFloat neg(APFloat X)
Returns the negated value of the argument.
Definition APFloat.h:1727
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
MCRegister getRegister() const
unsigned getStackOffset() const
DenormalModeKind Input
Denormal treatment kind for floating point instruction inputs in the default floating-point environme...
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
static constexpr DenormalMode getPreserveSign()
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
EVT getPow2VectorType(LLVMContext &Context) const
Widens the length of the given vector EVT up to the nearest power of 2 and returns that type.
Definition ValueTypes.h:508
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
Definition ValueTypes.h:129
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
Definition ValueTypes.h:307
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
bool isByteSized() const
Return true if the bit size is a multiple of 8.
Definition ValueTypes.h:266
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
EVT getHalfSizedIntegerVT(LLVMContext &Context) const
Finds the smallest simple value type that is greater than or equal to half the width of this EVT.
Definition ValueTypes.h:453
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
Definition ValueTypes.h:501
TypeSize getStoreSizeInBits() const
Return the number of bits overwritten by a store of the specified value type.
Definition ValueTypes.h:435
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
EVT getRoundIntegerType(LLVMContext &Context) const
Rounds the bit-width of the given integer EVT up to the nearest power of two (and at least to eight),...
Definition ValueTypes.h:442
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool bitsGE(EVT VT) const
Return true if this has no less bits than VT.
Definition ValueTypes.h:315
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isExtended() const
Test if the given EVT is extended (as opposed to being simple).
Definition ValueTypes.h:150
EVT changeElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a type whose attributes match ourselves with the exception of the element type that i...
Definition ValueTypes.h:121
LLVM_ABI const fltSemantics & getFltSemantics() const
Returns an APFloat semantics tag appropriate for the value type.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool bitsLE(EVT VT) const
Return true if this has no more bits than VT.
Definition ValueTypes.h:331
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
InputArg - This struct carries flags and type information about a single incoming (formal) argument o...
MVT VT
Legalized type of this argument part.
bool isUnknown() const
Returns true if we don't know any bits.
Definition KnownBits.h:64
KnownBits trunc(unsigned BitWidth) const
Return known bits for a truncation of the value we're tracking.
Definition KnownBits.h:165
KnownBits zext(unsigned BitWidth) const
Return known bits for a zero extension of the value we're tracking.
Definition KnownBits.h:176
unsigned countMaxActiveBits() const
Returns the maximum number of bits needed to represent all possible unsigned values with these known ...
Definition KnownBits.h:310
KnownBits intersectWith(const KnownBits &RHS) const
Returns KnownBits information that is known to be true for both this and RHS.
Definition KnownBits.h:325
KnownBits sext(unsigned BitWidth) const
Return known bits for a sign extension of the value we're tracking.
Definition KnownBits.h:184
unsigned countMinLeadingZeros() const
Returns the minimum number of leading zero bits.
Definition KnownBits.h:262
bool isNegative() const
Returns true if this value is known to be negative.
Definition KnownBits.h:103
static LLVM_ABI KnownBits mul(const KnownBits &LHS, const KnownBits &RHS, bool NoUndefSelfMultiply=false)
Compute known bits resulting from multiplying LHS and RHS.
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
LLVM_ABI bool isDereferenceable(unsigned Size, LLVMContext &C, const DataLayout &DL) const
Return true if memory region [V, V+Offset+Size) is known to be dereferenceable.
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
MachinePointerInfo getWithOffset(int64_t O) const
These are IR-level optimization flags that may be propagated to SDNodes.
void setAllowContract(bool b)
bool hasNoSignedZeros() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
LLVM_ABI void AddToWorklist(SDNode *N)
LLVM_ABI SDValue CombineTo(SDNode *N, ArrayRef< SDValue > To, bool AddTo=true)
LLVM_ABI void CommitTargetLoweringOpt(const TargetLoweringOpt &TLO)
A convenience struct that encapsulates a DAG, and two SDValues for returning information from TargetL...