LLVM 24.0.0git
LoongArchISelLowering.cpp
Go to the documentation of this file.
1//=- LoongArchISelLowering.cpp - LoongArch DAG Lowering Implementation ---===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines the interfaces that LoongArch uses to lower LLVM code into
10// a selection DAG.
11//
12//===----------------------------------------------------------------------===//
13
15#include "LoongArch.h"
19#include "LoongArchSubtarget.h"
23#include "llvm/ADT/SmallSet.h"
24#include "llvm/ADT/Statistic.h"
30#include "llvm/IR/IRBuilder.h"
32#include "llvm/IR/IntrinsicsLoongArch.h"
34#include "llvm/Support/Debug.h"
39
40using namespace llvm;
41
42#define DEBUG_TYPE "loongarch-isel-lowering"
43
44STATISTIC(NumTailCalls, "Number of tail calls");
45
54
56 "loongarch-materialize-float-imm", cl::Hidden,
57 cl::desc("Maximum number of instructions used (including code sequence "
58 "to generate the value and moving the value to FPR) when "
59 "materializing floating-point immediates (default = 3)"),
61 cl::values(clEnumValN(NoMaterializeFPImm, "0", "Use constant pool"),
63 "Materialize FP immediate within 2 instructions"),
65 "Materialize FP immediate within 3 instructions"),
67 "Materialize FP immediate within 4 instructions"),
69 "Materialize FP immediate within 5 instructions"),
71 "Materialize FP immediate within 6 instructions "
72 "(behaves same as 5 on loongarch64)")));
73
74static cl::opt<bool> ZeroDivCheck("loongarch-check-zero-division", cl::Hidden,
75 cl::desc("Trap on integer division by zero."),
76 cl::init(false));
77
79 const LoongArchSubtarget &STI)
80 : TargetLowering(TM, STI), Subtarget(STI) {
81
82 MVT GRLenVT = Subtarget.getGRLenVT();
83
84 // Set up the register classes.
85
86 addRegisterClass(GRLenVT, &LoongArch::GPRRegClass);
87 if (Subtarget.hasBasicF())
88 addRegisterClass(MVT::f32, &LoongArch::FPR32RegClass);
89 if (Subtarget.hasBasicD())
90 addRegisterClass(MVT::f64, &LoongArch::FPR64RegClass);
91
92 static const MVT::SimpleValueType LSXVTs[] = {
93 MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64, MVT::v4f32, MVT::v2f64};
94 static const MVT::SimpleValueType LASXVTs[] = {
95 MVT::v32i8, MVT::v16i16, MVT::v8i32, MVT::v4i64, MVT::v8f32, MVT::v4f64};
96
97 if (Subtarget.hasExtLSX())
98 for (MVT VT : LSXVTs)
99 addRegisterClass(VT, &LoongArch::LSX128RegClass);
100
101 if (Subtarget.hasExtLASX())
102 for (MVT VT : LASXVTs)
103 addRegisterClass(VT, &LoongArch::LASX256RegClass);
104
105 // Set operations for LA32 and LA64.
106
108 MVT::i1, Promote);
109
116
119 GRLenVT, Custom);
120
122
127
129 setOperationAction(ISD::TRAP, MVT::Other, Legal);
130
134
136
137 // BITREV/REVB requires the 32S feature.
138 if (STI.has32S()) {
139 // Expand bitreverse.i16 with native-width bitrev and shift for now, before
140 // we get to know which of sll and revb.2h is faster.
143
144 // LA32 does not have REVB.2W and REVB.D due to the 64-bit operands, and
145 // the narrower REVB.W does not exist. But LA32 does have REVB.2H, so i16
146 // and i32 could still be byte-swapped relatively cheaply.
148 } else {
156 }
157
164
167
168 // Set operations for LA64 only.
169
170 if (Subtarget.is64Bit()) {
188
192 Custom);
194 }
195
196 // Set operations for LA32 only.
197
198 if (!Subtarget.is64Bit()) {
204 if (Subtarget.hasBasicD())
206 }
207
209
210 static const ISD::CondCode FPCCToExpand[] = {
213
214 // Set operations for 'F' feature.
215
216 if (Subtarget.hasBasicF()) {
217 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::f16, Expand);
218 setTruncStoreAction(MVT::f32, MVT::f16, Expand);
219 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::bf16, Expand);
220 setTruncStoreAction(MVT::f32, MVT::bf16, Expand);
221 setCondCodeAction(FPCCToExpand, MVT::f32, Expand);
222
241 Subtarget.isSoftFPABI() ? LibCall : Custom);
243 Subtarget.isSoftFPABI() ? LibCall : Custom);
246 Subtarget.isSoftFPABI() ? LibCall : Custom);
249
250 if (Subtarget.is64Bit())
252
253 if (!Subtarget.hasBasicD()) {
255 if (Subtarget.is64Bit()) {
258 }
259 }
260 }
261
262 // Set operations for 'D' feature.
263
264 if (Subtarget.hasBasicD()) {
265 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f16, Expand);
266 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand);
267 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::bf16, Expand);
268 setTruncStoreAction(MVT::f64, MVT::bf16, Expand);
269 setTruncStoreAction(MVT::f64, MVT::f16, Expand);
270 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
271 setCondCodeAction(FPCCToExpand, MVT::f64, Expand);
272
292 Subtarget.isSoftFPABI() ? LibCall : Custom);
295 Subtarget.isSoftFPABI() ? LibCall : Custom);
296
297 if (Subtarget.is64Bit())
299 }
300
301 // Set operations for 'LSX' feature.
302
303 if (Subtarget.hasExtLSX()) {
305 // Expand all truncating stores and extending loads.
306 for (MVT InnerVT : MVT::fixedlen_vector_valuetypes()) {
307 setTruncStoreAction(VT, InnerVT, Expand);
310 setLoadExtAction(ISD::EXTLOAD, VT, InnerVT, Expand);
311 }
312 // By default everything must be expanded. Then we will selectively turn
313 // on ones that can be effectively codegen'd.
314 for (unsigned Op = 0; Op < ISD::BUILTIN_OP_END; ++Op)
316 }
317
318 for (MVT VT : LSXVTs) {
322
326
331 }
332 for (MVT VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64}) {
335 Legal);
337 VT, Legal);
344 Expand);
359 }
360 for (MVT VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32})
362 for (MVT VT : {MVT::v8i16, MVT::v4i32, MVT::v2i64})
364 for (MVT VT : {MVT::v4i32, MVT::v2i64}) {
367 }
369 for (MVT VT : {MVT::v4f32, MVT::v2f64}) {
377 VT, Expand);
385 }
387 setOperationAction(ISD::FCEIL, {MVT::f32, MVT::f64}, Legal);
388 setOperationAction(ISD::FFLOOR, {MVT::f32, MVT::f64}, Legal);
389 setOperationAction(ISD::FTRUNC, {MVT::f32, MVT::f64}, Legal);
390 setOperationAction(ISD::FROUNDEVEN, {MVT::f32, MVT::f64}, Legal);
391
392 for (MVT VT :
393 {MVT::v16i8, MVT::v8i8, MVT::v4i8, MVT::v2i8, MVT::v8i16, MVT::v4i16,
394 MVT::v2i16, MVT::v4i32, MVT::v2i32, MVT::v2i64}) {
404 }
407 // We want to legalize this to an f64 load rather than an i64 load.
408 setOperationAction(ISD::LOAD, MVT::v2f32, Custom);
409 for (MVT VT : {MVT::v2i64, MVT::v4i32, MVT::v8i16})
411 for (MVT VT : {MVT::v16i16, MVT::v8i32, MVT::v4i64, MVT::v16i32, MVT::v8i64,
412 MVT::v16i64})
414 }
415
416 // Set operations for 'LASX' feature.
417
418 if (Subtarget.hasExtLASX()) {
419 for (MVT VT : LASXVTs) {
423
429
433 }
434 for (MVT VT : {MVT::v4i64, MVT::v8i32, MVT::v16i16, MVT::v32i8}) {
437 Legal);
439 VT, Legal);
446 Expand);
462 }
463 for (MVT VT : {MVT::v32i8, MVT::v16i16, MVT::v8i32})
465 for (MVT VT : {MVT::v16i16, MVT::v8i32, MVT::v4i64})
467 for (MVT VT : {MVT::v8i32, MVT::v4i32, MVT::v4i64}) {
471 }
472 for (MVT VT : {MVT::v8f32, MVT::v4f64}) {
480 VT, Expand);
488 }
491 for (MVT VT : {MVT::v4i64, MVT::v8i32, MVT::v16i16}) {
495 }
496 for (MVT VT :
497 {MVT::v2i64, MVT::v4i32, MVT::v4i64, MVT::v8i16, MVT::v8i32}) {
500 }
501 for (MVT VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32})
503 }
504
505 // Set DAG combine for LA32 and LA64.
506 if (Subtarget.hasBasicF()) {
508 }
509
514
515 // Set DAG combine for 'LSX' feature.
516
517 if (Subtarget.hasExtLSX()) {
529 }
530
531 // Set DAG combine for 'LASX' feature.
532 if (Subtarget.hasExtLASX()) {
535 }
536
537 // Compute derived properties from the register classes.
538 computeRegisterProperties(Subtarget.getRegisterInfo());
539
541
544
545 setMaxAtomicSizeInBitsSupported(Subtarget.getGRLen());
546
548
549 // Function alignments.
551 // Set preferred alignments.
552 setPrefFunctionAlignment(Subtarget.getPrefFunctionAlignment());
553 setPrefLoopAlignment(Subtarget.getPrefLoopAlignment());
554 setMaxBytesForAlignment(Subtarget.getMaxBytesForAlignment());
555
556 // cmpxchg sizes down to 8 bits become legal if LAMCAS is available.
557 if (Subtarget.hasLAMCAS())
559
560 if (Subtarget.hasSCQ()) {
563 }
564
565 // Disable strict node mutation.
566 IsStrictFPEnabled = true;
567}
568
570 const GlobalAddressSDNode *GA) const {
571 // In order to maximise the opportunity for common subexpression elimination,
572 // keep a separate ADD node for the global address offset instead of folding
573 // it in the global address node. Later peephole optimisations may choose to
574 // fold it back in when profitable.
575 return false;
576}
577
579 SelectionDAG &DAG) const {
580 switch (Op.getOpcode()) {
582 return lowerATOMIC_FENCE(Op, DAG);
584 return lowerEH_DWARF_CFA(Op, DAG);
586 return lowerGlobalAddress(Op, DAG);
588 return lowerGlobalTLSAddress(Op, DAG);
590 return lowerINTRINSIC_WO_CHAIN(Op, DAG);
592 return lowerINTRINSIC_W_CHAIN(Op, DAG);
594 return lowerINTRINSIC_VOID(Op, DAG);
596 return lowerBlockAddress(Op, DAG);
597 case ISD::JumpTable:
598 return lowerJumpTable(Op, DAG);
599 case ISD::SHL_PARTS:
600 return lowerShiftLeftParts(Op, DAG);
601 case ISD::SRA_PARTS:
602 return lowerShiftRightParts(Op, DAG, true);
603 case ISD::SRL_PARTS:
604 return lowerShiftRightParts(Op, DAG, false);
606 return lowerConstantPool(Op, DAG);
607 case ISD::FP_TO_SINT:
608 return lowerFP_TO_SINT(Op, DAG);
609 case ISD::FP_TO_UINT:
610 return lowerFP_TO_UINT(Op, DAG);
611 case ISD::BITCAST:
612 return lowerBITCAST(Op, DAG);
613 case ISD::UINT_TO_FP:
614 return lowerUINT_TO_FP(Op, DAG);
615 case ISD::SINT_TO_FP:
616 return lowerSINT_TO_FP(Op, DAG);
617 case ISD::VASTART:
618 return lowerVASTART(Op, DAG);
619 case ISD::FRAMEADDR:
620 return lowerFRAMEADDR(Op, DAG);
621 case ISD::RETURNADDR:
622 return lowerRETURNADDR(Op, DAG);
624 return lowerSET_ROUNDING(Op, DAG);
626 return lowerGET_ROUNDING(Op, DAG);
628 return lowerWRITE_REGISTER(Op, DAG);
630 return lowerINSERT_VECTOR_ELT(Op, DAG);
632 return lowerEXTRACT_VECTOR_ELT(Op, DAG);
634 return lowerBUILD_VECTOR(Op, DAG);
636 return lowerCONCAT_VECTORS(Op, DAG);
638 return lowerVECTOR_SHUFFLE(Op, DAG);
639 case ISD::BITREVERSE:
640 return lowerBITREVERSE(Op, DAG);
642 return lowerSCALAR_TO_VECTOR(Op, DAG);
643 case ISD::PREFETCH:
644 return lowerPREFETCH(Op, DAG);
645 case ISD::SELECT:
646 return lowerSELECT(Op, DAG);
647 case ISD::BRCOND:
648 return lowerBRCOND(Op, DAG);
649 case ISD::FP_TO_FP16:
650 return lowerFP_TO_FP16(Op, DAG);
651 case ISD::FP16_TO_FP:
652 return lowerFP16_TO_FP(Op, DAG);
653 case ISD::FP_TO_BF16:
654 return lowerFP_TO_BF16(Op, DAG);
655 case ISD::BF16_TO_FP:
656 return lowerBF16_TO_FP(Op, DAG);
658 return lowerVECREDUCE_ADD(Op, DAG);
659 case ISD::ROTL:
660 case ISD::ROTR:
661 return lowerRotate(Op, DAG);
669 return lowerVECREDUCE(Op, DAG);
670 case ISD::ConstantFP:
671 return lowerConstantFP(Op, DAG);
672 case ISD::SETCC:
673 return lowerSETCC(Op, DAG);
674 case ISD::FP_ROUND:
675 return lowerFP_ROUND(Op, DAG);
676 case ISD::FP_EXTEND:
677 return lowerFP_EXTEND(Op, DAG);
679 return lowerSIGN_EXTEND_VECTOR_INREG(Op, DAG);
681 return lowerDYNAMIC_STACKALLOC(Op, DAG);
682 case ISD::ANY_EXTEND:
683 return lowerANY_EXTEND(Op, DAG);
684 }
685 return SDValue();
686}
687
688// Helper to attempt to return a cheaper, bit-inverted version of \p V.
690 // TODO: don't always ignore oneuse constraints.
691 V = peekThroughBitcasts(V);
692 EVT VT = V.getValueType();
693
694 // Match not(xor X, -1) -> X.
695 if (V.getOpcode() == ISD::XOR &&
696 (ISD::isBuildVectorAllOnes(V.getOperand(1).getNode()) ||
697 isAllOnesConstant(V.getOperand(1))))
698 return V.getOperand(0);
699
700 // Match not(extract_subvector(not(X)) -> extract_subvector(X).
701 if (V.getOpcode() == ISD::EXTRACT_SUBVECTOR &&
702 (isNullConstant(V.getOperand(1)) || V.getOperand(0).hasOneUse())) {
703 if (SDValue Not = isNOT(V.getOperand(0), DAG)) {
704 Not = DAG.getBitcast(V.getOperand(0).getValueType(), Not);
705 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(Not), VT, Not,
706 V.getOperand(1));
707 }
708 }
709
710 // Match not(SplatVector(not(X)) -> SplatVector(X).
711 if (V.getOpcode() == ISD::BUILD_VECTOR) {
712 if (SDValue SplatValue =
713 cast<BuildVectorSDNode>(V.getNode())->getSplatValue()) {
714 if (!V->isOnlyUserOf(SplatValue.getNode()))
715 return SDValue();
716
717 if (SDValue Not = isNOT(SplatValue, DAG)) {
718 Not = DAG.getBitcast(V.getOperand(0).getValueType(), Not);
719 return DAG.getSplat(VT, SDLoc(Not), Not);
720 }
721 }
722 }
723
724 // Match not(or(not(X),not(Y))) -> and(X, Y).
725 if (V.getOpcode() == ISD::OR && DAG.getTargetLoweringInfo().isTypeLegal(VT) &&
726 V.getOperand(0).hasOneUse() && V.getOperand(1).hasOneUse()) {
727 // TODO: Handle cases with single NOT operand -> VANDN
728 if (SDValue Op1 = isNOT(V.getOperand(1), DAG))
729 if (SDValue Op0 = isNOT(V.getOperand(0), DAG))
730 return DAG.getNode(ISD::AND, SDLoc(V), VT, DAG.getBitcast(VT, Op0),
731 DAG.getBitcast(VT, Op1));
732 }
733
734 // TODO: Add more matching patterns. Such as,
735 // not(concat_vectors(not(X), not(Y))) -> concat_vectors(X, Y).
736 // not(slt(C, X)) -> slt(X - 1, C)
737 return SDValue();
738}
739
740// Combine two ISD::FP_ROUND / LoongArchISD::VFCVT nodes with same type to
741// LoongArchISD::VFCVT. For example:
742// x1 = fp_round x, 0
743// y1 = fp_round y, 0
744// z = concat_vectors x1, y1
745// Or
746// x1 = LoongArch::VFCVT undef, x
747// y1 = LoongArch::VFCVT undef, y
748// z = LoongArchISD::VPACKEV y1, x1; or LoongArchISD::VPERMI y1, x1, 68
749// can be combined to:
750// z = LoongArch::VFCVT y, x
752 const LoongArchSubtarget &Subtarget) {
753 assert(((N->getOpcode() == ISD::CONCAT_VECTORS && N->getNumOperands() == 2) ||
754 (N->getOpcode() == LoongArchISD::VPACKEV) ||
755 (N->getOpcode() == LoongArchISD::VPERMI)) &&
756 "Invalid Node");
757
758 SDValue Op0 = peekThroughBitcasts(N->getOperand(0));
759 SDValue Op1 = peekThroughBitcasts(N->getOperand(1));
760 unsigned Opcode0 = Op0.getOpcode();
761 unsigned Opcode1 = Op1.getOpcode();
762 if (Opcode0 != Opcode1)
763 return SDValue();
764
765 if (Opcode0 != ISD::FP_ROUND && Opcode0 != LoongArchISD::VFCVT)
766 return SDValue();
767
768 // Check if two nodes have only one use.
769 if (!Op0.hasOneUse() || !Op1.hasOneUse())
770 return SDValue();
771
772 EVT VT = N.getValueType();
773 EVT SVT0 = Op0.getValueType();
774 EVT SVT1 = Op1.getValueType();
775 // Check if two nodes have the same result type.
776 if (SVT0 != SVT1)
777 return SDValue();
778
779 // Check if two nodes have the same operand type.
780 EVT SSVT0 = Op0.getOperand(0).getValueType();
781 EVT SSVT1 = Op1.getOperand(0).getValueType();
782 if (SSVT0 != SSVT1)
783 return SDValue();
784
785 if (N->getOpcode() == ISD::CONCAT_VECTORS && Opcode0 == ISD::FP_ROUND) {
786 if (Subtarget.hasExtLASX() && VT.is256BitVector() && SVT0 == MVT::v4f32 &&
787 SSVT0 == MVT::v4f64) {
788 // A vector_shuffle is required in the final step, as xvfcvt instruction
789 // operates on each 128-bit segament as a lane.
790 SDValue Res = DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v8f32,
791 Op1.getOperand(0), Op0.getOperand(0));
792 SDValue Undef = DAG.getUNDEF(Res.getValueType());
793 // After VFCVT, the high part of Res comes from the high parts of Op0 and
794 // Op1, and the low part comes from the low parts of Op0 and Op1. However,
795 // the desired order requires Op0 to fully occupy the lower half and Op1
796 // the upper half of Res. The Mask reorders the elements of Res to achieve
797 // this:
798 // - The first four elements (0, 1, 4, 5) come from Op0.
799 // - The next four elements (2, 3, 6, 7) come from Op1.
800 SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
801 Res = DAG.getVectorShuffle(Res.getValueType(), DL, Res, Undef, Mask);
802 return DAG.getBitcast(VT, Res);
803 }
804 }
805
806 if ((N->getOpcode() == LoongArchISD::VPACKEV ||
807 N->getOpcode() == LoongArchISD::VPERMI) &&
808 Opcode0 == LoongArchISD::VFCVT) {
809 // For VPACKEV or VPERMI, check if the first operation of VFCVT is undef.
810 if (!Op0.getOperand(0).isUndef() || !Op1.getOperand(0).isUndef())
811 return SDValue();
812
813 if (!Subtarget.hasExtLSX() || SVT0 != MVT::v4f32 || SSVT0 != MVT::v2f64)
814 return SDValue();
815
816 if (N->getOpcode() == LoongArchISD::VPACKEV &&
817 (VT == MVT::v2i64 || VT == MVT::v2f64)) {
818 SDValue Res = DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v4f32,
819 Op0.getOperand(1), Op1.getOperand(1));
820 return DAG.getBitcast(VT, Res);
821 }
822
823 if (N->getOpcode() == LoongArchISD::VPERMI && VT == MVT::v4f32) {
824 int64_t Imm = cast<ConstantSDNode>(N->getOperand(2))->getSExtValue();
825 if (Imm != 68)
826 return SDValue();
827 return DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v4f32, Op0.getOperand(1),
828 Op1.getOperand(1));
829 }
830 }
831
832 return SDValue();
833}
834
835SDValue LoongArchTargetLowering::lowerFP_ROUND(SDValue Op,
836 SelectionDAG &DAG) const {
837 SDLoc DL(Op);
838 SDValue In = Op.getOperand(0);
839 MVT VT = Op.getSimpleValueType();
840 MVT SVT = In.getSimpleValueType();
841
842 if (VT == MVT::v4f32 && SVT == MVT::v4f64) {
843 SDValue Lo, Hi;
844 std::tie(Lo, Hi) = DAG.SplitVector(In, DL);
845 return DAG.getNode(LoongArchISD::VFCVT, DL, VT, Hi, Lo);
846 }
847
848 return SDValue();
849}
850
851SDValue LoongArchTargetLowering::lowerFP_EXTEND(SDValue Op,
852 SelectionDAG &DAG) const {
853
854 SDLoc DL(Op);
855 EVT VT = Op.getValueType();
856 SDValue Src = Op->getOperand(0);
857 EVT SVT = Src.getValueType();
858
859 bool V2F32ToV2F64 =
860 VT == MVT::v2f64 && SVT == MVT::v2f32 && Subtarget.hasExtLSX();
861 bool V4F32ToV4F64 =
862 VT == MVT::v4f64 && SVT == MVT::v4f32 && Subtarget.hasExtLASX();
863 if (!V2F32ToV2F64 && !V4F32ToV4F64)
864 return SDValue();
865
866 // Check if Op is the high part of vector.
867 auto CheckVecHighPart = [](SDValue Op) {
869 if (Op.getOpcode() == ISD::EXTRACT_SUBVECTOR) {
870 SDValue SOp = Op.getOperand(0);
871 EVT SVT = SOp.getValueType();
872 if (!SVT.isVector() || (SVT.getVectorNumElements() % 2 != 0))
873 return SDValue();
874
875 const uint64_t Imm = Op.getConstantOperandVal(1);
876 if (Imm == SVT.getVectorNumElements() / 2)
877 return SOp;
878 return SDValue();
879 }
880 return SDValue();
881 };
882
883 unsigned Opcode;
884 SDValue VFCVTOp;
885 EVT WideOpVT = SVT.getSimpleVT().getDoubleNumVectorElementsVT();
886 SDValue ZeroIdx = DAG.getVectorIdxConstant(0, DL);
887
888 // If the operand of ISD::FP_EXTEND comes from the high part of vector,
889 // generate LoongArchISD::VFCVTH, otherwise LoongArchISD::VFCVTL.
890 if (SDValue V = CheckVecHighPart(Src)) {
891 assert(V.getValueSizeInBits() == WideOpVT.getSizeInBits() &&
892 "Unexpected wide vector");
893 Opcode = LoongArchISD::VFCVTH;
894 VFCVTOp = DAG.getBitcast(WideOpVT, V);
895 } else {
896 Opcode = LoongArchISD::VFCVTL;
897 VFCVTOp = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, WideOpVT,
898 DAG.getUNDEF(WideOpVT), Src, ZeroIdx);
899 }
900
901 // v2f64 = fp_extend v2f32
902 if (V2F32ToV2F64)
903 return DAG.getNode(Opcode, DL, VT, VFCVTOp);
904
905 // v4f64 = fp_extend v4f32
906 if (V4F32ToV4F64) {
907 // XVFCVT instruction operates on each 128-bit segment as a lane, so a
908 // vector_shuffle is required firstly.
909 SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
910 SDValue Res = DAG.getVectorShuffle(WideOpVT, DL, VFCVTOp,
911 DAG.getUNDEF(WideOpVT), Mask);
912 Res = DAG.getNode(Opcode, DL, VT, Res);
913 return Res;
914 }
915
916 return SDValue();
917}
918
919SDValue LoongArchTargetLowering::lowerConstantFP(SDValue Op,
920 SelectionDAG &DAG) const {
921 EVT VT = Op.getValueType();
922 ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Op);
923 const APFloat &FPVal = CFP->getValueAPF();
924 SDLoc DL(CFP);
925
926 assert((VT == MVT::f32 && Subtarget.hasBasicF()) ||
927 (VT == MVT::f64 && Subtarget.hasBasicD()));
928
929 // If value is 0.0 or -0.0, just ignore it.
930 if (FPVal.isZero())
931 return SDValue();
932
933 // If lsx enabled, use cheaper 'vldi' instruction if possible.
934 if (isFPImmVLDILegal(FPVal, VT))
935 return SDValue();
936
937 // Construct as integer, and move to float register.
938 APInt INTVal = FPVal.bitcastToAPInt();
939
940 // If more than MaterializeFPImmInsNum instructions will be used to
941 // generate the INTVal and move it to float register, fallback to
942 // use floating point load from the constant pool.
944 int InsNum = Seq.size() + ((VT == MVT::f64 && !Subtarget.is64Bit()) ? 2 : 1);
945 if (InsNum > MaterializeFPImmInsNum && !FPVal.isOne())
946 return SDValue();
947
948 switch (VT.getSimpleVT().SimpleTy) {
949 default:
950 llvm_unreachable("Unexpected floating point type!");
951 break;
952 case MVT::f32: {
953 SDValue NewVal = DAG.getConstant(INTVal, DL, MVT::i32);
954 if (Subtarget.is64Bit())
955 NewVal = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, NewVal);
956 return DAG.getNode(Subtarget.is64Bit() ? LoongArchISD::MOVGR2FR_W_LA64
957 : LoongArchISD::MOVGR2FR_W,
958 DL, VT, NewVal);
959 }
960 case MVT::f64: {
961 if (Subtarget.is64Bit()) {
962 SDValue NewVal = DAG.getConstant(INTVal, DL, MVT::i64);
963 return DAG.getNode(LoongArchISD::MOVGR2FR_D, DL, VT, NewVal);
964 }
965 SDValue Lo = DAG.getConstant(INTVal.trunc(32), DL, MVT::i32);
966 SDValue Hi = DAG.getConstant(INTVal.lshr(32).trunc(32), DL, MVT::i32);
967 return DAG.getNode(LoongArchISD::MOVGR2FR_D_LO_HI, DL, VT, Lo, Hi);
968 }
969 }
970
971 return SDValue();
972}
973
974// Ensure SETCC result and operand have the same bit width; isel does not
975// support mismatched widths.
976SDValue LoongArchTargetLowering::lowerSETCC(SDValue Op,
977 SelectionDAG &DAG) const {
978 SDLoc DL(Op);
979 EVT ResultVT = Op.getValueType();
980 EVT OperandVT = Op.getOperand(0).getValueType();
981
982 EVT SetCCResultVT =
983 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), OperandVT);
984
985 if (ResultVT == SetCCResultVT)
986 return Op;
987
988 assert(Op.getOperand(0).getValueType() == Op.getOperand(1).getValueType() &&
989 "SETCC operands must have the same type!");
990
991 SDValue SetCCNode =
992 DAG.getNode(ISD::SETCC, DL, SetCCResultVT, Op.getOperand(0),
993 Op.getOperand(1), Op.getOperand(2));
994
995 if (ResultVT.bitsGT(SetCCResultVT))
996 SetCCNode = DAG.getNode(ISD::SIGN_EXTEND, DL, ResultVT, SetCCNode);
997 else if (ResultVT.bitsLT(SetCCResultVT))
998 SetCCNode = DAG.getNode(ISD::TRUNCATE, DL, ResultVT, SetCCNode);
999
1000 return SetCCNode;
1001}
1002
1003// Lower sext_invec using vslti instructions.
1004// For example:
1005// %b = sext <4 x i16> %a to <4 x i32>
1006// can be lowered to:
1007// VSLTI_H vr2, vr1, 0
1008// VILVL.H vr1, vr2, vr1
1009SDValue LoongArchTargetLowering::lowerSIGN_EXTEND_VECTOR_INREG(
1010 SDValue Op, SelectionDAG &DAG) const {
1011 SDLoc DL(Op);
1012 SDValue Src = Op.getOperand(0);
1013 MVT SrcVT = Src.getSimpleValueType();
1014 MVT DstVT = Op.getSimpleValueType();
1015
1016 if (!SrcVT.is128BitVector())
1017 return SDValue();
1018
1019 // lower to VSLTI + VILVL if extend could be done in single step.
1020 if (DstVT.getScalarSizeInBits() / SrcVT.getScalarSizeInBits() == 2) {
1021 SDValue Zero = DAG.getConstant(0, DL, SrcVT);
1022 SDValue Mask = DAG.getNode(ISD::SETCC, DL, SrcVT, Src, Zero,
1023 DAG.getCondCode(ISD::SETLT));
1024 SDValue LoInterleaved =
1025 DAG.getNode(LoongArchISD::VILVL, DL, SrcVT, Mask, Src);
1026
1027 return DAG.getBitcast(DstVT, LoInterleaved);
1028 }
1029
1030 return SDValue();
1031}
1032
1033// ANY_EXTEND can be replaced by ZERO_EXTEND when LASX is enabled.
1034SDValue LoongArchTargetLowering::lowerANY_EXTEND(SDValue Op,
1035 SelectionDAG &DAG) const {
1036 assert(Subtarget.hasExtLASX());
1037 // We don't have corresponding instrunction for ANY_EXTEND, lowering it to
1038 // ZERO_EXTEND won't break its semantics, while avoid scalar extract/insert.
1039 return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(Op), Op.getValueType(),
1040 Op.getOperand(0));
1041}
1042
1043// Lower vecreduce_add using vhaddw instructions.
1044// For Example:
1045// call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %a)
1046// can be lowered to:
1047// VHADDW_D_W vr0, vr0, vr0
1048// VHADDW_Q_D vr0, vr0, vr0
1049// VPICKVE2GR_D a0, vr0, 0
1050// ADDI_W a0, a0, 0
1051SDValue LoongArchTargetLowering::lowerVECREDUCE_ADD(SDValue Op,
1052 SelectionDAG &DAG) const {
1053
1054 SDLoc DL(Op);
1055 MVT OpVT = Op.getSimpleValueType();
1056 SDValue Val = Op.getOperand(0);
1057
1058 unsigned NumEles = Val.getSimpleValueType().getVectorNumElements();
1059 unsigned EleBits = Val.getSimpleValueType().getScalarSizeInBits();
1060 unsigned ResBits = OpVT.getScalarSizeInBits();
1061
1062 unsigned LegalVecSize = 128;
1063 bool isLASX256Vector =
1064 Subtarget.hasExtLASX() && Val.getValueSizeInBits() == 256;
1065
1066 // Ensure operand type legal or enable it legal.
1067 while (!isTypeLegal(Val.getSimpleValueType())) {
1068 Val = DAG.WidenVector(Val, DL);
1069 }
1070
1071 // NumEles is designed for iterations count, v4i32 for LSX
1072 // and v8i32 for LASX should have the same count.
1073 if (isLASX256Vector) {
1074 NumEles /= 2;
1075 LegalVecSize = 256;
1076 }
1077
1078 EleBits *= 2;
1079 for (unsigned i = 1; i < NumEles; i *= 2, EleBits *= 2) {
1080 EleBits = std::min(EleBits, 64u);
1081 MVT IntTy = MVT::getIntegerVT(EleBits);
1082 MVT VecTy = MVT::getVectorVT(IntTy, LegalVecSize / EleBits);
1083 Val = DAG.getNode(LoongArchISD::VHADDW, DL, VecTy, Val, Val);
1084 }
1085
1086 if (isLASX256Vector) {
1087 SDValue Tmp = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, Val,
1088 DAG.getConstant(2, DL, Subtarget.getGRLenVT()));
1089 Val = DAG.getNode(ISD::ADD, DL, MVT::v4i64, Tmp, Val);
1090 }
1091
1092 Val = DAG.getBitcast(MVT::getVectorVT(OpVT, LegalVecSize / ResBits), Val);
1093 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, OpVT, Val,
1094 DAG.getConstant(0, DL, Subtarget.getGRLenVT()));
1095}
1096
1097// Lower vecreduce_and/or/xor/[s/u]max/[s/u]min.
1098// For Example:
1099// call i32 @llvm.vector.reduce.smax.v4i32(<4 x i32> %a)
1100// can be lowered to:
1101// VBSRL_V vr1, vr0, 8
1102// VMAX_W vr0, vr1, vr0
1103// VBSRL_V vr1, vr0, 4
1104// VMAX_W vr0, vr1, vr0
1105// VPICKVE2GR_W a0, vr0, 0
1106// For 256 bit vector, it is illegal and will be spilt into
1107// two 128 bit vector by default then processed by this.
1108SDValue LoongArchTargetLowering::lowerVECREDUCE(SDValue Op,
1109 SelectionDAG &DAG) const {
1110 SDLoc DL(Op);
1111
1112 MVT OpVT = Op.getSimpleValueType();
1113 SDValue Val = Op.getOperand(0);
1114
1115 unsigned NumEles = Val.getSimpleValueType().getVectorNumElements();
1116 unsigned EleBits = Val.getSimpleValueType().getScalarSizeInBits();
1117
1118 // Ensure operand type legal or enable it legal.
1119 while (!isTypeLegal(Val.getSimpleValueType())) {
1120 Val = DAG.WidenVector(Val, DL);
1121 }
1122
1123 unsigned Opcode = ISD::getVecReduceBaseOpcode(Op.getOpcode());
1124 MVT VecTy = Val.getSimpleValueType();
1125 MVT GRLenVT = Subtarget.getGRLenVT();
1126
1127 for (int i = NumEles; i > 1; i /= 2) {
1128 SDValue ShiftAmt = DAG.getConstant(i * EleBits / 16, DL, GRLenVT);
1129 SDValue Tmp = DAG.getNode(LoongArchISD::VBSRL, DL, VecTy, Val, ShiftAmt);
1130 Val = DAG.getNode(Opcode, DL, VecTy, Tmp, Val);
1131 }
1132
1133 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, OpVT, Val,
1134 DAG.getConstant(0, DL, GRLenVT));
1135}
1136
1137SDValue LoongArchTargetLowering::lowerPREFETCH(SDValue Op,
1138 SelectionDAG &DAG) const {
1139 unsigned IsData = Op.getConstantOperandVal(4);
1140
1141 // We don't support non-data prefetch.
1142 // Just preserve the chain.
1143 if (!IsData)
1144 return Op.getOperand(0);
1145
1146 return Op;
1147}
1148
1149SDValue LoongArchTargetLowering::lowerRotate(SDValue Op,
1150 SelectionDAG &DAG) const {
1151 MVT VT = Op.getSimpleValueType();
1152 assert(VT.isVector() && "Unexpected type");
1153
1154 SDLoc DL(Op);
1155 SDValue R = Op.getOperand(0);
1156 SDValue Amt = Op.getOperand(1);
1157 unsigned Opcode = Op.getOpcode();
1158 unsigned EltSizeInBits = VT.getScalarSizeInBits();
1159
1160 auto checkCstSplat = [](SDValue V, APInt &CstSplatValue) {
1161 if (V.getOpcode() != ISD::BUILD_VECTOR)
1162 return false;
1163 if (SDValue SplatValue =
1164 cast<BuildVectorSDNode>(V.getNode())->getSplatValue()) {
1165 if (auto *C = dyn_cast<ConstantSDNode>(SplatValue)) {
1166 CstSplatValue = C->getAPIntValue();
1167 return true;
1168 }
1169 }
1170 return false;
1171 };
1172
1173 // Check for constant splat rotation amount.
1174 APInt CstSplatValue;
1175 bool IsCstSplat = checkCstSplat(Amt, CstSplatValue);
1176 bool isROTL = Opcode == ISD::ROTL;
1177
1178 // Check for splat rotate by zero.
1179 if (IsCstSplat && CstSplatValue.urem(EltSizeInBits) == 0)
1180 return R;
1181
1182 // LoongArch targets always prefer ISD::ROTR.
1183 if (isROTL) {
1184 SDValue Zero = DAG.getConstant(0, DL, VT);
1185 return DAG.getNode(ISD::ROTR, DL, VT, R,
1186 DAG.getNode(ISD::SUB, DL, VT, Zero, Amt));
1187 }
1188
1189 // Rotate by a immediate.
1190 if (IsCstSplat) {
1191 // ISD::ROTR: Attemp to rotate by a positive immediate.
1192 SDValue Bits = DAG.getConstant(EltSizeInBits, DL, VT);
1193 if (SDValue Urem =
1194 DAG.FoldConstantArithmetic(ISD::UREM, DL, VT, {Amt, Bits}))
1195 return DAG.getNode(Opcode, DL, VT, R, Urem);
1196 }
1197
1198 return Op;
1199}
1200
1201// Return true if Val is equal to (setcc LHS, RHS, CC).
1202// Return false if Val is the inverse of (setcc LHS, RHS, CC).
1203// Otherwise, return std::nullopt.
1204static std::optional<bool> matchSetCC(SDValue LHS, SDValue RHS,
1205 ISD::CondCode CC, SDValue Val) {
1206 assert(Val->getOpcode() == ISD::SETCC);
1207 SDValue LHS2 = Val.getOperand(0);
1208 SDValue RHS2 = Val.getOperand(1);
1209 ISD::CondCode CC2 = cast<CondCodeSDNode>(Val.getOperand(2))->get();
1210
1211 if (LHS == LHS2 && RHS == RHS2) {
1212 if (CC == CC2)
1213 return true;
1214 if (CC == ISD::getSetCCInverse(CC2, LHS2.getValueType()))
1215 return false;
1216 } else if (LHS == RHS2 && RHS == LHS2) {
1218 if (CC == CC2)
1219 return true;
1220 if (CC == ISD::getSetCCInverse(CC2, LHS2.getValueType()))
1221 return false;
1222 }
1223
1224 return std::nullopt;
1225}
1226
1228 const LoongArchSubtarget &Subtarget) {
1229 SDValue CondV = N->getOperand(0);
1230 SDValue TrueV = N->getOperand(1);
1231 SDValue FalseV = N->getOperand(2);
1232 MVT VT = N->getSimpleValueType(0);
1233 SDLoc DL(N);
1234
1235 // (select c, -1, y) -> -c | y
1236 if (isAllOnesConstant(TrueV)) {
1237 SDValue Neg = DAG.getNegative(CondV, DL, VT);
1238 return DAG.getNode(ISD::OR, DL, VT, Neg, DAG.getFreeze(FalseV));
1239 }
1240 // (select c, y, -1) -> (c-1) | y
1241 if (isAllOnesConstant(FalseV)) {
1242 SDValue Neg =
1243 DAG.getNode(ISD::ADD, DL, VT, CondV, DAG.getAllOnesConstant(DL, VT));
1244 return DAG.getNode(ISD::OR, DL, VT, Neg, DAG.getFreeze(TrueV));
1245 }
1246
1247 // (select c, 0, y) -> (c-1) & y
1248 if (isNullConstant(TrueV)) {
1249 SDValue Neg =
1250 DAG.getNode(ISD::ADD, DL, VT, CondV, DAG.getAllOnesConstant(DL, VT));
1251 return DAG.getNode(ISD::AND, DL, VT, Neg, DAG.getFreeze(FalseV));
1252 }
1253 // (select c, y, 0) -> -c & y
1254 if (isNullConstant(FalseV)) {
1255 SDValue Neg = DAG.getNegative(CondV, DL, VT);
1256 return DAG.getNode(ISD::AND, DL, VT, Neg, DAG.getFreeze(TrueV));
1257 }
1258
1259 // select c, ~x, x --> xor -c, x
1260 if (isa<ConstantSDNode>(TrueV) && isa<ConstantSDNode>(FalseV)) {
1261 const APInt &TrueVal = TrueV->getAsAPIntVal();
1262 const APInt &FalseVal = FalseV->getAsAPIntVal();
1263 if (~TrueVal == FalseVal) {
1264 SDValue Neg = DAG.getNegative(CondV, DL, VT);
1265 return DAG.getNode(ISD::XOR, DL, VT, Neg, FalseV);
1266 }
1267 }
1268
1269 // Try to fold (select (setcc lhs, rhs, cc), truev, falsev) into bitwise ops
1270 // when both truev and falsev are also setcc.
1271 if (CondV.getOpcode() == ISD::SETCC && TrueV.getOpcode() == ISD::SETCC &&
1272 FalseV.getOpcode() == ISD::SETCC) {
1273 SDValue LHS = CondV.getOperand(0);
1274 SDValue RHS = CondV.getOperand(1);
1275 ISD::CondCode CC = cast<CondCodeSDNode>(CondV.getOperand(2))->get();
1276
1277 // (select x, x, y) -> x | y
1278 // (select !x, x, y) -> x & y
1279 if (std::optional<bool> MatchResult = matchSetCC(LHS, RHS, CC, TrueV)) {
1280 return DAG.getNode(*MatchResult ? ISD::OR : ISD::AND, DL, VT, TrueV,
1281 DAG.getFreeze(FalseV));
1282 }
1283 // (select x, y, x) -> x & y
1284 // (select !x, y, x) -> x | y
1285 if (std::optional<bool> MatchResult = matchSetCC(LHS, RHS, CC, FalseV)) {
1286 return DAG.getNode(*MatchResult ? ISD::AND : ISD::OR, DL, VT,
1287 DAG.getFreeze(TrueV), FalseV);
1288 }
1289 }
1290
1291 return SDValue();
1292}
1293
1294// Transform `binOp (select cond, x, c0), c1` where `c0` and `c1` are constants
1295// into `select cond, binOp(x, c1), binOp(c0, c1)` if profitable.
1296// For now we only consider transformation profitable if `binOp(c0, c1)` ends up
1297// being `0` or `-1`. In such cases we can replace `select` with `and`.
1298// TODO: Should we also do this if `binOp(c0, c1)` is cheaper to materialize
1299// than `c0`?
1300static SDValue
1302 const LoongArchSubtarget &Subtarget) {
1303 unsigned SelOpNo = 0;
1304 SDValue Sel = BO->getOperand(0);
1305 if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse()) {
1306 SelOpNo = 1;
1307 Sel = BO->getOperand(1);
1308 }
1309
1310 if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse())
1311 return SDValue();
1312
1313 unsigned ConstSelOpNo = 1;
1314 unsigned OtherSelOpNo = 2;
1315 if (!isa<ConstantSDNode>(Sel->getOperand(ConstSelOpNo))) {
1316 ConstSelOpNo = 2;
1317 OtherSelOpNo = 1;
1318 }
1319 SDValue ConstSelOp = Sel->getOperand(ConstSelOpNo);
1320 ConstantSDNode *ConstSelOpNode = dyn_cast<ConstantSDNode>(ConstSelOp);
1321 if (!ConstSelOpNode || ConstSelOpNode->isOpaque())
1322 return SDValue();
1323
1324 SDValue ConstBinOp = BO->getOperand(SelOpNo ^ 1);
1325 ConstantSDNode *ConstBinOpNode = dyn_cast<ConstantSDNode>(ConstBinOp);
1326 if (!ConstBinOpNode || ConstBinOpNode->isOpaque())
1327 return SDValue();
1328
1329 SDLoc DL(Sel);
1330 EVT VT = BO->getValueType(0);
1331
1332 SDValue NewConstOps[2] = {ConstSelOp, ConstBinOp};
1333 if (SelOpNo == 1)
1334 std::swap(NewConstOps[0], NewConstOps[1]);
1335
1336 SDValue NewConstOp =
1337 DAG.FoldConstantArithmetic(BO->getOpcode(), DL, VT, NewConstOps);
1338 if (!NewConstOp)
1339 return SDValue();
1340
1341 const APInt &NewConstAPInt = NewConstOp->getAsAPIntVal();
1342 if (!NewConstAPInt.isZero() && !NewConstAPInt.isAllOnes())
1343 return SDValue();
1344
1345 SDValue OtherSelOp = Sel->getOperand(OtherSelOpNo);
1346 SDValue NewNonConstOps[2] = {OtherSelOp, ConstBinOp};
1347 if (SelOpNo == 1)
1348 std::swap(NewNonConstOps[0], NewNonConstOps[1]);
1349 SDValue NewNonConstOp = DAG.getNode(BO->getOpcode(), DL, VT, NewNonConstOps);
1350
1351 SDValue NewT = (ConstSelOpNo == 1) ? NewConstOp : NewNonConstOp;
1352 SDValue NewF = (ConstSelOpNo == 1) ? NewNonConstOp : NewConstOp;
1353 return DAG.getSelect(DL, VT, Sel.getOperand(0), NewT, NewF);
1354}
1355
1356// Changes the condition code and swaps operands if necessary, so the SetCC
1357// operation matches one of the comparisons supported directly by branches
1358// in the LoongArch ISA. May adjust compares to favor compare with 0 over
1359// compare with 1/-1.
1361 ISD::CondCode &CC, SelectionDAG &DAG) {
1362 // If this is a single bit test that can't be handled by ANDI, shift the
1363 // bit to be tested to the MSB and perform a signed compare with 0.
1364 if (isIntEqualitySetCC(CC) && isNullConstant(RHS) &&
1365 LHS.getOpcode() == ISD::AND && LHS.hasOneUse() &&
1366 isa<ConstantSDNode>(LHS.getOperand(1))) {
1367 uint64_t Mask = LHS.getConstantOperandVal(1);
1368 if ((isPowerOf2_64(Mask) || isMask_64(Mask)) && !isInt<12>(Mask)) {
1369 unsigned ShAmt = 0;
1370 if (isPowerOf2_64(Mask)) {
1371 CC = CC == ISD::SETEQ ? ISD::SETGE : ISD::SETLT;
1372 ShAmt = LHS.getValueSizeInBits() - 1 - Log2_64(Mask);
1373 } else {
1374 ShAmt = LHS.getValueSizeInBits() - llvm::bit_width(Mask);
1375 }
1376
1377 LHS = LHS.getOperand(0);
1378 if (ShAmt != 0)
1379 LHS = DAG.getNode(ISD::SHL, DL, LHS.getValueType(), LHS,
1380 DAG.getConstant(ShAmt, DL, LHS.getValueType()));
1381 return;
1382 }
1383 }
1384
1385 if (auto *RHSC = dyn_cast<ConstantSDNode>(RHS)) {
1386 int64_t C = RHSC->getSExtValue();
1387 switch (CC) {
1388 default:
1389 break;
1390 case ISD::SETGT:
1391 // Convert X > -1 to X >= 0.
1392 if (C == -1) {
1393 RHS = DAG.getConstant(0, DL, RHS.getValueType());
1394 CC = ISD::SETGE;
1395 return;
1396 }
1397 break;
1398 case ISD::SETLT:
1399 // Convert X < 1 to 0 >= X.
1400 if (C == 1) {
1401 RHS = LHS;
1402 LHS = DAG.getConstant(0, DL, RHS.getValueType());
1403 CC = ISD::SETGE;
1404 return;
1405 }
1406 break;
1407 }
1408 }
1409
1410 switch (CC) {
1411 default:
1412 break;
1413 case ISD::SETGT:
1414 case ISD::SETLE:
1415 case ISD::SETUGT:
1416 case ISD::SETULE:
1418 std::swap(LHS, RHS);
1419 break;
1420 }
1421}
1422
1423SDValue LoongArchTargetLowering::lowerSELECT(SDValue Op,
1424 SelectionDAG &DAG) const {
1425 SDValue CondV = Op.getOperand(0);
1426 SDValue TrueV = Op.getOperand(1);
1427 SDValue FalseV = Op.getOperand(2);
1428 SDLoc DL(Op);
1429 MVT VT = Op.getSimpleValueType();
1430 MVT GRLenVT = Subtarget.getGRLenVT();
1431
1432 if (SDValue V = combineSelectToBinOp(Op.getNode(), DAG, Subtarget))
1433 return V;
1434
1435 if (Op.hasOneUse()) {
1436 unsigned UseOpc = Op->user_begin()->getOpcode();
1437 if (isBinOp(UseOpc) && DAG.isSafeToSpeculativelyExecute(UseOpc)) {
1438 SDNode *BinOp = *Op->user_begin();
1439 if (SDValue NewSel = foldBinOpIntoSelectIfProfitable(*Op->user_begin(),
1440 DAG, Subtarget)) {
1441 DAG.ReplaceAllUsesWith(BinOp, &NewSel);
1442 // Opcode check is necessary because foldBinOpIntoSelectIfProfitable
1443 // may return a constant node and cause crash in lowerSELECT.
1444 if (NewSel.getOpcode() == ISD::SELECT)
1445 return lowerSELECT(NewSel, DAG);
1446 return NewSel;
1447 }
1448 }
1449 }
1450
1451 // If the condition is not an integer SETCC which operates on GRLenVT, we need
1452 // to emit a LoongArchISD::SELECT_CC comparing the condition to zero. i.e.:
1453 // (select condv, truev, falsev)
1454 // -> (loongarchisd::select_cc condv, zero, setne, truev, falsev)
1455 if (CondV.getOpcode() != ISD::SETCC ||
1456 CondV.getOperand(0).getSimpleValueType() != GRLenVT) {
1457 SDValue Zero = DAG.getConstant(0, DL, GRLenVT);
1458 SDValue SetNE = DAG.getCondCode(ISD::SETNE);
1459
1460 SDValue Ops[] = {CondV, Zero, SetNE, TrueV, FalseV};
1461
1462 return DAG.getNode(LoongArchISD::SELECT_CC, DL, VT, Ops);
1463 }
1464
1465 // If the CondV is the output of a SETCC node which operates on GRLenVT
1466 // inputs, then merge the SETCC node into the lowered LoongArchISD::SELECT_CC
1467 // to take advantage of the integer compare+branch instructions. i.e.: (select
1468 // (setcc lhs, rhs, cc), truev, falsev)
1469 // -> (loongarchisd::select_cc lhs, rhs, cc, truev, falsev)
1470 SDValue LHS = CondV.getOperand(0);
1471 SDValue RHS = CondV.getOperand(1);
1472 ISD::CondCode CCVal = cast<CondCodeSDNode>(CondV.getOperand(2))->get();
1473
1474 // Special case for a select of 2 constants that have a difference of 1.
1475 // Normally this is done by DAGCombine, but if the select is introduced by
1476 // type legalization or op legalization, we miss it. Restricting to SETLT
1477 // case for now because that is what signed saturating add/sub need.
1478 // FIXME: We don't need the condition to be SETLT or even a SETCC,
1479 // but we would probably want to swap the true/false values if the condition
1480 // is SETGE/SETLE to avoid an XORI.
1481 if (isa<ConstantSDNode>(TrueV) && isa<ConstantSDNode>(FalseV) &&
1482 CCVal == ISD::SETLT) {
1483 const APInt &TrueVal = TrueV->getAsAPIntVal();
1484 const APInt &FalseVal = FalseV->getAsAPIntVal();
1485 if (TrueVal - 1 == FalseVal)
1486 return DAG.getNode(ISD::ADD, DL, VT, CondV, FalseV);
1487 if (TrueVal + 1 == FalseVal)
1488 return DAG.getNode(ISD::SUB, DL, VT, FalseV, CondV);
1489 }
1490
1491 translateSetCCForBranch(DL, LHS, RHS, CCVal, DAG);
1492 // 1 < x ? x : 1 -> 0 < x ? x : 1
1493 if (isOneConstant(LHS) && (CCVal == ISD::SETLT || CCVal == ISD::SETULT) &&
1494 RHS == TrueV && LHS == FalseV) {
1495 LHS = DAG.getConstant(0, DL, VT);
1496 // 0 <u x is the same as x != 0.
1497 if (CCVal == ISD::SETULT) {
1498 std::swap(LHS, RHS);
1499 CCVal = ISD::SETNE;
1500 }
1501 }
1502
1503 // x <s -1 ? x : -1 -> x <s 0 ? x : -1
1504 if (isAllOnesConstant(RHS) && CCVal == ISD::SETLT && LHS == TrueV &&
1505 RHS == FalseV) {
1506 RHS = DAG.getConstant(0, DL, VT);
1507 }
1508
1509 SDValue TargetCC = DAG.getCondCode(CCVal);
1510
1511 if (isa<ConstantSDNode>(TrueV) && !isa<ConstantSDNode>(FalseV)) {
1512 // (select (setcc lhs, rhs, CC), constant, falsev)
1513 // -> (select (setcc lhs, rhs, InverseCC), falsev, constant)
1514 std::swap(TrueV, FalseV);
1515 TargetCC = DAG.getCondCode(ISD::getSetCCInverse(CCVal, LHS.getValueType()));
1516 }
1517
1518 SDValue Ops[] = {LHS, RHS, TargetCC, TrueV, FalseV};
1519 return DAG.getNode(LoongArchISD::SELECT_CC, DL, VT, Ops);
1520}
1521
1522SDValue LoongArchTargetLowering::lowerBRCOND(SDValue Op,
1523 SelectionDAG &DAG) const {
1524 SDValue CondV = Op.getOperand(1);
1525 SDLoc DL(Op);
1526 MVT GRLenVT = Subtarget.getGRLenVT();
1527
1528 if (CondV.getOpcode() == ISD::SETCC) {
1529 if (CondV.getOperand(0).getValueType() == GRLenVT) {
1530 SDValue LHS = CondV.getOperand(0);
1531 SDValue RHS = CondV.getOperand(1);
1532 ISD::CondCode CCVal = cast<CondCodeSDNode>(CondV.getOperand(2))->get();
1533
1534 translateSetCCForBranch(DL, LHS, RHS, CCVal, DAG);
1535
1536 SDValue TargetCC = DAG.getCondCode(CCVal);
1537 return DAG.getNode(LoongArchISD::BR_CC, DL, Op.getValueType(),
1538 Op.getOperand(0), LHS, RHS, TargetCC,
1539 Op.getOperand(2));
1540 } else if (CondV.getOperand(0).getValueType().isFloatingPoint()) {
1541 return DAG.getNode(LoongArchISD::BRCOND, DL, Op.getValueType(),
1542 Op.getOperand(0), CondV, Op.getOperand(2));
1543 }
1544 }
1545
1546 return DAG.getNode(LoongArchISD::BR_CC, DL, Op.getValueType(),
1547 Op.getOperand(0), CondV, DAG.getConstant(0, DL, GRLenVT),
1548 DAG.getCondCode(ISD::SETNE), Op.getOperand(2));
1549}
1550
1551SDValue
1552LoongArchTargetLowering::lowerSCALAR_TO_VECTOR(SDValue Op,
1553 SelectionDAG &DAG) const {
1554 SDLoc DL(Op);
1555 MVT OpVT = Op.getSimpleValueType();
1556
1557 SDValue Vector = DAG.getUNDEF(OpVT);
1558 SDValue Val = Op.getOperand(0);
1559 SDValue Idx = DAG.getConstant(0, DL, Subtarget.getGRLenVT());
1560
1561 return DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, OpVT, Vector, Val, Idx);
1562}
1563
1564SDValue LoongArchTargetLowering::lowerBITREVERSE(SDValue Op,
1565 SelectionDAG &DAG) const {
1566 EVT ResTy = Op->getValueType(0);
1567 SDValue Src = Op->getOperand(0);
1568 SDLoc DL(Op);
1569
1570 // LoongArchISD::BITREV_8B is not supported on LA32.
1571 if (!Subtarget.is64Bit() && (ResTy == MVT::v16i8 || ResTy == MVT::v32i8))
1572 return SDValue();
1573
1574 EVT NewVT = ResTy.is128BitVector() ? MVT::v2i64 : MVT::v4i64;
1575 unsigned int OrigEltNum = ResTy.getVectorNumElements();
1576 unsigned int NewEltNum = NewVT.getVectorNumElements();
1577
1578 SDValue NewSrc = DAG.getNode(ISD::BITCAST, DL, NewVT, Src);
1579
1581 for (unsigned int i = 0; i < NewEltNum; i++) {
1582 SDValue Op = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i64, NewSrc,
1583 DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
1584 unsigned RevOp = (ResTy == MVT::v16i8 || ResTy == MVT::v32i8)
1585 ? (unsigned)LoongArchISD::BITREV_8B
1586 : (unsigned)ISD::BITREVERSE;
1587 Ops.push_back(DAG.getNode(RevOp, DL, MVT::i64, Op));
1588 }
1589 SDValue Res =
1590 DAG.getNode(ISD::BITCAST, DL, ResTy, DAG.getBuildVector(NewVT, DL, Ops));
1591
1592 switch (ResTy.getSimpleVT().SimpleTy) {
1593 default:
1594 return SDValue();
1595 case MVT::v16i8:
1596 case MVT::v32i8:
1597 return Res;
1598 case MVT::v8i16:
1599 case MVT::v16i16:
1600 case MVT::v4i32:
1601 case MVT::v8i32: {
1603 for (unsigned int i = 0; i < NewEltNum; i++)
1604 for (int j = OrigEltNum / NewEltNum - 1; j >= 0; j--)
1605 Mask.push_back(j + (OrigEltNum / NewEltNum) * i);
1606 return DAG.getVectorShuffle(ResTy, DL, Res, DAG.getUNDEF(ResTy), Mask);
1607 }
1608 }
1609}
1610
1611// Widen element type to get a new mask value (if possible).
1612// For example:
1613// shufflevector <4 x i32> %a, <4 x i32> %b,
1614// <4 x i32> <i32 6, i32 7, i32 2, i32 3>
1615// is equivalent to:
1616// shufflevector <2 x i64> %a, <2 x i64> %b, <2 x i32> <i32 3, i32 1>
1617// can be lowered to:
1618// VPACKOD_D vr0, vr0, vr1
1620 SDValue V1, SDValue V2, SelectionDAG &DAG) {
1621 unsigned EltBits = VT.getScalarSizeInBits();
1622
1623 if (EltBits > 32 || EltBits == 1)
1624 return SDValue();
1625
1626 SmallVector<int, 8> NewMask;
1627 if (widenShuffleMaskElts(Mask, NewMask)) {
1628 MVT NewEltVT = VT.isFloatingPoint() ? MVT::getFloatingPointVT(EltBits * 2)
1629 : MVT::getIntegerVT(EltBits * 2);
1630 MVT NewVT = MVT::getVectorVT(NewEltVT, VT.getVectorNumElements() / 2);
1631 if (DAG.getTargetLoweringInfo().isTypeLegal(NewVT)) {
1632 SDValue NewV1 = DAG.getBitcast(NewVT, V1);
1633 SDValue NewV2 = DAG.getBitcast(NewVT, V2);
1634 return DAG.getBitcast(
1635 VT, DAG.getVectorShuffle(NewVT, DL, NewV1, NewV2, NewMask));
1636 }
1637 }
1638
1639 return SDValue();
1640}
1641
1642/// Attempts to match a shuffle mask against the VBSLL, VBSRL, VSLLI and VSRLI
1643/// instruction.
1644// The funciton matches elements from one of the input vector shuffled to the
1645// left or right with zeroable elements 'shifted in'. It handles both the
1646// strictly bit-wise element shifts and the byte shfit across an entire 128-bit
1647// lane.
1648// Mostly copied from X86.
1649static int matchShuffleAsShift(MVT &ShiftVT, unsigned &Opcode,
1650 unsigned ScalarSizeInBits, ArrayRef<int> Mask,
1651 int MaskOffset, const APInt &Zeroable) {
1652 int Size = Mask.size();
1653 unsigned SizeInBits = Size * ScalarSizeInBits;
1654
1655 auto CheckZeros = [&](int Shift, int Scale, bool Left) {
1656 for (int i = 0; i < Size; i += Scale)
1657 for (int j = 0; j < Shift; ++j)
1658 if (!Zeroable[i + j + (Left ? 0 : (Scale - Shift))])
1659 return false;
1660
1661 return true;
1662 };
1663
1664 auto isSequentialOrUndefInRange = [&](unsigned Pos, unsigned Size, int Low,
1665 int Step = 1) {
1666 for (unsigned i = Pos, e = Pos + Size; i != e; ++i, Low += Step)
1667 if (!(Mask[i] == -1 || Mask[i] == Low))
1668 return false;
1669 return true;
1670 };
1671
1672 auto MatchShift = [&](int Shift, int Scale, bool Left) {
1673 for (int i = 0; i != Size; i += Scale) {
1674 unsigned Pos = Left ? i + Shift : i;
1675 unsigned Low = Left ? i : i + Shift;
1676 unsigned Len = Scale - Shift;
1677 if (!isSequentialOrUndefInRange(Pos, Len, Low + MaskOffset))
1678 return -1;
1679 }
1680
1681 int ShiftEltBits = ScalarSizeInBits * Scale;
1682 bool ByteShift = ShiftEltBits > 64;
1683 Opcode = Left ? (ByteShift ? LoongArchISD::VBSLL : LoongArchISD::VSLLI)
1684 : (ByteShift ? LoongArchISD::VBSRL : LoongArchISD::VSRLI);
1685 int ShiftAmt = Shift * ScalarSizeInBits / (ByteShift ? 8 : 1);
1686
1687 // Normalize the scale for byte shifts to still produce an i64 element
1688 // type.
1689 Scale = ByteShift ? Scale / 2 : Scale;
1690
1691 // We need to round trip through the appropriate type for the shift.
1692 MVT ShiftSVT = MVT::getIntegerVT(ScalarSizeInBits * Scale);
1693 ShiftVT = ByteShift ? MVT::getVectorVT(MVT::i8, SizeInBits / 8)
1694 : MVT::getVectorVT(ShiftSVT, Size / Scale);
1695 return (int)ShiftAmt;
1696 };
1697
1698 unsigned MaxWidth = 128;
1699 for (int Scale = 2; Scale * ScalarSizeInBits <= MaxWidth; Scale *= 2)
1700 for (int Shift = 1; Shift != Scale; ++Shift)
1701 for (bool Left : {true, false})
1702 if (CheckZeros(Shift, Scale, Left)) {
1703 int ShiftAmt = MatchShift(Shift, Scale, Left);
1704 if (0 < ShiftAmt)
1705 return ShiftAmt;
1706 }
1707
1708 // no match
1709 return -1;
1710}
1711
1712/// Lower VECTOR_SHUFFLE as shift (if possible).
1713///
1714/// For example:
1715/// %2 = shufflevector <4 x i32> %0, <4 x i32> zeroinitializer,
1716/// <4 x i32> <i32 4, i32 0, i32 1, i32 2>
1717/// is lowered to:
1718/// (VBSLL_V $v0, $v0, 4)
1719///
1720/// %2 = shufflevector <4 x i32> %0, <4 x i32> zeroinitializer,
1721/// <4 x i32> <i32 4, i32 0, i32 4, i32 2>
1722/// is lowered to:
1723/// (VSLLI_D $v0, $v0, 32)
1725 MVT VT, SDValue V1, SDValue V2,
1726 SelectionDAG &DAG,
1727 const LoongArchSubtarget &Subtarget,
1728 const APInt &Zeroable) {
1729 int Size = Mask.size();
1730 assert(Size == (int)VT.getVectorNumElements() && "Unexpected mask size");
1731
1732 MVT ShiftVT;
1733 SDValue V = V1;
1734 unsigned Opcode;
1735
1736 // Try to match shuffle against V1 shift.
1737 int ShiftAmt = matchShuffleAsShift(ShiftVT, Opcode, VT.getScalarSizeInBits(),
1738 Mask, 0, Zeroable);
1739
1740 // If V1 failed, try to match shuffle against V2 shift.
1741 if (ShiftAmt < 0) {
1742 ShiftAmt = matchShuffleAsShift(ShiftVT, Opcode, VT.getScalarSizeInBits(),
1743 Mask, Size, Zeroable);
1744 V = V2;
1745 }
1746
1747 if (ShiftAmt < 0)
1748 return SDValue();
1749
1750 assert(DAG.getTargetLoweringInfo().isTypeLegal(ShiftVT) &&
1751 "Illegal integer vector type");
1752 V = DAG.getBitcast(ShiftVT, V);
1753 V = DAG.getNode(Opcode, DL, ShiftVT, V,
1754 DAG.getConstant(ShiftAmt, DL, Subtarget.getGRLenVT()));
1755 return DAG.getBitcast(VT, V);
1756}
1757
1758/// Determine whether a range fits a regular pattern of values.
1759/// This function accounts for the possibility of jumping over the End iterator.
1760template <typename ValType>
1761static bool
1763 unsigned CheckStride,
1765 ValType ExpectedIndex, unsigned ExpectedIndexStride) {
1766 auto &I = Begin;
1767
1768 while (I != End) {
1769 if (*I != -1 && *I != ExpectedIndex)
1770 return false;
1771 ExpectedIndex += ExpectedIndexStride;
1772
1773 // Incrementing past End is undefined behaviour so we must increment one
1774 // step at a time and check for End at each step.
1775 for (unsigned n = 0; n < CheckStride && I != End; ++n, ++I)
1776 ; // Empty loop body.
1777 }
1778 return true;
1779}
1780
1781/// Compute whether each element of a shuffle is zeroable.
1782///
1783/// A "zeroable" vector shuffle element is one which can be lowered to zero.
1785 SDValue V2, APInt &KnownUndef,
1786 APInt &KnownZero) {
1787 int Size = Mask.size();
1788 KnownUndef = KnownZero = APInt::getZero(Size);
1789
1791 V2 = peekThroughBitcasts(V2);
1792
1793 bool V1IsZero = ISD::isBuildVectorAllZeros(V1.getNode());
1794 bool V2IsZero = ISD::isBuildVectorAllZeros(V2.getNode());
1795
1796 int VectorSizeInBits = V1.getValueSizeInBits();
1797 int ScalarSizeInBits = VectorSizeInBits / Size;
1798 assert(!(VectorSizeInBits % ScalarSizeInBits) && "Illegal shuffle mask size");
1799 (void)ScalarSizeInBits;
1800
1801 for (int i = 0; i < Size; ++i) {
1802 int M = Mask[i];
1803 if (M < 0) {
1804 KnownUndef.setBit(i);
1805 continue;
1806 }
1807 if ((M >= 0 && M < Size && V1IsZero) || (M >= Size && V2IsZero)) {
1808 KnownZero.setBit(i);
1809 continue;
1810 }
1811 }
1812}
1813
1814/// Test whether a shuffle mask is equivalent within each sub-lane.
1815///
1816/// The specific repeated shuffle mask is populated in \p RepeatedMask, as it is
1817/// non-trivial to compute in the face of undef lanes. The representation is
1818/// suitable for use with existing 128-bit shuffles as entries from the second
1819/// vector have been remapped to [LaneSize, 2*LaneSize).
1820static bool isRepeatedShuffleMask(unsigned LaneSizeInBits, MVT VT,
1821 ArrayRef<int> Mask,
1822 SmallVectorImpl<int> &RepeatedMask) {
1823 auto LaneSize = LaneSizeInBits / VT.getScalarSizeInBits();
1824 RepeatedMask.assign(LaneSize, -1);
1825 int Size = Mask.size();
1826 for (int i = 0; i < Size; ++i) {
1827 assert(Mask[i] == -1 || Mask[i] >= 0);
1828 if (Mask[i] < 0)
1829 continue;
1830 if ((Mask[i] % Size) / LaneSize != i / LaneSize)
1831 // This entry crosses lanes, so there is no way to model this shuffle.
1832 return false;
1833
1834 // Ok, handle the in-lane shuffles by detecting if and when they repeat.
1835 // Adjust second vector indices to start at LaneSize instead of Size.
1836 int LocalM =
1837 Mask[i] < Size ? Mask[i] % LaneSize : Mask[i] % LaneSize + LaneSize;
1838 if (RepeatedMask[i % LaneSize] < 0)
1839 // This is the first non-undef entry in this slot of a 128-bit lane.
1840 RepeatedMask[i % LaneSize] = LocalM;
1841 else if (RepeatedMask[i % LaneSize] != LocalM)
1842 // Found a mismatch with the repeated mask.
1843 return false;
1844 }
1845 return true;
1846}
1847
1848/// Attempts to match vector shuffle as byte rotation.
1850 ArrayRef<int> Mask) {
1851
1852 SDValue Lo, Hi;
1853 SmallVector<int, 16> RepeatedMask;
1854
1855 if (!isRepeatedShuffleMask(128, VT, Mask, RepeatedMask))
1856 return -1;
1857
1858 int NumElts = RepeatedMask.size();
1859 int Rotation = 0;
1860 int Scale = 16 / NumElts;
1861
1862 for (int i = 0; i < NumElts; ++i) {
1863 int M = RepeatedMask[i];
1864 assert((M == -1 || (0 <= M && M < (2 * NumElts))) &&
1865 "Unexpected mask index.");
1866 if (M < 0)
1867 continue;
1868
1869 // Determine where a rotated vector would have started.
1870 int StartIdx = i - (M % NumElts);
1871 if (StartIdx == 0)
1872 return -1;
1873
1874 // If we found the tail of a vector the rotation must be the missing
1875 // front. If we found the head of a vector, it must be how much of the
1876 // head.
1877 int CandidateRotation = StartIdx < 0 ? -StartIdx : NumElts - StartIdx;
1878
1879 if (Rotation == 0)
1880 Rotation = CandidateRotation;
1881 else if (Rotation != CandidateRotation)
1882 return -1;
1883
1884 // Compute which value this mask is pointing at.
1885 SDValue MaskV = M < NumElts ? V1 : V2;
1886
1887 // Compute which of the two target values this index should be assigned
1888 // to. This reflects whether the high elements are remaining or the low
1889 // elements are remaining.
1890 SDValue &TargetV = StartIdx < 0 ? Hi : Lo;
1891
1892 // Either set up this value if we've not encountered it before, or check
1893 // that it remains consistent.
1894 if (!TargetV)
1895 TargetV = MaskV;
1896 else if (TargetV != MaskV)
1897 return -1;
1898 }
1899
1900 // Check that we successfully analyzed the mask, and normalize the results.
1901 assert(Rotation != 0 && "Failed to locate a viable rotation!");
1902 assert((Lo || Hi) && "Failed to find a rotated input vector!");
1903 if (!Lo)
1904 Lo = Hi;
1905 else if (!Hi)
1906 Hi = Lo;
1907
1908 V1 = Lo;
1909 V2 = Hi;
1910
1911 return Rotation * Scale;
1912}
1913
1914/// Lower VECTOR_SHUFFLE as byte rotate (if possible).
1915///
1916/// For example:
1917/// %shuffle = shufflevector <2 x i64> %a, <2 x i64> %b,
1918/// <2 x i32> <i32 3, i32 0>
1919/// is lowered to:
1920/// (VBSRL_V $v1, $v1, 8)
1921/// (VBSLL_V $v0, $v0, 8)
1922/// (VOR_V $v0, $V0, $v1)
1923static SDValue
1925 SDValue V1, SDValue V2, SelectionDAG &DAG,
1926 const LoongArchSubtarget &Subtarget) {
1927
1928 SDValue Lo = V1, Hi = V2;
1929 int ByteRotation = matchShuffleAsByteRotate(VT, Lo, Hi, Mask);
1930 if (ByteRotation <= 0)
1931 return SDValue();
1932
1933 MVT ByteVT = MVT::getVectorVT(MVT::i8, VT.getSizeInBits() / 8);
1934 Lo = DAG.getBitcast(ByteVT, Lo);
1935 Hi = DAG.getBitcast(ByteVT, Hi);
1936
1937 int LoByteShift = 16 - ByteRotation;
1938 int HiByteShift = ByteRotation;
1939 MVT GRLenVT = Subtarget.getGRLenVT();
1940
1941 SDValue LoShift = DAG.getNode(LoongArchISD::VBSLL, DL, ByteVT, Lo,
1942 DAG.getConstant(LoByteShift, DL, GRLenVT));
1943 SDValue HiShift = DAG.getNode(LoongArchISD::VBSRL, DL, ByteVT, Hi,
1944 DAG.getConstant(HiByteShift, DL, GRLenVT));
1945 return DAG.getBitcast(VT, DAG.getNode(ISD::OR, DL, ByteVT, LoShift, HiShift));
1946}
1947
1948/// Lower VECTOR_SHUFFLE as ZERO_EXTEND Or ANY_EXTEND (if possible).
1949///
1950/// For example:
1951/// %2 = shufflevector <4 x i32> %0, <4 x i32> zeroinitializer,
1952/// <4 x i32> <i32 0, i32 4, i32 1, i32 4>
1953/// %3 = bitcast <4 x i32> %2 to <2 x i64>
1954/// is lowered to:
1955/// (VREPLI $v1, 0)
1956/// (VILVL $v0, $v1, $v0)
1958 ArrayRef<int> Mask, MVT VT,
1959 SDValue V1, SDValue V2,
1960 SelectionDAG &DAG,
1961 const APInt &Zeroable) {
1962 int Bits = VT.getSizeInBits();
1963 int EltBits = VT.getScalarSizeInBits();
1964 int NumElements = VT.getVectorNumElements();
1965
1966 if (Zeroable.isAllOnes())
1967 return DAG.getConstant(0, DL, VT);
1968
1969 // Define a helper function to check a particular ext-scale and lower to it if
1970 // valid.
1971 auto Lower = [&](int Scale) -> SDValue {
1972 SDValue InputV;
1973 bool AnyExt = true;
1974 int Offset = 0;
1975 for (int i = 0; i < NumElements; i++) {
1976 int M = Mask[i];
1977 if (M < 0)
1978 continue;
1979 if (i % Scale != 0) {
1980 // Each of the extended elements need to be zeroable.
1981 if (!Zeroable[i])
1982 return SDValue();
1983
1984 AnyExt = false;
1985 continue;
1986 }
1987
1988 // Each of the base elements needs to be consecutive indices into the
1989 // same input vector.
1990 SDValue V = M < NumElements ? V1 : V2;
1991 M = M % NumElements;
1992 if (!InputV) {
1993 InputV = V;
1994 Offset = M - (i / Scale);
1995
1996 // These offset can't be handled
1997 if (Offset % (NumElements / Scale))
1998 return SDValue();
1999 } else if (InputV != V)
2000 return SDValue();
2001
2002 if (M != (Offset + (i / Scale)))
2003 return SDValue(); // Non-consecutive strided elements.
2004 }
2005
2006 // If we fail to find an input, we have a zero-shuffle which should always
2007 // have already been handled.
2008 if (!InputV)
2009 return SDValue();
2010
2011 do {
2012 unsigned VilVLoHi = LoongArchISD::VILVL;
2013 if (Offset >= (NumElements / 2)) {
2014 VilVLoHi = LoongArchISD::VILVH;
2015 Offset -= (NumElements / 2);
2016 }
2017
2018 MVT InputVT = MVT::getVectorVT(MVT::getIntegerVT(EltBits), NumElements);
2019 SDValue Ext =
2020 AnyExt ? DAG.getFreeze(InputV) : DAG.getConstant(0, DL, InputVT);
2021 InputV = DAG.getBitcast(InputVT, InputV);
2022 InputV = DAG.getNode(VilVLoHi, DL, InputVT, Ext, InputV);
2023 Scale /= 2;
2024 EltBits *= 2;
2025 NumElements /= 2;
2026 } while (Scale > 1);
2027 return DAG.getBitcast(VT, InputV);
2028 };
2029
2030 // Each iteration, try extending the elements half as much, but into twice as
2031 // many elements.
2032 for (int NumExtElements = Bits / 64; NumExtElements < NumElements;
2033 NumExtElements *= 2) {
2034 if (SDValue V = Lower(NumElements / NumExtElements))
2035 return V;
2036 }
2037 return SDValue();
2038}
2039
2040/// Lower VECTOR_SHUFFLE into VREPLVEI (if possible).
2041///
2042/// VREPLVEI performs vector broadcast based on an element specified by an
2043/// integer immediate, with its mask being similar to:
2044/// <x, x, x, ...>
2045/// where x is any valid index.
2046///
2047/// When undef's appear in the mask they are treated as if they were whatever
2048/// value is necessary in order to fit the above form.
2049static SDValue
2051 SDValue V1, SelectionDAG &DAG,
2052 const LoongArchSubtarget &Subtarget) {
2053 int SplatIndex = -1;
2054 for (const auto &M : Mask) {
2055 if (M != -1) {
2056 SplatIndex = M;
2057 break;
2058 }
2059 }
2060
2061 if (SplatIndex == -1)
2062 return DAG.getUNDEF(VT);
2063
2064 assert(SplatIndex < (int)Mask.size() && "Out of bounds mask index");
2065 if (fitsRegularPattern<int>(Mask.begin(), 1, Mask.end(), SplatIndex, 0)) {
2066 return DAG.getNode(LoongArchISD::VREPLVEI, DL, VT, V1,
2067 DAG.getConstant(SplatIndex, DL, Subtarget.getGRLenVT()));
2068 }
2069
2070 return SDValue();
2071}
2072
2073/// Lower VECTOR_SHUFFLE into VSHUF4I (if possible).
2074///
2075/// VSHUF4I splits the vector into blocks of four elements, then shuffles these
2076/// elements according to a <4 x i2> constant (encoded as an integer immediate).
2077///
2078/// It is therefore possible to lower into VSHUF4I when the mask takes the form:
2079/// <a, b, c, d, a+4, b+4, c+4, d+4, a+8, b+8, c+8, d+8, ...>
2080/// When undef's appear they are treated as if they were whatever value is
2081/// necessary in order to fit the above forms.
2082///
2083/// For example:
2084/// %2 = shufflevector <8 x i16> %0, <8 x i16> undef,
2085/// <8 x i32> <i32 3, i32 2, i32 1, i32 0,
2086/// i32 7, i32 6, i32 5, i32 4>
2087/// is lowered to:
2088/// (VSHUF4I_H $v0, $v1, 27)
2089/// where the 27 comes from:
2090/// 3 + (2 << 2) + (1 << 4) + (0 << 6)
2091static SDValue
2093 SDValue V1, SDValue V2, SelectionDAG &DAG,
2094 const LoongArchSubtarget &Subtarget) {
2095
2096 unsigned SubVecSize = 4;
2097 if (VT == MVT::v2f64 || VT == MVT::v2i64)
2098 SubVecSize = 2;
2099
2100 int SubMask[4] = {-1, -1, -1, -1};
2101 for (unsigned i = 0; i < SubVecSize; ++i) {
2102 for (unsigned j = i; j < Mask.size(); j += SubVecSize) {
2103 int M = Mask[j];
2104
2105 // Convert from vector index to 4-element subvector index
2106 // If an index refers to an element outside of the subvector then give up
2107 if (M != -1) {
2108 M -= 4 * (j / SubVecSize);
2109 if (M < 0 || M >= 4)
2110 return SDValue();
2111 }
2112
2113 // If the mask has an undef, replace it with the current index.
2114 // Note that it might still be undef if the current index is also undef
2115 if (SubMask[i] == -1)
2116 SubMask[i] = M;
2117 // Check that non-undef values are the same as in the mask. If they
2118 // aren't then give up
2119 else if (M != -1 && M != SubMask[i])
2120 return SDValue();
2121 }
2122 }
2123
2124 // Calculate the immediate. Replace any remaining undefs with zero
2125 int Imm = 0;
2126 for (int i = SubVecSize - 1; i >= 0; --i) {
2127 int M = SubMask[i];
2128
2129 if (M == -1)
2130 M = 0;
2131
2132 Imm <<= 2;
2133 Imm |= M & 0x3;
2134 }
2135
2136 MVT GRLenVT = Subtarget.getGRLenVT();
2137
2138 // Return vshuf4i.d
2139 if (VT == MVT::v2f64 || VT == MVT::v2i64)
2140 return DAG.getNode(LoongArchISD::VSHUF4I_D, DL, VT, V1, V2,
2141 DAG.getConstant(Imm, DL, GRLenVT));
2142
2143 return DAG.getNode(LoongArchISD::VSHUF4I, DL, VT, V1,
2144 DAG.getConstant(Imm, DL, GRLenVT));
2145}
2146
2147/// Lower VECTOR_SHUFFLE whose result is the reversed source vector.
2148///
2149/// It is possible to do optimization for VECTOR_SHUFFLE performing vector
2150/// reverse whose mask likes:
2151/// <7, 6, 5, 4, 3, 2, 1, 0>
2152///
2153/// When undef's appear in the mask they are treated as if they were whatever
2154/// value is necessary in order to fit the above forms.
2155static SDValue
2157 SDValue V1, SelectionDAG &DAG,
2158 const LoongArchSubtarget &Subtarget) {
2159 // Only vectors with i8/i16 elements which cannot match other patterns
2160 // directly needs to do this.
2161 if (VT != MVT::v16i8 && VT != MVT::v8i16 && VT != MVT::v32i8 &&
2162 VT != MVT::v16i16)
2163 return SDValue();
2164
2165 if (!ShuffleVectorInst::isReverseMask(Mask, Mask.size()))
2166 return SDValue();
2167
2168 int WidenNumElts = VT.getVectorNumElements() / 4;
2169 SmallVector<int, 16> WidenMask(WidenNumElts, -1);
2170 for (int i = 0; i < WidenNumElts; ++i)
2171 WidenMask[i] = WidenNumElts - 1 - i;
2172
2173 MVT WidenVT = MVT::getVectorVT(
2174 VT.getVectorElementType() == MVT::i8 ? MVT::i32 : MVT::i64, WidenNumElts);
2175 SDValue NewV1 = DAG.getBitcast(WidenVT, V1);
2176 SDValue WidenRev = DAG.getVectorShuffle(WidenVT, DL, NewV1,
2177 DAG.getUNDEF(WidenVT), WidenMask);
2178
2179 return DAG.getNode(LoongArchISD::VSHUF4I, DL, VT,
2180 DAG.getBitcast(VT, WidenRev),
2181 DAG.getConstant(27, DL, Subtarget.getGRLenVT()));
2182}
2183
2184/// Lower VECTOR_SHUFFLE into VPACKEV (if possible).
2185///
2186/// VPACKEV interleaves the even elements from each vector.
2187///
2188/// It is possible to lower into VPACKEV when the mask consists of two of the
2189/// following forms interleaved:
2190/// <0, 2, 4, ...>
2191/// <n, n+2, n+4, ...>
2192/// where n is the number of elements in the vector.
2193/// For example:
2194/// <0, 0, 2, 2, 4, 4, ...>
2195/// <0, n, 2, n+2, 4, n+4, ...>
2196///
2197/// When undef's appear in the mask they are treated as if they were whatever
2198/// value is necessary in order to fit the above forms.
2200 MVT VT, SDValue V1, SDValue V2,
2201 SelectionDAG &DAG) {
2202
2203 const auto &Begin = Mask.begin();
2204 const auto &End = Mask.end();
2205 SDValue OriV1 = V1, OriV2 = V2;
2206
2207 if (fitsRegularPattern<int>(Begin, 2, End, 0, 2))
2208 V1 = OriV1;
2209 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size(), 2))
2210 V1 = OriV2;
2211 else
2212 return SDValue();
2213
2214 if (fitsRegularPattern<int>(Begin + 1, 2, End, 0, 2))
2215 V2 = OriV1;
2216 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size(), 2))
2217 V2 = OriV2;
2218 else
2219 return SDValue();
2220
2221 return DAG.getNode(LoongArchISD::VPACKEV, DL, VT, V2, V1);
2222}
2223
2224/// Lower VECTOR_SHUFFLE into VPACKOD (if possible).
2225///
2226/// VPACKOD interleaves the odd elements from each vector.
2227///
2228/// It is possible to lower into VPACKOD when the mask consists of two of the
2229/// following forms interleaved:
2230/// <1, 3, 5, ...>
2231/// <n+1, n+3, n+5, ...>
2232/// where n is the number of elements in the vector.
2233/// For example:
2234/// <1, 1, 3, 3, 5, 5, ...>
2235/// <1, n+1, 3, n+3, 5, n+5, ...>
2236///
2237/// When undef's appear in the mask they are treated as if they were whatever
2238/// value is necessary in order to fit the above forms.
2240 MVT VT, SDValue V1, SDValue V2,
2241 SelectionDAG &DAG) {
2242
2243 const auto &Begin = Mask.begin();
2244 const auto &End = Mask.end();
2245 SDValue OriV1 = V1, OriV2 = V2;
2246
2247 if (fitsRegularPattern<int>(Begin, 2, End, 1, 2))
2248 V1 = OriV1;
2249 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size() + 1, 2))
2250 V1 = OriV2;
2251 else
2252 return SDValue();
2253
2254 if (fitsRegularPattern<int>(Begin + 1, 2, End, 1, 2))
2255 V2 = OriV1;
2256 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size() + 1, 2))
2257 V2 = OriV2;
2258 else
2259 return SDValue();
2260
2261 return DAG.getNode(LoongArchISD::VPACKOD, DL, VT, V2, V1);
2262}
2263
2264/// Lower VECTOR_SHUFFLE into VILVH (if possible).
2265///
2266/// VILVH interleaves consecutive elements from the left (highest-indexed) half
2267/// of each vector.
2268///
2269/// It is possible to lower into VILVH when the mask consists of two of the
2270/// following forms interleaved:
2271/// <x, x+1, x+2, ...>
2272/// <n+x, n+x+1, n+x+2, ...>
2273/// where n is the number of elements in the vector and x is half n.
2274/// For example:
2275/// <x, x, x+1, x+1, x+2, x+2, ...>
2276/// <x, n+x, x+1, n+x+1, x+2, n+x+2, ...>
2277///
2278/// When undef's appear in the mask they are treated as if they were whatever
2279/// value is necessary in order to fit the above forms.
2281 MVT VT, SDValue V1, SDValue V2,
2282 SelectionDAG &DAG) {
2283
2284 const auto &Begin = Mask.begin();
2285 const auto &End = Mask.end();
2286 unsigned HalfSize = Mask.size() / 2;
2287 SDValue OriV1 = V1, OriV2 = V2;
2288
2289 if (fitsRegularPattern<int>(Begin, 2, End, HalfSize, 1))
2290 V1 = OriV1;
2291 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size() + HalfSize, 1))
2292 V1 = OriV2;
2293 else
2294 return SDValue();
2295
2296 if (fitsRegularPattern<int>(Begin + 1, 2, End, HalfSize, 1))
2297 V2 = OriV1;
2298 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size() + HalfSize,
2299 1))
2300 V2 = OriV2;
2301 else
2302 return SDValue();
2303
2304 return DAG.getNode(LoongArchISD::VILVH, DL, VT, V2, V1);
2305}
2306
2307/// Lower VECTOR_SHUFFLE into VILVL (if possible).
2308///
2309/// VILVL interleaves consecutive elements from the right (lowest-indexed) half
2310/// of each vector.
2311///
2312/// It is possible to lower into VILVL when the mask consists of two of the
2313/// following forms interleaved:
2314/// <0, 1, 2, ...>
2315/// <n, n+1, n+2, ...>
2316/// where n is the number of elements in the vector.
2317/// For example:
2318/// <0, 0, 1, 1, 2, 2, ...>
2319/// <0, n, 1, n+1, 2, n+2, ...>
2320///
2321/// When undef's appear in the mask they are treated as if they were whatever
2322/// value is necessary in order to fit the above forms.
2324 MVT VT, SDValue V1, SDValue V2,
2325 SelectionDAG &DAG) {
2326
2327 const auto &Begin = Mask.begin();
2328 const auto &End = Mask.end();
2329 SDValue OriV1 = V1, OriV2 = V2;
2330
2331 if (fitsRegularPattern<int>(Begin, 2, End, 0, 1))
2332 V1 = OriV1;
2333 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size(), 1))
2334 V1 = OriV2;
2335 else
2336 return SDValue();
2337
2338 if (fitsRegularPattern<int>(Begin + 1, 2, End, 0, 1))
2339 V2 = OriV1;
2340 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size(), 1))
2341 V2 = OriV2;
2342 else
2343 return SDValue();
2344
2345 return DAG.getNode(LoongArchISD::VILVL, DL, VT, V2, V1);
2346}
2347
2348/// Lower VECTOR_SHUFFLE into VPICKEV (if possible).
2349///
2350/// VPICKEV copies the even elements of each vector into the result vector.
2351///
2352/// It is possible to lower into VPICKEV when the mask consists of two of the
2353/// following forms concatenated:
2354/// <0, 2, 4, ...>
2355/// <n, n+2, n+4, ...>
2356/// where n is the number of elements in the vector.
2357/// For example:
2358/// <0, 2, 4, ..., 0, 2, 4, ...>
2359/// <0, 2, 4, ..., n, n+2, n+4, ...>
2360///
2361/// When undef's appear in the mask they are treated as if they were whatever
2362/// value is necessary in order to fit the above forms.
2364 MVT VT, SDValue V1, SDValue V2,
2365 SelectionDAG &DAG) {
2366
2367 const auto &Begin = Mask.begin();
2368 const auto &Mid = Mask.begin() + Mask.size() / 2;
2369 const auto &End = Mask.end();
2370 SDValue OriV1 = V1, OriV2 = V2;
2371
2372 if (fitsRegularPattern<int>(Begin, 1, Mid, 0, 2))
2373 V1 = OriV1;
2374 else if (fitsRegularPattern<int>(Begin, 1, Mid, Mask.size(), 2))
2375 V1 = OriV2;
2376 else
2377 return SDValue();
2378
2379 if (fitsRegularPattern<int>(Mid, 1, End, 0, 2))
2380 V2 = OriV1;
2381 else if (fitsRegularPattern<int>(Mid, 1, End, Mask.size(), 2))
2382 V2 = OriV2;
2383
2384 else
2385 return SDValue();
2386
2387 return DAG.getNode(LoongArchISD::VPICKEV, DL, VT, V2, V1);
2388}
2389
2390/// Lower VECTOR_SHUFFLE into VPICKOD (if possible).
2391///
2392/// VPICKOD copies the odd elements of each vector into the result vector.
2393///
2394/// It is possible to lower into VPICKOD when the mask consists of two of the
2395/// following forms concatenated:
2396/// <1, 3, 5, ...>
2397/// <n+1, n+3, n+5, ...>
2398/// where n is the number of elements in the vector.
2399/// For example:
2400/// <1, 3, 5, ..., 1, 3, 5, ...>
2401/// <1, 3, 5, ..., n+1, n+3, n+5, ...>
2402///
2403/// When undef's appear in the mask they are treated as if they were whatever
2404/// value is necessary in order to fit the above forms.
2406 MVT VT, SDValue V1, SDValue V2,
2407 SelectionDAG &DAG) {
2408
2409 const auto &Begin = Mask.begin();
2410 const auto &Mid = Mask.begin() + Mask.size() / 2;
2411 const auto &End = Mask.end();
2412 SDValue OriV1 = V1, OriV2 = V2;
2413
2414 if (fitsRegularPattern<int>(Begin, 1, Mid, 1, 2))
2415 V1 = OriV1;
2416 else if (fitsRegularPattern<int>(Begin, 1, Mid, Mask.size() + 1, 2))
2417 V1 = OriV2;
2418 else
2419 return SDValue();
2420
2421 if (fitsRegularPattern<int>(Mid, 1, End, 1, 2))
2422 V2 = OriV1;
2423 else if (fitsRegularPattern<int>(Mid, 1, End, Mask.size() + 1, 2))
2424 V2 = OriV2;
2425 else
2426 return SDValue();
2427
2428 return DAG.getNode(LoongArchISD::VPICKOD, DL, VT, V2, V1);
2429}
2430
2431/// Lower VECTOR_SHUFFLE into VEXTRINS (if possible).
2432///
2433/// VEXTRINS copies one element of a vector into any place of the result
2434/// vector and makes no change to the rest elements of the result vector.
2435///
2436/// It is possible to lower into VEXTRINS when the mask takes the form:
2437/// <0, 1, 2, ..., n+i, ..., n-1> or <n, n+1, n+2, ..., i, ..., 2n-1> or
2438/// <0, 1, 2, ..., i, ..., n-1> or <n, n+1, n+2, ..., n+i, ..., 2n-1>
2439/// where n is the number of elements in the vector and i is in [0, n).
2440/// For example:
2441/// <0, 1, 2, 3, 4, 5, 6, 8> , <2, 9, 10, 11, 12, 13, 14, 15> ,
2442/// <0, 1, 2, 6, 4, 5, 6, 7> , <8, 9, 10, 11, 12, 9, 14, 15>
2443///
2444/// When undef's appear in the mask they are treated as if they were whatever
2445/// value is necessary in order to fit the above forms.
2446static SDValue
2448 SDValue V1, SDValue V2, SelectionDAG &DAG,
2449 const LoongArchSubtarget &Subtarget) {
2450 unsigned NumElts = VT.getVectorNumElements();
2451 MVT EltVT = VT.getVectorElementType();
2452 MVT GRLenVT = Subtarget.getGRLenVT();
2453
2454 if (Mask.size() != NumElts)
2455 return SDValue();
2456
2457 auto tryLowerToExtrAndIns = [&](unsigned Base) -> SDValue {
2458 int DiffCount = 0;
2459 int DiffPos = -1;
2460 for (unsigned i = 0; i < NumElts; ++i) {
2461 if (Mask[i] == -1)
2462 continue;
2463 if (Mask[i] != int(Base + i)) {
2464 ++DiffCount;
2465 DiffPos = int(i);
2466 if (DiffCount > 1)
2467 return SDValue();
2468 }
2469 }
2470
2471 // Need exactly one differing element to lower into VEXTRINS.
2472 if (DiffCount != 1)
2473 return SDValue();
2474
2475 // DiffMask must be in [0, 2N).
2476 int DiffMask = Mask[DiffPos];
2477 if (DiffMask < 0 || DiffMask >= int(2 * NumElts))
2478 return SDValue();
2479
2480 // Determine source vector and source index.
2481 SDValue SrcVec;
2482 unsigned SrcIdx;
2483 if (unsigned(DiffMask) < NumElts) {
2484 SrcVec = V1;
2485 SrcIdx = unsigned(DiffMask);
2486 } else {
2487 SrcVec = V2;
2488 SrcIdx = unsigned(DiffMask) - NumElts;
2489 }
2490
2491 // Replace with EXTRACT_VECTOR_ELT + INSERT_VECTOR_ELT, it will match the
2492 // patterns of VEXTRINS in tablegen.
2493 SDValue Extracted = DAG.getNode(
2494 ISD::EXTRACT_VECTOR_ELT, DL, EltVT.isFloatingPoint() ? EltVT : GRLenVT,
2495 SrcVec, DAG.getConstant(SrcIdx, DL, GRLenVT));
2496 SDValue Result =
2497 DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, (Base == 0) ? V1 : V2,
2498 Extracted, DAG.getConstant(DiffPos, DL, GRLenVT));
2499
2500 return Result;
2501 };
2502
2503 // Try [0, n-1) insertion then [n, 2n-1) insertion.
2504 if (SDValue Result = tryLowerToExtrAndIns(0))
2505 return Result;
2506 return tryLowerToExtrAndIns(NumElts);
2507}
2508
2509// Check the Mask and then build SrcVec and MaskImm infos which will
2510// be used to build LoongArchISD nodes for VPERMI_W or XVPERMI_W.
2511// On success, return true. Otherwise, return false.
2514 unsigned &MaskImm) {
2515 unsigned MaskSize = Mask.size();
2516
2517 auto isValid = [&](int M, int Off) {
2518 return (M == -1) || (M >= Off && M < Off + 4);
2519 };
2520
2521 auto buildImm = [&](int MLo, int MHi, unsigned Off, unsigned I) {
2522 auto immPart = [&](int M, unsigned Off) {
2523 return (M == -1 ? 0 : (M - Off)) & 0x3;
2524 };
2525 MaskImm |= immPart(MLo, Off) << (I * 2);
2526 MaskImm |= immPart(MHi, Off) << ((I + 1) * 2);
2527 };
2528
2529 for (unsigned i = 0; i < 4; i += 2) {
2530 int MLo = Mask[i];
2531 int MHi = Mask[i + 1];
2532
2533 if (MaskSize == 8) { // Only v8i32/v8f32 need this check.
2534 auto isValid2 = [&](int &M, int M2) {
2535 // If high half index is undef, it's always valid.
2536 if (M2 == -1)
2537 return true;
2538 if (M == -1) {
2539 // If low half index is undef, use index from high half,
2540 // remapped to low half.
2541 if ((M2 % MaskSize) < 4)
2542 return false;
2543 M = M2 - 4;
2544 return true;
2545 }
2546 // Index in low half must be same as index in high half.
2547 return M2 == M + 4;
2548 };
2549 if (!isValid2(MLo, Mask[i + 4]) || !isValid2(MHi, Mask[i + 5]))
2550 return false;
2551 }
2552
2553 if (isValid(MLo, 0) && isValid(MHi, 0)) {
2554 SrcVec.push_back(V1);
2555 buildImm(MLo, MHi, 0, i);
2556 } else if (isValid(MLo, MaskSize) && isValid(MHi, MaskSize)) {
2557 SrcVec.push_back(V2);
2558 buildImm(MLo, MHi, MaskSize, i);
2559 } else {
2560 return false;
2561 }
2562 }
2563
2564 return true;
2565}
2566
2567/// Lower VECTOR_SHUFFLE into VPERMI (if possible).
2568///
2569/// VPERMI selects two elements from each of the two vectors based on the
2570/// mask and places them in the corresponding positions of the result vector
2571/// in order. Only v4i32 and v4f32 types are allowed.
2572///
2573/// It is possible to lower into VPERMI when the mask consists of two of the
2574/// following forms concatenated:
2575/// <i, j, u, v>
2576/// <u, v, i, j>
2577/// where i,j are in [0,4) and u,v are in [4, 8).
2578/// For example:
2579/// <2, 3, 4, 5>
2580/// <5, 7, 0, 2>
2581///
2582/// When undef's appear in the mask they are treated as if they were whatever
2583/// value is necessary in order to fit the above forms.
2585 MVT VT, SDValue V1, SDValue V2,
2586 SelectionDAG &DAG,
2587 const LoongArchSubtarget &Subtarget) {
2588 if ((VT != MVT::v4i32 && VT != MVT::v4f32) ||
2589 Mask.size() != VT.getVectorNumElements())
2590 return SDValue();
2591
2593 unsigned MaskImm = 0;
2594 if (!buildVPERMIInfo(Mask, V1, V2, SrcVec, MaskImm))
2595 return SDValue();
2596
2597 return DAG.getNode(LoongArchISD::VPERMI, DL, VT, SrcVec[1], SrcVec[0],
2598 DAG.getConstant(MaskImm, DL, Subtarget.getGRLenVT()));
2599}
2600
2601/// Lower VECTOR_SHUFFLE into VSHUF.
2602///
2603/// This mostly consists of converting the shuffle mask into a BUILD_VECTOR and
2604/// adding it as an operand to the resulting VSHUF.
2606 MVT VT, SDValue V1, SDValue V2,
2607 SelectionDAG &DAG,
2608 const LoongArchSubtarget &Subtarget) {
2609
2611 for (auto M : Mask)
2612 Ops.push_back(DAG.getSignedConstant(M, DL, Subtarget.getGRLenVT()));
2613
2614 EVT MaskVecTy = VT.changeVectorElementTypeToInteger();
2615 SDValue MaskVec = DAG.getBuildVector(MaskVecTy, DL, Ops);
2616
2617 // VECTOR_SHUFFLE concatenates the vectors in an vectorwise fashion.
2618 // <0b00, 0b01> + <0b10, 0b11> -> <0b00, 0b01, 0b10, 0b11>
2619 // VSHF concatenates the vectors in a bitwise fashion:
2620 // <0b00, 0b01> + <0b10, 0b11> ->
2621 // 0b0100 + 0b1110 -> 0b01001110
2622 // <0b10, 0b11, 0b00, 0b01>
2623 // We must therefore swap the operands to get the correct result.
2624 return DAG.getNode(LoongArchISD::VSHUF, DL, VT, MaskVec, V2, V1);
2625}
2626
2627/// Dispatching routine to lower various 128-bit LoongArch vector shuffles.
2628///
2629/// This routine breaks down the specific type of 128-bit shuffle and
2630/// dispatches to the lowering routines accordingly.
2632 SDValue V1, SDValue V2, SelectionDAG &DAG,
2633 const LoongArchSubtarget &Subtarget) {
2634 assert((VT.SimpleTy == MVT::v16i8 || VT.SimpleTy == MVT::v8i16 ||
2635 VT.SimpleTy == MVT::v4i32 || VT.SimpleTy == MVT::v2i64 ||
2636 VT.SimpleTy == MVT::v4f32 || VT.SimpleTy == MVT::v2f64) &&
2637 "Vector type is unsupported for lsx!");
2638 assert(V1.getSimpleValueType() == V2.getSimpleValueType() &&
2639 "Two operands have different types!");
2640 assert(VT.getVectorNumElements() == Mask.size() &&
2641 "Unexpected mask size for shuffle!");
2642 assert(Mask.size() % 2 == 0 && "Expected even mask size.");
2643
2644 APInt KnownUndef, KnownZero;
2645 computeZeroableShuffleElements(Mask, V1, V2, KnownUndef, KnownZero);
2646 APInt Zeroable = KnownUndef | KnownZero;
2647
2648 SDValue Result;
2649 // TODO: Add more comparison patterns.
2650 if (V2.isUndef()) {
2651 if ((Result =
2652 lowerVECTOR_SHUFFLE_VREPLVEI(DL, Mask, VT, V1, DAG, Subtarget)))
2653 return Result;
2654 if ((Result =
2655 lowerVECTOR_SHUFFLE_VSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2656 return Result;
2657 if ((Result =
2658 lowerVECTOR_SHUFFLE_IsReverse(DL, Mask, VT, V1, DAG, Subtarget)))
2659 return Result;
2660
2661 // TODO: This comment may be enabled in the future to better match the
2662 // pattern for instruction selection.
2663 /* V2 = V1; */
2664 }
2665
2666 // It is recommended not to change the pattern comparison order for better
2667 // performance.
2668 if ((Result = lowerVECTOR_SHUFFLE_VPACKEV(DL, Mask, VT, V1, V2, DAG)))
2669 return Result;
2670 if ((Result = lowerVECTOR_SHUFFLE_VPACKOD(DL, Mask, VT, V1, V2, DAG)))
2671 return Result;
2672 if ((Result = lowerVECTOR_SHUFFLE_VILVH(DL, Mask, VT, V1, V2, DAG)))
2673 return Result;
2674 if ((Result = lowerVECTOR_SHUFFLE_VILVL(DL, Mask, VT, V1, V2, DAG)))
2675 return Result;
2676 if ((Result = lowerVECTOR_SHUFFLE_VPICKEV(DL, Mask, VT, V1, V2, DAG)))
2677 return Result;
2678 if ((Result = lowerVECTOR_SHUFFLE_VPICKOD(DL, Mask, VT, V1, V2, DAG)))
2679 return Result;
2680 if ((VT.SimpleTy == MVT::v2i64 || VT.SimpleTy == MVT::v2f64) &&
2681 (Result =
2682 lowerVECTOR_SHUFFLE_VSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2683 return Result;
2684 if ((Result =
2685 lowerVECTOR_SHUFFLE_VEXTRINS(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2686 return Result;
2687 if ((Result = lowerVECTOR_SHUFFLEAsShift(DL, Mask, VT, V1, V2, DAG, Subtarget,
2688 Zeroable)))
2689 return Result;
2690 if ((Result =
2691 lowerVECTOR_SHUFFLE_VPERMI(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2692 return Result;
2693 if ((Result = lowerVECTOR_SHUFFLEAsZeroOrAnyExtend(DL, Mask, VT, V1, V2, DAG,
2694 Zeroable)))
2695 return Result;
2696 if ((Result = lowerVECTOR_SHUFFLEAsByteRotate(DL, Mask, VT, V1, V2, DAG,
2697 Subtarget)))
2698 return Result;
2699 if (SDValue NewShuffle = widenShuffleMask(DL, Mask, VT, V1, V2, DAG))
2700 return NewShuffle;
2701 if ((Result =
2702 lowerVECTOR_SHUFFLE_VSHUF(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2703 return Result;
2704 return SDValue();
2705}
2706
2707/// Lower VECTOR_SHUFFLE into XVREPLVEI (if possible).
2708///
2709/// It is a XVREPLVEI when the mask is:
2710/// <x, x, x, ..., x+n, x+n, x+n, ...>
2711/// where the number of x is equal to n and n is half the length of vector.
2712///
2713/// When undef's appear in the mask they are treated as if they were whatever
2714/// value is necessary in order to fit the above form.
2715static SDValue
2717 SDValue V1, SelectionDAG &DAG,
2718 const LoongArchSubtarget &Subtarget) {
2719 int SplatIndex = -1;
2720 for (const auto &M : Mask) {
2721 if (M != -1) {
2722 SplatIndex = M;
2723 break;
2724 }
2725 }
2726
2727 if (SplatIndex == -1)
2728 return DAG.getUNDEF(VT);
2729
2730 const auto &Begin = Mask.begin();
2731 const auto &End = Mask.end();
2732 int HalfSize = Mask.size() / 2;
2733
2734 if (SplatIndex >= HalfSize)
2735 return SDValue();
2736
2737 assert(SplatIndex < (int)Mask.size() && "Out of bounds mask index");
2738 if (fitsRegularPattern<int>(Begin, 1, End - HalfSize, SplatIndex, 0) &&
2739 fitsRegularPattern<int>(Begin + HalfSize, 1, End, SplatIndex + HalfSize,
2740 0)) {
2741 return DAG.getNode(LoongArchISD::VREPLVEI, DL, VT, V1,
2742 DAG.getConstant(SplatIndex, DL, Subtarget.getGRLenVT()));
2743 }
2744
2745 return SDValue();
2746}
2747
2748/// Lower VECTOR_SHUFFLE into XVSHUF4I (if possible).
2749static SDValue
2751 SDValue V1, SDValue V2, SelectionDAG &DAG,
2752 const LoongArchSubtarget &Subtarget) {
2753 // XVSHUF4I_D must be handled separately because it is different from other
2754 // types of [X]VSHUF4I instructions.
2755 if (Mask.size() == 4) {
2756 unsigned MaskImm = 0;
2757 for (int i = 1; i >= 0; --i) {
2758 int MLo = Mask[i];
2759 int MHi = Mask[i + 2];
2760 if (!(MLo == -1 || (MLo >= 0 && MLo <= 1) || (MLo >= 4 && MLo <= 5)) ||
2761 !(MHi == -1 || (MHi >= 2 && MHi <= 3) || (MHi >= 6 && MHi <= 7)))
2762 return SDValue();
2763 if (MHi != -1 && MLo != -1 && MHi != MLo + 2)
2764 return SDValue();
2765
2766 MaskImm <<= 2;
2767 if (MLo != -1)
2768 MaskImm |= ((MLo <= 1) ? MLo : (MLo - 2)) & 0x3;
2769 else if (MHi != -1)
2770 MaskImm |= ((MHi <= 3) ? (MHi - 2) : (MHi - 4)) & 0x3;
2771 }
2772
2773 return DAG.getNode(LoongArchISD::VSHUF4I_D, DL, VT, V1, V2,
2774 DAG.getConstant(MaskImm, DL, Subtarget.getGRLenVT()));
2775 }
2776
2777 return lowerVECTOR_SHUFFLE_VSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget);
2778}
2779
2780/// Lower VECTOR_SHUFFLE into XVPERMI (if possible).
2781static SDValue
2783 SDValue V1, SDValue V2, SelectionDAG &DAG,
2784 const LoongArchSubtarget &Subtarget) {
2785 MVT GRLenVT = Subtarget.getGRLenVT();
2786 unsigned MaskSize = Mask.size();
2787 if (MaskSize != VT.getVectorNumElements())
2788 return SDValue();
2789
2790 // Consider XVPERMI_W.
2791 if (VT == MVT::v8i32 || VT == MVT::v8f32) {
2793 unsigned MaskImm = 0;
2794 if (!buildVPERMIInfo(Mask, V1, V2, SrcVec, MaskImm))
2795 return SDValue();
2796
2797 return DAG.getNode(LoongArchISD::VPERMI, DL, VT, SrcVec[1], SrcVec[0],
2798 DAG.getConstant(MaskImm, DL, GRLenVT));
2799 }
2800
2801 // Consider XVPERMI_D.
2802 if (VT == MVT::v4i64 || VT == MVT::v4f64) {
2803 unsigned MaskImm = 0;
2804 for (unsigned i = 0; i < MaskSize; ++i) {
2805 if (Mask[i] == -1)
2806 continue;
2807 if (Mask[i] >= (int)MaskSize)
2808 return SDValue();
2809 MaskImm |= Mask[i] << (i * 2);
2810 }
2811
2812 return DAG.getNode(LoongArchISD::XVPERMI, DL, VT, V1,
2813 DAG.getConstant(MaskImm, DL, GRLenVT));
2814 }
2815
2816 return SDValue();
2817}
2818
2819/// Lower VECTOR_SHUFFLE into XVPERM (if possible).
2821 MVT VT, SDValue V1, SelectionDAG &DAG,
2822 const LoongArchSubtarget &Subtarget) {
2823 // LoongArch LASX only have XVPERM_W.
2824 if (Mask.size() != 8 || (VT != MVT::v8i32 && VT != MVT::v8f32))
2825 return SDValue();
2826
2827 unsigned NumElts = VT.getVectorNumElements();
2828 unsigned HalfSize = NumElts / 2;
2829 bool FrontLo = true, FrontHi = true;
2830 bool BackLo = true, BackHi = true;
2831
2832 auto inRange = [](int val, int low, int high) {
2833 return (val == -1) || (val >= low && val < high);
2834 };
2835
2836 for (unsigned i = 0; i < HalfSize; ++i) {
2837 int Fronti = Mask[i];
2838 int Backi = Mask[i + HalfSize];
2839
2840 FrontLo &= inRange(Fronti, 0, HalfSize);
2841 FrontHi &= inRange(Fronti, HalfSize, NumElts);
2842 BackLo &= inRange(Backi, 0, HalfSize);
2843 BackHi &= inRange(Backi, HalfSize, NumElts);
2844 }
2845
2846 // If both the lower and upper 128-bit parts access only one half of the
2847 // vector (either lower or upper), avoid using xvperm.w. The latency of
2848 // xvperm.w(3) is higher than using xvshuf(1) and xvori(1).
2849 if ((FrontLo || FrontHi) && (BackLo || BackHi))
2850 return SDValue();
2851
2853 MVT GRLenVT = Subtarget.getGRLenVT();
2854 for (unsigned i = 0; i < NumElts; ++i)
2855 Masks.push_back(Mask[i] == -1 ? DAG.getUNDEF(GRLenVT)
2856 : DAG.getConstant(Mask[i], DL, GRLenVT));
2857 SDValue MaskVec = DAG.getBuildVector(MVT::v8i32, DL, Masks);
2858
2859 return DAG.getNode(LoongArchISD::XVPERM, DL, VT, V1, MaskVec);
2860}
2861
2862/// Lower VECTOR_SHUFFLE into XVPACKEV (if possible).
2864 MVT VT, SDValue V1, SDValue V2,
2865 SelectionDAG &DAG) {
2866 return lowerVECTOR_SHUFFLE_VPACKEV(DL, Mask, VT, V1, V2, DAG);
2867}
2868
2869/// Lower VECTOR_SHUFFLE into XVPACKOD (if possible).
2871 MVT VT, SDValue V1, SDValue V2,
2872 SelectionDAG &DAG) {
2873 return lowerVECTOR_SHUFFLE_VPACKOD(DL, Mask, VT, V1, V2, DAG);
2874}
2875
2876/// Lower VECTOR_SHUFFLE into XVILVH (if possible).
2878 MVT VT, SDValue V1, SDValue V2,
2879 SelectionDAG &DAG) {
2880
2881 const auto &Begin = Mask.begin();
2882 const auto &End = Mask.end();
2883 unsigned HalfSize = Mask.size() / 2;
2884 unsigned LeftSize = HalfSize / 2;
2885 SDValue OriV1 = V1, OriV2 = V2;
2886
2887 if (fitsRegularPattern<int>(Begin, 2, End - HalfSize, HalfSize - LeftSize,
2888 1) &&
2889 fitsRegularPattern<int>(Begin + HalfSize, 2, End, HalfSize + LeftSize, 1))
2890 V1 = OriV1;
2891 else if (fitsRegularPattern<int>(Begin, 2, End - HalfSize,
2892 Mask.size() + HalfSize - LeftSize, 1) &&
2893 fitsRegularPattern<int>(Begin + HalfSize, 2, End,
2894 Mask.size() + HalfSize + LeftSize, 1))
2895 V1 = OriV2;
2896 else
2897 return SDValue();
2898
2899 if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize, HalfSize - LeftSize,
2900 1) &&
2901 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End, HalfSize + LeftSize,
2902 1))
2903 V2 = OriV1;
2904 else if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize,
2905 Mask.size() + HalfSize - LeftSize, 1) &&
2906 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End,
2907 Mask.size() + HalfSize + LeftSize, 1))
2908 V2 = OriV2;
2909 else
2910 return SDValue();
2911
2912 return DAG.getNode(LoongArchISD::VILVH, DL, VT, V2, V1);
2913}
2914
2915/// Lower VECTOR_SHUFFLE into XVILVL (if possible).
2917 MVT VT, SDValue V1, SDValue V2,
2918 SelectionDAG &DAG) {
2919
2920 const auto &Begin = Mask.begin();
2921 const auto &End = Mask.end();
2922 unsigned HalfSize = Mask.size() / 2;
2923 SDValue OriV1 = V1, OriV2 = V2;
2924
2925 if (fitsRegularPattern<int>(Begin, 2, End - HalfSize, 0, 1) &&
2926 fitsRegularPattern<int>(Begin + HalfSize, 2, End, HalfSize, 1))
2927 V1 = OriV1;
2928 else if (fitsRegularPattern<int>(Begin, 2, End - HalfSize, Mask.size(), 1) &&
2929 fitsRegularPattern<int>(Begin + HalfSize, 2, End,
2930 Mask.size() + HalfSize, 1))
2931 V1 = OriV2;
2932 else
2933 return SDValue();
2934
2935 if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize, 0, 1) &&
2936 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End, HalfSize, 1))
2937 V2 = OriV1;
2938 else if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize, Mask.size(),
2939 1) &&
2940 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End,
2941 Mask.size() + HalfSize, 1))
2942 V2 = OriV2;
2943 else
2944 return SDValue();
2945
2946 return DAG.getNode(LoongArchISD::VILVL, DL, VT, V2, V1);
2947}
2948
2949/// Lower VECTOR_SHUFFLE into XVPICKEV (if possible).
2951 MVT VT, SDValue V1, SDValue V2,
2952 SelectionDAG &DAG) {
2953
2954 const auto &Begin = Mask.begin();
2955 const auto &LeftMid = Mask.begin() + Mask.size() / 4;
2956 const auto &Mid = Mask.begin() + Mask.size() / 2;
2957 const auto &RightMid = Mask.end() - Mask.size() / 4;
2958 const auto &End = Mask.end();
2959 unsigned HalfSize = Mask.size() / 2;
2960 SDValue OriV1 = V1, OriV2 = V2;
2961
2962 if (fitsRegularPattern<int>(Begin, 1, LeftMid, 0, 2) &&
2963 fitsRegularPattern<int>(Mid, 1, RightMid, HalfSize, 2))
2964 V1 = OriV1;
2965 else if (fitsRegularPattern<int>(Begin, 1, LeftMid, Mask.size(), 2) &&
2966 fitsRegularPattern<int>(Mid, 1, RightMid, Mask.size() + HalfSize, 2))
2967 V1 = OriV2;
2968 else
2969 return SDValue();
2970
2971 if (fitsRegularPattern<int>(LeftMid, 1, Mid, 0, 2) &&
2972 fitsRegularPattern<int>(RightMid, 1, End, HalfSize, 2))
2973 V2 = OriV1;
2974 else if (fitsRegularPattern<int>(LeftMid, 1, Mid, Mask.size(), 2) &&
2975 fitsRegularPattern<int>(RightMid, 1, End, Mask.size() + HalfSize, 2))
2976 V2 = OriV2;
2977
2978 else
2979 return SDValue();
2980
2981 return DAG.getNode(LoongArchISD::VPICKEV, DL, VT, V2, V1);
2982}
2983
2984/// Lower VECTOR_SHUFFLE into XVPICKOD (if possible).
2986 MVT VT, SDValue V1, SDValue V2,
2987 SelectionDAG &DAG) {
2988
2989 const auto &Begin = Mask.begin();
2990 const auto &LeftMid = Mask.begin() + Mask.size() / 4;
2991 const auto &Mid = Mask.begin() + Mask.size() / 2;
2992 const auto &RightMid = Mask.end() - Mask.size() / 4;
2993 const auto &End = Mask.end();
2994 unsigned HalfSize = Mask.size() / 2;
2995 SDValue OriV1 = V1, OriV2 = V2;
2996
2997 if (fitsRegularPattern<int>(Begin, 1, LeftMid, 1, 2) &&
2998 fitsRegularPattern<int>(Mid, 1, RightMid, HalfSize + 1, 2))
2999 V1 = OriV1;
3000 else if (fitsRegularPattern<int>(Begin, 1, LeftMid, Mask.size() + 1, 2) &&
3001 fitsRegularPattern<int>(Mid, 1, RightMid, Mask.size() + HalfSize + 1,
3002 2))
3003 V1 = OriV2;
3004 else
3005 return SDValue();
3006
3007 if (fitsRegularPattern<int>(LeftMid, 1, Mid, 1, 2) &&
3008 fitsRegularPattern<int>(RightMid, 1, End, HalfSize + 1, 2))
3009 V2 = OriV1;
3010 else if (fitsRegularPattern<int>(LeftMid, 1, Mid, Mask.size() + 1, 2) &&
3011 fitsRegularPattern<int>(RightMid, 1, End, Mask.size() + HalfSize + 1,
3012 2))
3013 V2 = OriV2;
3014 else
3015 return SDValue();
3016
3017 return DAG.getNode(LoongArchISD::VPICKOD, DL, VT, V2, V1);
3018}
3019
3020/// Lower VECTOR_SHUFFLE into XVEXTRINS (if possible).
3021static SDValue
3023 SDValue V1, SDValue V2, SelectionDAG &DAG,
3024 const LoongArchSubtarget &Subtarget) {
3025 int NumElts = VT.getVectorNumElements();
3026 int HalfSize = NumElts / 2;
3027 MVT EltVT = VT.getVectorElementType();
3028 MVT GRLenVT = Subtarget.getGRLenVT();
3029
3030 if ((int)Mask.size() != NumElts)
3031 return SDValue();
3032
3033 auto tryLowerToExtrAndIns = [&](int Base) -> SDValue {
3034 SmallVector<int> DiffPos;
3035 for (int i = 0; i < NumElts; ++i) {
3036 if (Mask[i] == -1)
3037 continue;
3038 if (Mask[i] != Base + i) {
3039 DiffPos.push_back(i);
3040 if (DiffPos.size() > 2)
3041 return SDValue();
3042 }
3043 }
3044
3045 // Need exactly two differing element to lower into XVEXTRINS.
3046 // If only one differing element, the element at a distance of
3047 // HalfSize from it must be undef.
3048 if (DiffPos.size() == 1) {
3049 if (DiffPos[0] < HalfSize && Mask[DiffPos[0] + HalfSize] == -1)
3050 DiffPos.push_back(DiffPos[0] + HalfSize);
3051 else if (DiffPos[0] >= HalfSize && Mask[DiffPos[0] - HalfSize] == -1)
3052 DiffPos.insert(DiffPos.begin(), DiffPos[0] - HalfSize);
3053 else
3054 return SDValue();
3055 }
3056 if (DiffPos.size() != 2 || DiffPos[1] != DiffPos[0] + HalfSize)
3057 return SDValue();
3058
3059 // DiffMask must be in its low or high part.
3060 int DiffMaskLo = Mask[DiffPos[0]];
3061 int DiffMaskHi = Mask[DiffPos[1]];
3062 DiffMaskLo = DiffMaskLo == -1 ? DiffMaskHi - HalfSize : DiffMaskLo;
3063 DiffMaskHi = DiffMaskHi == -1 ? DiffMaskLo + HalfSize : DiffMaskHi;
3064 if (!(DiffMaskLo >= 0 && DiffMaskLo < HalfSize) &&
3065 !(DiffMaskLo >= NumElts && DiffMaskLo < NumElts + HalfSize))
3066 return SDValue();
3067 if (!(DiffMaskHi >= HalfSize && DiffMaskHi < NumElts) &&
3068 !(DiffMaskHi >= NumElts + HalfSize && DiffMaskHi < 2 * NumElts))
3069 return SDValue();
3070 if (DiffMaskHi != DiffMaskLo + HalfSize)
3071 return SDValue();
3072
3073 // Determine source vector and source index.
3074 SDValue SrcVec = (DiffMaskLo < HalfSize) ? V1 : V2;
3075 int SrcIdxLo =
3076 (DiffMaskLo < HalfSize) ? DiffMaskLo : (DiffMaskLo - NumElts);
3077 bool IsEltFP = EltVT.isFloatingPoint();
3078
3079 // Replace with 2*EXTRACT_VECTOR_ELT + 2*INSERT_VECTOR_ELT, it will match
3080 // the patterns of XVEXTRINS in tablegen.
3081 SDValue BaseVec = (Base == 0) ? V1 : V2;
3082 SDValue EltLo =
3083 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, IsEltFP ? EltVT : GRLenVT,
3084 SrcVec, DAG.getConstant(SrcIdxLo, DL, GRLenVT));
3085 SDValue InsLo = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, BaseVec, EltLo,
3086 DAG.getConstant(DiffPos[0], DL, GRLenVT));
3087 SDValue EltHi =
3088 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, IsEltFP ? EltVT : GRLenVT,
3089 SrcVec, DAG.getConstant(SrcIdxLo + HalfSize, DL, GRLenVT));
3090 SDValue Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, InsLo, EltHi,
3091 DAG.getConstant(DiffPos[1], DL, GRLenVT));
3092
3093 return Result;
3094 };
3095
3096 // Try [0, n-1) insertion then [n, 2n-1) insertion.
3097 if (SDValue Result = tryLowerToExtrAndIns(0))
3098 return Result;
3099 return tryLowerToExtrAndIns(NumElts);
3100}
3101
3102/// Lower VECTOR_SHUFFLE into XVINSVE0 (if possible).
3103static SDValue
3105 SDValue V1, SDValue V2, SelectionDAG &DAG,
3106 const LoongArchSubtarget &Subtarget) {
3107 // LoongArch LASX only supports xvinsve0.{w/d}.
3108 if (VT != MVT::v8i32 && VT != MVT::v8f32 && VT != MVT::v4i64 &&
3109 VT != MVT::v4f64)
3110 return SDValue();
3111
3112 MVT GRLenVT = Subtarget.getGRLenVT();
3113 int MaskSize = Mask.size();
3114 assert(MaskSize == (int)VT.getVectorNumElements() && "Unexpected mask size");
3115
3116 // Check if exactly one element of the Mask is replaced by 'Replaced', while
3117 // all other elements are either 'Base + i' or undef (-1). On success, return
3118 // the index of the replaced element. Otherwise, just return -1.
3119 auto checkReplaceOne = [&](int Base, int Replaced) -> int {
3120 int Idx = -1;
3121 for (int i = 0; i < MaskSize; ++i) {
3122 if (Mask[i] == Base + i || Mask[i] == -1)
3123 continue;
3124 if (Mask[i] != Replaced)
3125 return -1;
3126 if (Idx == -1)
3127 Idx = i;
3128 else
3129 return -1;
3130 }
3131 return Idx;
3132 };
3133
3134 // Case 1: the lowest element of V2 replaces one element in V1.
3135 int Idx = checkReplaceOne(0, MaskSize);
3136 if (Idx != -1)
3137 return DAG.getNode(LoongArchISD::XVINSVE0, DL, VT, V1, V2,
3138 DAG.getConstant(Idx, DL, GRLenVT));
3139
3140 // Case 2: the lowest element of V1 replaces one element in V2.
3141 Idx = checkReplaceOne(MaskSize, 0);
3142 if (Idx != -1)
3143 return DAG.getNode(LoongArchISD::XVINSVE0, DL, VT, V2, V1,
3144 DAG.getConstant(Idx, DL, GRLenVT));
3145
3146 return SDValue();
3147}
3148
3149/// Lower VECTOR_SHUFFLE into XVSHUF (if possible).
3151 MVT VT, SDValue V1, SDValue V2,
3152 SelectionDAG &DAG) {
3153
3154 int MaskSize = Mask.size();
3155 int HalfSize = Mask.size() / 2;
3156 const auto &Begin = Mask.begin();
3157 const auto &Mid = Mask.begin() + HalfSize;
3158 const auto &End = Mask.end();
3159
3160 // VECTOR_SHUFFLE concatenates the vectors:
3161 // <0, 1, 2, 3, 4, 5, 6, 7> + <8, 9, 10, 11, 12, 13, 14, 15>
3162 // shuffling ->
3163 // <0, 1, 2, 3, 8, 9, 10, 11> <4, 5, 6, 7, 12, 13, 14, 15>
3164 //
3165 // XVSHUF concatenates the vectors:
3166 // <a0, a1, a2, a3, b0, b1, b2, b3> + <a4, a5, a6, a7, b4, b5, b6, b7>
3167 // shuffling ->
3168 // <a0, a1, a2, a3, a4, a5, a6, a7> + <b0, b1, b2, b3, b4, b5, b6, b7>
3169 SmallVector<SDValue, 8> MaskAlloc;
3170 for (auto it = Begin; it < Mid; it++) {
3171 if (*it < 0) // UNDEF
3172 MaskAlloc.push_back(DAG.getTargetConstant(0, DL, MVT::i64));
3173 else if ((*it >= 0 && *it < HalfSize) ||
3174 (*it >= MaskSize && *it < MaskSize + HalfSize)) {
3175 int M = *it < HalfSize ? *it : *it - HalfSize;
3176 MaskAlloc.push_back(DAG.getTargetConstant(M, DL, MVT::i64));
3177 } else
3178 return SDValue();
3179 }
3180 assert((int)MaskAlloc.size() == HalfSize && "xvshuf convert failed!");
3181
3182 for (auto it = Mid; it < End; it++) {
3183 if (*it < 0) // UNDEF
3184 MaskAlloc.push_back(DAG.getTargetConstant(0, DL, MVT::i64));
3185 else if ((*it >= HalfSize && *it < MaskSize) ||
3186 (*it >= MaskSize + HalfSize && *it < MaskSize * 2)) {
3187 int M = *it < MaskSize ? *it - HalfSize : *it - MaskSize;
3188 MaskAlloc.push_back(DAG.getTargetConstant(M, DL, MVT::i64));
3189 } else
3190 return SDValue();
3191 }
3192 assert((int)MaskAlloc.size() == MaskSize && "xvshuf convert failed!");
3193
3194 EVT MaskVecTy = VT.changeVectorElementTypeToInteger();
3195 SDValue MaskVec = DAG.getBuildVector(MaskVecTy, DL, MaskAlloc);
3196 return DAG.getNode(LoongArchISD::VSHUF, DL, VT, MaskVec, V2, V1);
3197}
3198
3199/// Shuffle vectors by lane to generate more optimized instructions.
3200/// 256-bit shuffles are always considered as 2-lane 128-bit shuffles.
3201///
3202/// Therefore, except for the following four cases, other cases are regarded
3203/// as cross-lane shuffles, where optimization is relatively limited.
3204///
3205/// - Shuffle high, low lanes of two inputs vector
3206/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <0, 5, 3, 6>
3207/// - Shuffle low, high lanes of two inputs vector
3208/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <3, 6, 0, 5>
3209/// - Shuffle low, low lanes of two inputs vector
3210/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <3, 6, 3, 6>
3211/// - Shuffle high, high lanes of two inputs vector
3212/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <0, 5, 0, 5>
3213///
3214/// The first case is the closest to LoongArch instructions and the other
3215/// cases need to be converted to it for processing.
3216///
3217/// This function will return true for the last three cases above and will
3218/// modify V1, V2 and Mask. Otherwise, return false for the first case and
3219/// cross-lane shuffle cases.
3221 const SDLoc &DL, MutableArrayRef<int> Mask, MVT VT, SDValue &V1,
3222 SDValue &V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget) {
3223
3224 enum HalfMaskType { HighLaneTy, LowLaneTy, None };
3225
3226 int MaskSize = Mask.size();
3227 int HalfSize = Mask.size() / 2;
3228 MVT GRLenVT = Subtarget.getGRLenVT();
3229
3230 HalfMaskType preMask = None, postMask = None;
3231
3232 if (std::all_of(Mask.begin(), Mask.begin() + HalfSize, [&](int M) {
3233 return M < 0 || (M >= 0 && M < HalfSize) ||
3234 (M >= MaskSize && M < MaskSize + HalfSize);
3235 }))
3236 preMask = HighLaneTy;
3237 else if (std::all_of(Mask.begin(), Mask.begin() + HalfSize, [&](int M) {
3238 return M < 0 || (M >= HalfSize && M < MaskSize) ||
3239 (M >= MaskSize + HalfSize && M < MaskSize * 2);
3240 }))
3241 preMask = LowLaneTy;
3242
3243 if (std::all_of(Mask.begin() + HalfSize, Mask.end(), [&](int M) {
3244 return M < 0 || (M >= HalfSize && M < MaskSize) ||
3245 (M >= MaskSize + HalfSize && M < MaskSize * 2);
3246 }))
3247 postMask = LowLaneTy;
3248 else if (std::all_of(Mask.begin() + HalfSize, Mask.end(), [&](int M) {
3249 return M < 0 || (M >= 0 && M < HalfSize) ||
3250 (M >= MaskSize && M < MaskSize + HalfSize);
3251 }))
3252 postMask = HighLaneTy;
3253
3254 // The pre-half of mask is high lane type, and the post-half of mask
3255 // is low lane type, which is closest to the LoongArch instructions.
3256 //
3257 // Note: In the LoongArch architecture, the high lane of mask corresponds
3258 // to the lower 128-bit of vector register, and the low lane of mask
3259 // corresponds the higher 128-bit of vector register.
3260 if (preMask == HighLaneTy && postMask == LowLaneTy) {
3261 return false;
3262 }
3263 if (preMask == LowLaneTy && postMask == HighLaneTy) {
3264 V1 = DAG.getBitcast(MVT::v4i64, V1);
3265 V1 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V1,
3266 DAG.getConstant(0b01001110, DL, GRLenVT));
3267 V1 = DAG.getBitcast(VT, V1);
3268
3269 if (!V2.isUndef()) {
3270 V2 = DAG.getBitcast(MVT::v4i64, V2);
3271 V2 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V2,
3272 DAG.getConstant(0b01001110, DL, GRLenVT));
3273 V2 = DAG.getBitcast(VT, V2);
3274 }
3275
3276 for (auto it = Mask.begin(); it < Mask.begin() + HalfSize; it++) {
3277 *it = *it < 0 ? *it : *it - HalfSize;
3278 }
3279 for (auto it = Mask.begin() + HalfSize; it < Mask.end(); it++) {
3280 *it = *it < 0 ? *it : *it + HalfSize;
3281 }
3282 } else if (preMask == LowLaneTy && postMask == LowLaneTy) {
3283 V1 = DAG.getBitcast(MVT::v4i64, V1);
3284 V1 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V1,
3285 DAG.getConstant(0b11101110, DL, GRLenVT));
3286 V1 = DAG.getBitcast(VT, V1);
3287
3288 if (!V2.isUndef()) {
3289 V2 = DAG.getBitcast(MVT::v4i64, V2);
3290 V2 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V2,
3291 DAG.getConstant(0b11101110, DL, GRLenVT));
3292 V2 = DAG.getBitcast(VT, V2);
3293 }
3294
3295 for (auto it = Mask.begin(); it < Mask.begin() + HalfSize; it++) {
3296 *it = *it < 0 ? *it : *it - HalfSize;
3297 }
3298 } else if (preMask == HighLaneTy && postMask == HighLaneTy) {
3299 V1 = DAG.getBitcast(MVT::v4i64, V1);
3300 V1 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V1,
3301 DAG.getConstant(0b01000100, DL, GRLenVT));
3302 V1 = DAG.getBitcast(VT, V1);
3303
3304 if (!V2.isUndef()) {
3305 V2 = DAG.getBitcast(MVT::v4i64, V2);
3306 V2 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V2,
3307 DAG.getConstant(0b01000100, DL, GRLenVT));
3308 V2 = DAG.getBitcast(VT, V2);
3309 }
3310
3311 for (auto it = Mask.begin() + HalfSize; it < Mask.end(); it++) {
3312 *it = *it < 0 ? *it : *it + HalfSize;
3313 }
3314 } else { // cross-lane
3315 return false;
3316 }
3317
3318 return true;
3319}
3320
3321/// Lower VECTOR_SHUFFLE as lane permute and then shuffle (if possible).
3322/// Only for 256-bit vector.
3323///
3324/// For example:
3325/// %2 = shufflevector <4 x i64> %0, <4 x i64> posion,
3326/// <4 x i64> <i32 0, i32 3, i32 2, i32 0>
3327/// is lowerded to:
3328/// (XVPERMI $xr2, $xr0, 78)
3329/// (XVSHUF $xr1, $xr2, $xr0)
3330/// (XVORI $xr0, $xr1, 0)
3332 ArrayRef<int> Mask,
3333 MVT VT, SDValue V1,
3334 SDValue V2,
3335 SelectionDAG &DAG) {
3336 assert(VT.is256BitVector() && "Only for 256-bit vector shuffles!");
3337 int Size = Mask.size();
3338 int LaneSize = Size / 2;
3339
3340 bool LaneCrossing[2] = {false, false};
3341 for (int i = 0; i < Size; ++i)
3342 if (Mask[i] >= 0 && ((Mask[i] % Size) / LaneSize) != (i / LaneSize))
3343 LaneCrossing[(Mask[i] % Size) / LaneSize] = true;
3344
3345 // Ensure that all lanes ared involved.
3346 if (!LaneCrossing[0] && !LaneCrossing[1])
3347 return SDValue();
3348
3349 SmallVector<int> InLaneMask;
3350 InLaneMask.assign(Mask.begin(), Mask.end());
3351 for (int i = 0; i < Size; ++i) {
3352 int &M = InLaneMask[i];
3353 if (M < 0)
3354 continue;
3355 if (((M % Size) / LaneSize) != (i / LaneSize))
3356 M = (M % LaneSize) + ((i / LaneSize) * LaneSize) + Size;
3357 }
3358
3359 SDValue Flipped = DAG.getBitcast(MVT::v4i64, V1);
3360 Flipped = DAG.getVectorShuffle(MVT::v4i64, DL, Flipped,
3361 DAG.getUNDEF(MVT::v4i64), {2, 3, 0, 1});
3362 Flipped = DAG.getBitcast(VT, Flipped);
3363 return DAG.getVectorShuffle(VT, DL, V1, Flipped, InLaneMask);
3364}
3365
3366/// Dispatching routine to lower various 256-bit LoongArch vector shuffles.
3367///
3368/// This routine breaks down the specific type of 256-bit shuffle and
3369/// dispatches to the lowering routines accordingly.
3371 SDValue V1, SDValue V2, SelectionDAG &DAG,
3372 const LoongArchSubtarget &Subtarget) {
3373 assert((VT.SimpleTy == MVT::v32i8 || VT.SimpleTy == MVT::v16i16 ||
3374 VT.SimpleTy == MVT::v8i32 || VT.SimpleTy == MVT::v4i64 ||
3375 VT.SimpleTy == MVT::v8f32 || VT.SimpleTy == MVT::v4f64) &&
3376 "Vector type is unsupported for lasx!");
3377 assert(V1.getSimpleValueType() == V2.getSimpleValueType() &&
3378 "Two operands have different types!");
3379 assert(VT.getVectorNumElements() == Mask.size() &&
3380 "Unexpected mask size for shuffle!");
3381 assert(Mask.size() % 2 == 0 && "Expected even mask size.");
3382 assert(Mask.size() >= 4 && "Mask size is less than 4.");
3383
3384 APInt KnownUndef, KnownZero;
3385 computeZeroableShuffleElements(Mask, V1, V2, KnownUndef, KnownZero);
3386 APInt Zeroable = KnownUndef | KnownZero;
3387
3388 SDValue Result;
3389 // TODO: Add more comparison patterns.
3390 if (V2.isUndef()) {
3391 if ((Result =
3392 lowerVECTOR_SHUFFLE_XVREPLVEI(DL, Mask, VT, V1, DAG, Subtarget)))
3393 return Result;
3394 if ((Result = lowerVECTOR_SHUFFLE_XVSHUF4I(DL, Mask, VT, V1, V2, DAG,
3395 Subtarget)))
3396 return Result;
3397 // Try to widen vectors to gain more optimization opportunities.
3398 if (SDValue NewShuffle = widenShuffleMask(DL, Mask, VT, V1, V2, DAG))
3399 return NewShuffle;
3400 if ((Result =
3401 lowerVECTOR_SHUFFLE_XVPERMI(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3402 return Result;
3403 if ((Result = lowerVECTOR_SHUFFLE_XVPERM(DL, Mask, VT, V1, DAG, Subtarget)))
3404 return Result;
3405 if ((Result =
3406 lowerVECTOR_SHUFFLE_IsReverse(DL, Mask, VT, V1, DAG, Subtarget)))
3407 return Result;
3408
3409 // TODO: This comment may be enabled in the future to better match the
3410 // pattern for instruction selection.
3411 /* V2 = V1; */
3412 }
3413
3414 // It is recommended not to change the pattern comparison order for better
3415 // performance.
3416 if ((Result = lowerVECTOR_SHUFFLE_XVPACKEV(DL, Mask, VT, V1, V2, DAG)))
3417 return Result;
3418 if ((Result = lowerVECTOR_SHUFFLE_XVPACKOD(DL, Mask, VT, V1, V2, DAG)))
3419 return Result;
3420 if ((Result = lowerVECTOR_SHUFFLE_XVILVH(DL, Mask, VT, V1, V2, DAG)))
3421 return Result;
3422 if ((Result = lowerVECTOR_SHUFFLE_XVILVL(DL, Mask, VT, V1, V2, DAG)))
3423 return Result;
3424 if ((Result = lowerVECTOR_SHUFFLE_XVPICKEV(DL, Mask, VT, V1, V2, DAG)))
3425 return Result;
3426 if ((Result = lowerVECTOR_SHUFFLE_XVPICKOD(DL, Mask, VT, V1, V2, DAG)))
3427 return Result;
3428 if ((VT.SimpleTy == MVT::v4i64 || VT.SimpleTy == MVT::v4f64) &&
3429 (Result =
3430 lowerVECTOR_SHUFFLE_XVSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3431 return Result;
3432 if ((Result =
3433 lowerVECTOR_SHUFFLE_XVEXTRINS(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3434 return Result;
3435 if ((Result = lowerVECTOR_SHUFFLEAsShift(DL, Mask, VT, V1, V2, DAG, Subtarget,
3436 Zeroable)))
3437 return Result;
3438 if ((Result =
3439 lowerVECTOR_SHUFFLE_XVPERMI(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3440 return Result;
3441 if ((Result =
3442 lowerVECTOR_SHUFFLE_XVINSVE0(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3443 return Result;
3444 if ((Result = lowerVECTOR_SHUFFLEAsByteRotate(DL, Mask, VT, V1, V2, DAG,
3445 Subtarget)))
3446 return Result;
3447
3448 // canonicalize non cross-lane shuffle vector
3449 SmallVector<int> NewMask(Mask);
3450 if (canonicalizeShuffleVectorByLane(DL, NewMask, VT, V1, V2, DAG, Subtarget))
3451 return lower256BitShuffle(DL, NewMask, VT, V1, V2, DAG, Subtarget);
3452
3453 // FIXME: Handling the remaining cases earlier can degrade performance
3454 // in some situations. Further analysis is required to enable more
3455 // effective optimizations.
3456 if (V2.isUndef()) {
3457 if ((Result = lowerVECTOR_SHUFFLEAsLanePermuteAndShuffle(DL, NewMask, VT,
3458 V1, V2, DAG)))
3459 return Result;
3460 }
3461
3462 if (SDValue NewShuffle = widenShuffleMask(DL, NewMask, VT, V1, V2, DAG))
3463 return NewShuffle;
3464 if ((Result = lowerVECTOR_SHUFFLE_XVSHUF(DL, NewMask, VT, V1, V2, DAG)))
3465 return Result;
3466
3467 return SDValue();
3468}
3469
3470SDValue LoongArchTargetLowering::lowerVECTOR_SHUFFLE(SDValue Op,
3471 SelectionDAG &DAG) const {
3472 ShuffleVectorSDNode *SVOp = cast<ShuffleVectorSDNode>(Op);
3473 ArrayRef<int> OrigMask = SVOp->getMask();
3474 SDValue V1 = Op.getOperand(0);
3475 SDValue V2 = Op.getOperand(1);
3476 MVT VT = Op.getSimpleValueType();
3477 int NumElements = VT.getVectorNumElements();
3478 SDLoc DL(Op);
3479
3480 bool V1IsUndef = V1.isUndef();
3481 bool V2IsUndef = V2.isUndef();
3482 if (V1IsUndef && V2IsUndef)
3483 return DAG.getUNDEF(VT);
3484
3485 // When we create a shuffle node we put the UNDEF node to second operand,
3486 // but in some cases the first operand may be transformed to UNDEF.
3487 // In this case we should just commute the node.
3488 if (V1IsUndef)
3489 return DAG.getCommutedVectorShuffle(*SVOp);
3490
3491 // Check for non-undef masks pointing at an undef vector and make the masks
3492 // undef as well. This makes it easier to match the shuffle based solely on
3493 // the mask.
3494 if (V2IsUndef &&
3495 any_of(OrigMask, [NumElements](int M) { return M >= NumElements; })) {
3496 SmallVector<int, 8> NewMask(OrigMask);
3497 for (int &M : NewMask)
3498 if (M >= NumElements)
3499 M = -1;
3500 return DAG.getVectorShuffle(VT, DL, V1, V2, NewMask);
3501 }
3502
3503 // Check for illegal shuffle mask element index values.
3504 int MaskUpperLimit = OrigMask.size() * (V2IsUndef ? 1 : 2);
3505 (void)MaskUpperLimit;
3506 assert(llvm::all_of(OrigMask,
3507 [&](int M) { return -1 <= M && M < MaskUpperLimit; }) &&
3508 "Out of bounds shuffle index");
3509
3510 // For each vector width, delegate to a specialized lowering routine.
3511 if (VT.is128BitVector())
3512 return lower128BitShuffle(DL, OrigMask, VT, V1, V2, DAG, Subtarget);
3513
3514 if (VT.is256BitVector())
3515 return lower256BitShuffle(DL, OrigMask, VT, V1, V2, DAG, Subtarget);
3516
3517 return SDValue();
3518}
3519
3520SDValue LoongArchTargetLowering::lowerFP_TO_FP16(SDValue Op,
3521 SelectionDAG &DAG) const {
3522 // Custom lower to ensure the libcall return is passed in an FPR on hard
3523 // float ABIs.
3524 SDLoc DL(Op);
3525 MakeLibCallOptions CallOptions;
3526 SDValue Op0 = Op.getOperand(0);
3527 SDValue Chain = SDValue();
3528 RTLIB::Libcall LC = RTLIB::getFPROUND(Op0.getValueType(), MVT::f16);
3529 SDValue Res;
3530 std::tie(Res, Chain) =
3531 makeLibCall(DAG, LC, MVT::f32, Op0, CallOptions, DL, Chain);
3532 if (Subtarget.is64Bit())
3533 return DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Res);
3534 return DAG.getBitcast(MVT::i32, Res);
3535}
3536
3537SDValue LoongArchTargetLowering::lowerFP16_TO_FP(SDValue Op,
3538 SelectionDAG &DAG) const {
3539 // Custom lower to ensure the libcall argument is passed in an FPR on hard
3540 // float ABIs.
3541 SDLoc DL(Op);
3542 MakeLibCallOptions CallOptions;
3543 SDValue Op0 = Op.getOperand(0);
3544 SDValue Chain = SDValue();
3545 SDValue Arg = Subtarget.is64Bit() ? DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64,
3546 DL, MVT::f32, Op0)
3547 : DAG.getBitcast(MVT::f32, Op0);
3548 SDValue Res;
3549 std::tie(Res, Chain) = makeLibCall(DAG, RTLIB::FPEXT_F16_F32, MVT::f32, Arg,
3550 CallOptions, DL, Chain);
3551 return Res;
3552}
3553
3554SDValue LoongArchTargetLowering::lowerFP_TO_BF16(SDValue Op,
3555 SelectionDAG &DAG) const {
3556 assert(Subtarget.hasBasicF() && "Unexpected custom legalization");
3557 SDLoc DL(Op);
3558 MakeLibCallOptions CallOptions;
3559 RTLIB::Libcall LC =
3560 RTLIB::getFPROUND(Op.getOperand(0).getValueType(), MVT::bf16);
3561 SDValue Res =
3562 makeLibCall(DAG, LC, MVT::f32, Op.getOperand(0), CallOptions, DL).first;
3563 if (Subtarget.is64Bit())
3564 return DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Res);
3565 return DAG.getBitcast(MVT::i32, Res);
3566}
3567
3568SDValue LoongArchTargetLowering::lowerBF16_TO_FP(SDValue Op,
3569 SelectionDAG &DAG) const {
3570 assert(Subtarget.hasBasicF() && "Unexpected custom legalization");
3571 MVT VT = Op.getSimpleValueType();
3572 SDLoc DL(Op);
3573 Op = DAG.getNode(
3574 ISD::SHL, DL, Op.getOperand(0).getValueType(), Op.getOperand(0),
3575 DAG.getShiftAmountConstant(16, Op.getOperand(0).getValueType(), DL));
3576 SDValue Res = Subtarget.is64Bit() ? DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64,
3577 DL, MVT::f32, Op)
3578 : DAG.getBitcast(MVT::f32, Op);
3579 if (VT != MVT::f32)
3580 return DAG.getNode(ISD::FP_EXTEND, DL, VT, Res);
3581 return Res;
3582}
3583
3584// Lower BUILD_VECTOR as broadcast load (if possible).
3585// For example:
3586// %a = load i8, ptr %ptr
3587// %b = build_vector %a, %a, %a, %a
3588// is lowered to :
3589// (VLDREPL_B $a0, 0)
3591 const SDLoc &DL,
3592 SelectionDAG &DAG) {
3593 MVT VT = BVOp->getSimpleValueType(0);
3594 int NumOps = BVOp->getNumOperands();
3595
3596 assert((VT.is128BitVector() || VT.is256BitVector()) &&
3597 "Unsupported vector type for broadcast.");
3598
3599 SDValue IdentitySrc;
3600 bool IsIdeneity = true;
3601
3602 for (int i = 0; i != NumOps; i++) {
3603 SDValue Op = BVOp->getOperand(i);
3604 if (Op.getOpcode() != ISD::LOAD || (IdentitySrc && Op != IdentitySrc)) {
3605 IsIdeneity = false;
3606 break;
3607 }
3608 IdentitySrc = BVOp->getOperand(0);
3609 }
3610
3611 // make sure that this load is valid and only has one user.
3612 if (!IsIdeneity || !IdentitySrc || !BVOp->isOnlyUserOf(IdentitySrc.getNode()))
3613 return SDValue();
3614
3615 auto *LN = cast<LoadSDNode>(IdentitySrc);
3616 auto ExtType = LN->getExtensionType();
3617
3618 if ((ExtType == ISD::EXTLOAD || ExtType == ISD::NON_EXTLOAD) &&
3619 VT.getScalarSizeInBits() == LN->getMemoryVT().getScalarSizeInBits()) {
3620 // Indexed loads and stores are not supported on LoongArch.
3621 assert(LN->isUnindexed() && "Unexpected indexed load.");
3622
3623 SDVTList Tys = DAG.getVTList(VT, MVT::Other);
3624 // The offset operand of unindexed load is always undefined, so there is
3625 // no need to pass it to VLDREPL.
3626 SDValue Ops[] = {LN->getChain(), LN->getBasePtr()};
3627 SDValue BCast = DAG.getNode(LoongArchISD::VLDREPL, DL, Tys, Ops);
3628 DAG.ReplaceAllUsesOfValueWith(SDValue(LN, 1), BCast.getValue(1));
3629 return BCast;
3630 }
3631 return SDValue();
3632}
3633
3634// Sequentially insert elements from Ops into Vector, from low to high indices.
3635// Note: Ops can have fewer elements than Vector.
3637 const LoongArchSubtarget &Subtarget, SDValue &Vector,
3638 EVT ResTy) {
3639 assert(Ops.size() <= ResTy.getVectorNumElements());
3640
3641 SDValue Op0 = Ops[0];
3642 if (!Op0.isUndef())
3643 Vector = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, ResTy, Op0);
3644 for (unsigned i = 1; i < Ops.size(); ++i) {
3645 SDValue Opi = Ops[i];
3646 if (Opi.isUndef())
3647 continue;
3648 Vector = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, ResTy, Vector, Opi,
3649 DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
3650 }
3651}
3652
3653// Build a ResTy subvector from Node, taking NumElts elements starting at index
3654// 'first'.
3656 SelectionDAG &DAG, SDLoc DL,
3657 const LoongArchSubtarget &Subtarget,
3658 EVT ResTy, unsigned first) {
3659 unsigned NumElts = ResTy.getVectorNumElements();
3660
3661 assert(first + NumElts <= Node->getSimpleValueType(0).getVectorNumElements());
3662
3663 SmallVector<SDValue, 16> Ops(Node->op_begin() + first,
3664 Node->op_begin() + first + NumElts);
3665 SDValue Vector = DAG.getUNDEF(ResTy);
3666 fillVector(Ops, DAG, DL, Subtarget, Vector, ResTy);
3667 return Vector;
3668}
3669
3670SDValue LoongArchTargetLowering::lowerBUILD_VECTOR(SDValue Op,
3671 SelectionDAG &DAG) const {
3672 BuildVectorSDNode *Node = cast<BuildVectorSDNode>(Op);
3673 MVT VT = Node->getSimpleValueType(0);
3674 EVT ResTy = Op->getValueType(0);
3675 unsigned NumElts = ResTy.getVectorNumElements();
3676 SDLoc DL(Op);
3677 APInt SplatValue, SplatUndef;
3678 unsigned SplatBitSize;
3679 bool HasAnyUndefs;
3680 bool IsConstant = false;
3681 bool UseSameConstant = true;
3682 SDValue ConstantValue;
3683 bool Is128Vec = ResTy.is128BitVector();
3684 bool Is256Vec = ResTy.is256BitVector();
3685
3686 if ((!Subtarget.hasExtLSX() || !Is128Vec) &&
3687 (!Subtarget.hasExtLASX() || !Is256Vec))
3688 return SDValue();
3689
3690 if (SDValue Result = lowerBUILD_VECTORAsBroadCastLoad(Node, DL, DAG))
3691 return Result;
3692
3693 if (Node->isConstantSplat(SplatValue, SplatUndef, SplatBitSize, HasAnyUndefs,
3694 /*MinSplatBits=*/8) &&
3695 SplatBitSize <= 64) {
3696 // We can only cope with 8, 16, 32, or 64-bit elements.
3697 if (SplatBitSize != 8 && SplatBitSize != 16 && SplatBitSize != 32 &&
3698 SplatBitSize != 64)
3699 return SDValue();
3700
3701 if (SplatBitSize == 64 && !Subtarget.is64Bit()) {
3702 // We can only handle 64-bit elements that are within
3703 // the signed 10-bit range or match vldi patterns on 32-bit targets.
3704 // See the BUILD_VECTOR case in LoongArchDAGToDAGISel::Select().
3705 if (!SplatValue.isSignedIntN(10) &&
3706 !isImmVLDILegalForMode1(SplatValue, SplatBitSize).first)
3707 return SDValue();
3708 if ((Is128Vec && ResTy == MVT::v4i32) ||
3709 (Is256Vec && ResTy == MVT::v8i32))
3710 return Op;
3711 }
3712
3713 EVT ViaVecTy;
3714
3715 switch (SplatBitSize) {
3716 default:
3717 return SDValue();
3718 case 8:
3719 ViaVecTy = Is128Vec ? MVT::v16i8 : MVT::v32i8;
3720 break;
3721 case 16:
3722 ViaVecTy = Is128Vec ? MVT::v8i16 : MVT::v16i16;
3723 break;
3724 case 32:
3725 ViaVecTy = Is128Vec ? MVT::v4i32 : MVT::v8i32;
3726 break;
3727 case 64:
3728 ViaVecTy = Is128Vec ? MVT::v2i64 : MVT::v4i64;
3729 break;
3730 }
3731
3732 // SelectionDAG::getConstant will promote SplatValue appropriately.
3733 SDValue Result = DAG.getConstant(SplatValue, DL, ViaVecTy);
3734
3735 // Bitcast to the type we originally wanted.
3736 if (ViaVecTy != ResTy)
3737 Result = DAG.getNode(ISD::BITCAST, SDLoc(Node), ResTy, Result);
3738
3739 return Result;
3740 }
3741
3742 if (DAG.isSplatValue(Op, /*AllowUndefs=*/false))
3743 return Op;
3744
3745 for (unsigned i = 0; i < NumElts; ++i) {
3746 SDValue Opi = Node->getOperand(i);
3747 if (isIntOrFPConstant(Opi)) {
3748 IsConstant = true;
3749 if (!ConstantValue.getNode())
3750 ConstantValue = Opi;
3751 else if (ConstantValue != Opi)
3752 UseSameConstant = false;
3753 }
3754 }
3755
3756 // If the type of BUILD_VECTOR is v2f64, custom legalizing it has no benefits.
3757 if (IsConstant && UseSameConstant && ResTy != MVT::v2f64) {
3758 SDValue Result = DAG.getSplatBuildVector(ResTy, DL, ConstantValue);
3759 for (unsigned i = 0; i < NumElts; ++i) {
3760 SDValue Opi = Node->getOperand(i);
3761 if (!isIntOrFPConstant(Opi))
3762 Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, ResTy, Result, Opi,
3763 DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
3764 }
3765 return Result;
3766 }
3767
3768 if (!IsConstant) {
3769 // If the BUILD_VECTOR has a repeated pattern, use INSERT_VECTOR_ELT to fill
3770 // the sub-sequence of the vector and then broadcast the sub-sequence.
3771 //
3772 // TODO: If the BUILD_VECTOR contains undef elements, consider falling
3773 // back to use INSERT_VECTOR_ELT to materialize the vector, because it
3774 // generates worse code in some cases. This could be further optimized
3775 // with more consideration.
3777 BitVector UndefElements;
3778 if (Node->getRepeatedSequence(Sequence, &UndefElements) &&
3779 UndefElements.count() == 0) {
3780 // Using LSX instructions to fill the sub-sequence of 256-bits vector,
3781 // because the high part can be simply treated as undef.
3782 SDValue Vector = DAG.getUNDEF(ResTy);
3783 EVT FillTy = Is256Vec
3785 : ResTy;
3786 SDValue FillVec =
3787 Is256Vec ? DAG.getExtractSubvector(DL, FillTy, Vector, 0) : Vector;
3788
3789 fillVector(Sequence, DAG, DL, Subtarget, FillVec, FillTy);
3790
3791 unsigned SeqLen = Sequence.size();
3792 unsigned SplatLen = NumElts / SeqLen;
3793 MVT SplatEltTy = MVT::getIntegerVT(VT.getScalarSizeInBits() * SeqLen);
3794 MVT SplatTy = MVT::getVectorVT(SplatEltTy, SplatLen);
3795
3796 // If size of the sub-sequence is half of a 256-bits vector, bitcast the
3797 // vector to v4i64 type in order to match the pattern of XVREPLVE0Q.
3798 if (SplatEltTy == MVT::i128)
3799 SplatTy = MVT::v4i64;
3800
3801 SDValue SplatVec;
3802 SDValue SrcVec = DAG.getBitcast(
3803 SplatTy,
3804 Is256Vec ? DAG.getInsertSubvector(DL, Vector, FillVec, 0) : FillVec);
3805 if (Is256Vec) {
3806 SplatVec =
3807 DAG.getNode((SplatEltTy == MVT::i128) ? LoongArchISD::XVREPLVE0Q
3808 : LoongArchISD::XVREPLVE0,
3809 DL, SplatTy, SrcVec);
3810 } else {
3811 SplatVec = DAG.getNode(LoongArchISD::VREPLVEI, DL, SplatTy, SrcVec,
3812 DAG.getConstant(0, DL, Subtarget.getGRLenVT()));
3813 }
3814
3815 return DAG.getBitcast(ResTy, SplatVec);
3816 }
3817
3818 // Use INSERT_VECTOR_ELT operations rather than expand to stores, because
3819 // using memory operations is much lower.
3820 //
3821 // For 256-bit vectors, normally split into two halves and concatenate.
3822 // Special case: for v8i32/v8f32/v4i64/v4f64, if the upper half has only
3823 // one non-undef element, skip spliting to avoid a worse result.
3824 if (ResTy == MVT::v8i32 || ResTy == MVT::v8f32 || ResTy == MVT::v4i64 ||
3825 ResTy == MVT::v4f64) {
3826 unsigned NonUndefCount = 0;
3827 for (unsigned i = NumElts / 2; i < NumElts; ++i) {
3828 if (!Node->getOperand(i).isUndef()) {
3829 ++NonUndefCount;
3830 if (NonUndefCount > 1)
3831 break;
3832 }
3833 }
3834 if (NonUndefCount == 1)
3835 return fillSubVectorFromBuildVector(Node, DAG, DL, Subtarget, ResTy, 0);
3836 }
3837
3838 EVT VecTy =
3839 Is256Vec ? ResTy.getHalfNumVectorElementsVT(*DAG.getContext()) : ResTy;
3840 SDValue Vector =
3841 fillSubVectorFromBuildVector(Node, DAG, DL, Subtarget, VecTy, 0);
3842
3843 if (Is128Vec)
3844 return Vector;
3845
3846 SDValue VectorHi = fillSubVectorFromBuildVector(Node, DAG, DL, Subtarget,
3847 VecTy, NumElts / 2);
3848
3849 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ResTy, Vector, VectorHi);
3850 }
3851
3852 return SDValue();
3853}
3854
3855SDValue LoongArchTargetLowering::lowerCONCAT_VECTORS(SDValue Op,
3856 SelectionDAG &DAG) const {
3857 SDLoc DL(Op);
3858 MVT ResVT = Op.getSimpleValueType();
3859 assert(ResVT.is256BitVector() && Op.getNumOperands() == 2);
3860
3861 if (Op.getOperand(0).getOpcode() == ISD::TRUNCATE &&
3862 Op.getOperand(1).getOpcode() == ISD::TRUNCATE)
3863 return Op;
3864
3865 unsigned NumOperands = Op.getNumOperands();
3866 unsigned NumFreezeUndef = 0;
3867 unsigned NumZero = 0;
3868 unsigned NumNonZero = 0;
3869 unsigned NonZeros = 0;
3870 SmallSet<SDValue, 4> Undefs;
3871 for (unsigned i = 0; i != NumOperands; ++i) {
3872 SDValue SubVec = Op.getOperand(i);
3873 if (SubVec.isUndef())
3874 continue;
3875 if (ISD::isFreezeUndef(SubVec.getNode())) {
3876 // If the freeze(undef) has multiple uses then we must fold to zero.
3877 if (SubVec.hasOneUse()) {
3878 ++NumFreezeUndef;
3879 } else {
3880 ++NumZero;
3881 Undefs.insert(SubVec);
3882 }
3883 } else if (ISD::isBuildVectorAllZeros(SubVec.getNode()))
3884 ++NumZero;
3885 else {
3886 assert(i < sizeof(NonZeros) * CHAR_BIT); // Ensure the shift is in range.
3887 NonZeros |= 1 << i;
3888 ++NumNonZero;
3889 }
3890 }
3891
3892 // If we have more than 2 non-zeros, build each half separately.
3893 if (NumNonZero > 2) {
3894 MVT HalfVT = ResVT.getHalfNumVectorElementsVT();
3895 ArrayRef<SDUse> Ops = Op->ops();
3896 SDValue Lo = DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
3897 Ops.slice(0, NumOperands / 2));
3898 SDValue Hi = DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
3899 Ops.slice(NumOperands / 2));
3900 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ResVT, Lo, Hi);
3901 }
3902
3903 // Otherwise, build it up through insert_subvectors.
3904 SDValue Vec = NumZero ? DAG.getConstant(0, DL, ResVT)
3905 : (NumFreezeUndef ? DAG.getFreeze(DAG.getUNDEF(ResVT))
3906 : DAG.getUNDEF(ResVT));
3907
3908 // Replace Undef operands with ZeroVector.
3909 for (SDValue U : Undefs)
3910 DAG.ReplaceAllUsesWith(U, DAG.getConstant(0, DL, U.getSimpleValueType()));
3911
3912 MVT SubVT = Op.getOperand(0).getSimpleValueType();
3913 unsigned NumSubElems = SubVT.getVectorNumElements();
3914 for (unsigned i = 0; i != NumOperands; ++i) {
3915 if ((NonZeros & (1 << i)) == 0)
3916 continue;
3917
3918 Vec = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, ResVT, Vec, Op.getOperand(i),
3919 DAG.getVectorIdxConstant(i * NumSubElems, DL));
3920 }
3921
3922 return Vec;
3923}
3924
3925SDValue
3926LoongArchTargetLowering::lowerEXTRACT_VECTOR_ELT(SDValue Op,
3927 SelectionDAG &DAG) const {
3928 MVT EltVT = Op.getSimpleValueType();
3929 SDValue Vec = Op->getOperand(0);
3930 EVT VecTy = Vec->getValueType(0);
3931 SDValue Idx = Op->getOperand(1);
3932 SDLoc DL(Op);
3933 MVT GRLenVT = Subtarget.getGRLenVT();
3934
3935 assert(VecTy.is256BitVector() && "Unexpected EXTRACT_VECTOR_ELT vector type");
3936
3937 if (isa<ConstantSDNode>(Idx))
3938 return Op;
3939
3940 switch (VecTy.getSimpleVT().SimpleTy) {
3941 default:
3942 llvm_unreachable("Unexpected type");
3943 case MVT::v32i8:
3944 case MVT::v16i16:
3945 case MVT::v4i64:
3946 case MVT::v4f64: {
3947 // Extract the high half subvector and place it to the low half of a new
3948 // vector. It doesn't matter what the high half of the new vector is.
3949 EVT HalfTy = VecTy.getHalfNumVectorElementsVT(*DAG.getContext());
3950 SDValue VecHi =
3951 DAG.getExtractSubvector(DL, HalfTy, Vec, HalfTy.getVectorNumElements());
3952 SDValue TmpVec =
3953 DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VecTy, DAG.getUNDEF(VecTy),
3954 VecHi, DAG.getConstant(0, DL, GRLenVT));
3955
3956 // Shuffle the origin Vec and the TmpVec using MaskVec, the lowest element
3957 // of MaskVec is Idx, the rest do not matter. ResVec[0] will hold the
3958 // desired element.
3959 SDValue IdxCp =
3960 Subtarget.is64Bit()
3961 ? DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64, DL, MVT::f32, Idx)
3962 : DAG.getBitcast(MVT::f32, Idx);
3963 SDValue IdxVec = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v8f32, IdxCp);
3964 SDValue MaskVec =
3965 DAG.getBitcast((VecTy == MVT::v4f64) ? MVT::v4i64 : VecTy, IdxVec);
3966 SDValue ResVec =
3967 DAG.getNode(LoongArchISD::VSHUF, DL, VecTy, MaskVec, TmpVec, Vec);
3968
3969 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, ResVec,
3970 DAG.getConstant(0, DL, GRLenVT));
3971 }
3972 case MVT::v8i32:
3973 case MVT::v8f32: {
3974 SDValue SplatIdx = DAG.getSplatBuildVector(MVT::v8i32, DL, Idx);
3975 SDValue SplatValue =
3976 DAG.getNode(LoongArchISD::XVPERM, DL, VecTy, Vec, SplatIdx);
3977
3978 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, SplatValue,
3979 DAG.getConstant(0, DL, GRLenVT));
3980 }
3981 }
3982}
3983
3984SDValue
3985LoongArchTargetLowering::lowerINSERT_VECTOR_ELT(SDValue Op,
3986 SelectionDAG &DAG) const {
3987 MVT VT = Op.getSimpleValueType();
3988 MVT EltVT = VT.getVectorElementType();
3989 unsigned NumElts = VT.getVectorNumElements();
3990 unsigned EltSizeInBits = EltVT.getScalarSizeInBits();
3991 SDLoc DL(Op);
3992 SDValue Op0 = Op.getOperand(0);
3993 SDValue Op1 = Op.getOperand(1);
3994 SDValue Op2 = Op.getOperand(2);
3995
3996 if (isa<ConstantSDNode>(Op2))
3997 return Op;
3998
3999 MVT IdxTy = MVT::getIntegerVT(EltSizeInBits);
4000 MVT IdxVTy = MVT::getVectorVT(IdxTy, NumElts);
4001
4002 if (!isTypeLegal(VT) || !isTypeLegal(IdxVTy))
4003 return SDValue();
4004
4005 SDValue SplatElt = DAG.getSplatBuildVector(VT, DL, Op1);
4006 SmallVector<SDValue, 32> RawIndices;
4007 SDValue SplatIdx;
4008 SDValue Indices;
4009
4010 if (!Subtarget.is64Bit() && IdxTy == MVT::i64) {
4011 MVT PairVTy = MVT::getVectorVT(MVT::i32, NumElts * 2);
4012 for (unsigned i = 0; i < NumElts; ++i) {
4013 RawIndices.push_back(Op2);
4014 RawIndices.push_back(DAG.getConstant(0, DL, MVT::i32));
4015 }
4016 SplatIdx = DAG.getBuildVector(PairVTy, DL, RawIndices);
4017 SplatIdx = DAG.getBitcast(IdxVTy, SplatIdx);
4018
4019 RawIndices.clear();
4020 for (unsigned i = 0; i < NumElts; ++i) {
4021 RawIndices.push_back(DAG.getConstant(i, DL, MVT::i32));
4022 RawIndices.push_back(DAG.getConstant(0, DL, MVT::i32));
4023 }
4024 Indices = DAG.getBuildVector(PairVTy, DL, RawIndices);
4025 Indices = DAG.getBitcast(IdxVTy, Indices);
4026 } else {
4027 SplatIdx = DAG.getSplatBuildVector(IdxVTy, DL, Op2);
4028
4029 for (unsigned i = 0; i < NumElts; ++i)
4030 RawIndices.push_back(DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
4031 Indices = DAG.getBuildVector(IdxVTy, DL, RawIndices);
4032 }
4033
4034 // insert vec, elt, idx
4035 // =>
4036 // select (splatidx == {0,1,2...}) ? splatelt : vec
4037 SDValue SelectCC =
4038 DAG.getSetCC(DL, IdxVTy, SplatIdx, Indices, ISD::CondCode::SETEQ);
4039 return DAG.getNode(ISD::VSELECT, DL, VT, SelectCC, SplatElt, Op0);
4040}
4041
4042SDValue LoongArchTargetLowering::lowerATOMIC_FENCE(SDValue Op,
4043 SelectionDAG &DAG) const {
4044 SDLoc DL(Op);
4045 SyncScope::ID FenceSSID =
4046 static_cast<SyncScope::ID>(Op.getConstantOperandVal(2));
4047
4048 // singlethread fences only synchronize with signal handlers on the same
4049 // thread and thus only need to preserve instruction order, not actually
4050 // enforce memory ordering.
4051 if (FenceSSID == SyncScope::SingleThread)
4052 // MEMBARRIER is a compiler barrier; it codegens to a no-op.
4053 return DAG.getNode(ISD::MEMBARRIER, DL, MVT::Other, Op.getOperand(0));
4054
4055 return Op;
4056}
4057
4059 MVT GRLenVT, SDValue RMValue) {
4060 // LLVM rounding mode encoding differs from LoongArch FCSR encoding:
4061 // LLVM: 0=RTZ, 1=RNE, 2=RUP, 3=RDN
4062 // FCSR: 0=RNE, 1=RZ, 2=RP, 3=RN
4063 //
4064 // The conversion swaps encodings 0 and 1 while preserving 2 and 3.
4065 // Since the transformation is self-inverse, it applies in both directions:
4066 // LLVM RM <-> LoongArch FCSR RM
4067 //
4068 // Transformation: RM ^ (~(RM >> 1) & 1)
4069 SDValue ShiftRight1 = DAG.getNode(ISD::SRL, DL, GRLenVT, RMValue,
4070 DAG.getConstant(1, DL, GRLenVT));
4071
4072 SDValue SwapMask = DAG.getNode(ISD::AND, DL, GRLenVT,
4073 DAG.getNode(ISD::XOR, DL, GRLenVT, ShiftRight1,
4074 DAG.getConstant(1, DL, GRLenVT)),
4075 DAG.getConstant(1, DL, GRLenVT));
4076
4077 return DAG.getNode(ISD::XOR, DL, GRLenVT, RMValue, SwapMask);
4078}
4079
4080SDValue LoongArchTargetLowering::lowerSET_ROUNDING(SDValue Op,
4081 SelectionDAG &DAG) const {
4082 MVT GRLenVT = Subtarget.getGRLenVT();
4083 SDLoc DL(Op);
4084 SDValue Chain = Op.getOperand(0);
4085 SDValue RMValue = Op.getOperand(1);
4086
4087 if (auto *CVal = dyn_cast<ConstantSDNode>(RMValue)) {
4088 uint64_t RM = CVal->getZExtValue();
4089 if (RM > 3) {
4091 LLVMContext &C = MF.getFunction().getContext();
4092 C.diagnose(DiagnosticInfoUnsupported(
4093 MF.getFunction(),
4094 "rounding mode is not supported by LoongArch hardware",
4095 DiagnosticLocation(DL.getDebugLoc()), DS_Error));
4096 return Chain;
4097 }
4098 }
4099
4100 RMValue = DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT, RMValue);
4101 RMValue = convertRMEncoding(DAG, DL, GRLenVT, RMValue);
4102
4103 // The RM field in FCSR is at bits [9:8]. Shift the rounding mode value
4104 // into position before writing via WRFCSR.
4105 RMValue = DAG.getNode(ISD::SHL, DL, GRLenVT, RMValue,
4106 DAG.getConstant(8, DL, GRLenVT));
4107
4108 // FCSR3 is an alias of the RM field; writing it avoids clobbering
4109 // unrelated fields in FCSR0.
4110 SDValue FCSRNo = DAG.getTargetConstant(3, DL, GRLenVT);
4111 MachineSDNode *RN = DAG.getMachineNode(LoongArch::WRFCSR, DL, MVT::Other,
4112 FCSRNo, RMValue, Chain);
4113 return SDValue(RN, 0);
4114}
4115
4116SDValue LoongArchTargetLowering::lowerGET_ROUNDING(SDValue Op,
4117 SelectionDAG &DAG) const {
4118 MVT GRLenVT = Subtarget.getGRLenVT();
4119 SDLoc DL(Op);
4120 SDValue Chain = Op->getOperand(0);
4121
4122 // FCSR3 is an alias of the RM field.
4123 SDValue FCSRNo = DAG.getTargetConstant(3, DL, GRLenVT);
4124 MachineSDNode *FCSR = DAG.getMachineNode(LoongArch::RDFCSR, DL, GRLenVT,
4125 MVT::Other, FCSRNo, Chain);
4126 SDValue RMValue = SDValue(FCSR, 0);
4127 Chain = SDValue(FCSR, 1);
4128
4129 // The RM field in FCSR is at bits [9:8].
4130 RMValue = DAG.getNode(ISD::SRL, DL, GRLenVT, RMValue,
4131 DAG.getConstant(8, DL, GRLenVT));
4132 RMValue = convertRMEncoding(DAG, DL, GRLenVT, RMValue);
4133
4134 SDValue RetVal = DAG.getZExtOrTrunc(RMValue, DL, Op.getValueType());
4135 return DAG.getMergeValues({RetVal, Chain}, DL);
4136}
4137
4138SDValue LoongArchTargetLowering::lowerWRITE_REGISTER(SDValue Op,
4139 SelectionDAG &DAG) const {
4140
4141 if (Subtarget.is64Bit() && Op.getOperand(2).getValueType() == MVT::i32) {
4142 DAG.getContext()->emitError(
4143 "On LA64, only 64-bit registers can be written.");
4144 return Op.getOperand(0);
4145 }
4146
4147 if (!Subtarget.is64Bit() && Op.getOperand(2).getValueType() == MVT::i64) {
4148 DAG.getContext()->emitError(
4149 "On LA32, only 32-bit registers can be written.");
4150 return Op.getOperand(0);
4151 }
4152
4153 return Op;
4154}
4155
4156SDValue LoongArchTargetLowering::lowerFRAMEADDR(SDValue Op,
4157 SelectionDAG &DAG) const {
4158 if (!isa<ConstantSDNode>(Op.getOperand(0))) {
4159 DAG.getContext()->emitError("argument to '__builtin_frame_address' must "
4160 "be a constant integer");
4161 return SDValue();
4162 }
4163
4166 Register FrameReg = Subtarget.getRegisterInfo()->getFrameRegister(MF);
4167 EVT VT = Op.getValueType();
4168 SDLoc DL(Op);
4169 SDValue FrameAddr = DAG.getCopyFromReg(DAG.getEntryNode(), DL, FrameReg, VT);
4170 unsigned Depth = Op.getConstantOperandVal(0);
4171 int GRLenInBytes = Subtarget.getGRLen() / 8;
4172
4173 while (Depth--) {
4174 int Offset = -(GRLenInBytes * 2);
4175 SDValue Ptr = DAG.getNode(ISD::ADD, DL, VT, FrameAddr,
4176 DAG.getSignedConstant(Offset, DL, VT));
4177 FrameAddr =
4178 DAG.getLoad(VT, DL, DAG.getEntryNode(), Ptr, MachinePointerInfo());
4179 }
4180 return FrameAddr;
4181}
4182
4183SDValue LoongArchTargetLowering::lowerRETURNADDR(SDValue Op,
4184 SelectionDAG &DAG) const {
4185 // Currently only support lowering return address for current frame.
4186 if (Op.getConstantOperandVal(0) != 0) {
4187 DAG.getContext()->emitError(
4188 "return address can only be determined for the current frame");
4189 return SDValue();
4190 }
4191
4194 MVT GRLenVT = Subtarget.getGRLenVT();
4195
4196 // Return the value of the return address register, marking it an implicit
4197 // live-in.
4198 Register Reg = MF.addLiveIn(Subtarget.getRegisterInfo()->getRARegister(),
4199 getRegClassFor(GRLenVT));
4200 return DAG.getCopyFromReg(DAG.getEntryNode(), SDLoc(Op), Reg, GRLenVT);
4201}
4202
4203SDValue LoongArchTargetLowering::lowerEH_DWARF_CFA(SDValue Op,
4204 SelectionDAG &DAG) const {
4206 auto Size = Subtarget.getGRLen() / 8;
4207 auto FI = MF.getFrameInfo().CreateFixedObject(Size, 0, false);
4208 return DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
4209}
4210
4211SDValue LoongArchTargetLowering::lowerVASTART(SDValue Op,
4212 SelectionDAG &DAG) const {
4214 auto *FuncInfo = MF.getInfo<LoongArchMachineFunctionInfo>();
4215
4216 SDLoc DL(Op);
4217 SDValue FI = DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(),
4219
4220 // vastart just stores the address of the VarArgsFrameIndex slot into the
4221 // memory location argument.
4222 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
4223 return DAG.getStore(Op.getOperand(0), DL, FI, Op.getOperand(1),
4224 MachinePointerInfo(SV));
4225}
4226
4227SDValue LoongArchTargetLowering::lowerUINT_TO_FP(SDValue Op,
4228 SelectionDAG &DAG) const {
4229 SDLoc DL(Op);
4230 SDValue Op0 = Op.getOperand(0);
4231 EVT VT = Op.getValueType();
4232 EVT Op0VT = Op0.getValueType();
4233
4234 if (VT.isVector()) {
4235 if (VT.getScalarSizeInBits() != Op0VT.getScalarSizeInBits())
4236 return SDValue();
4237 return Op;
4238 }
4239
4240 if ((DAG.SignBitIsZero(Op0) || Op->getFlags().hasNonNeg()) &&
4243 return DAG.getNode(ISD::SINT_TO_FP, DL, VT, Op0);
4244
4245 // We can't do uint64 -> double -> float because of double-rounding issue.
4246 if (Subtarget.hasExtLSX() && Op0VT == MVT::i64 && VT == MVT::f64) {
4247 Op0 = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2i64, Op0);
4248 SDValue Conv = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::v2f64, Op0);
4249 Conv = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f64, Conv,
4250 DAG.getIntPtrConstant(0, DL));
4251 return Conv;
4252 }
4253
4254 if (!Subtarget.is64Bit() || !Subtarget.hasBasicF() || Subtarget.hasBasicD())
4255 return SDValue();
4256
4257 assert(Subtarget.is64Bit() && Subtarget.hasBasicF() &&
4258 !Subtarget.hasBasicD() && "unexpected target features");
4259
4260 if (Op0->getOpcode() == ISD::AND) {
4261 auto *C = dyn_cast<ConstantSDNode>(Op0.getOperand(1));
4262 if (C && C->getZExtValue() < UINT64_C(0xFFFFFFFF))
4263 return Op;
4264 }
4265
4266 if (Op0->getOpcode() == LoongArchISD::BSTRPICK &&
4267 Op0.getConstantOperandVal(1) < UINT64_C(0X1F) &&
4268 Op0.getConstantOperandVal(2) == UINT64_C(0))
4269 return Op;
4270
4271 if (Op0.getOpcode() == ISD::AssertZext &&
4272 dyn_cast<VTSDNode>(Op0.getOperand(1))->getVT().bitsLT(MVT::i32))
4273 return Op;
4274
4275 EVT OpVT = Op0.getValueType();
4276 EVT RetVT = Op.getValueType();
4277 RTLIB::Libcall LC = RTLIB::getUINTTOFP(OpVT, RetVT);
4278 MakeLibCallOptions CallOptions;
4279 CallOptions.setTypeListBeforeSoften(OpVT, RetVT);
4280 SDValue Chain = SDValue();
4282 std::tie(Result, Chain) =
4283 makeLibCall(DAG, LC, Op.getValueType(), Op0, CallOptions, DL, Chain);
4284 return Result;
4285}
4286
4287SDValue LoongArchTargetLowering::lowerSINT_TO_FP(SDValue Op,
4288 SelectionDAG &DAG) const {
4289 assert(Subtarget.is64Bit() && Subtarget.hasBasicF() &&
4290 !Subtarget.hasBasicD() && "unexpected target features");
4291
4292 SDLoc DL(Op);
4293 SDValue Op0 = Op.getOperand(0);
4294
4295 if ((Op0.getOpcode() == ISD::AssertSext ||
4297 dyn_cast<VTSDNode>(Op0.getOperand(1))->getVT().bitsLE(MVT::i32))
4298 return Op;
4299
4300 EVT OpVT = Op0.getValueType();
4301 EVT RetVT = Op.getValueType();
4302 RTLIB::Libcall LC = RTLIB::getSINTTOFP(OpVT, RetVT);
4303 MakeLibCallOptions CallOptions;
4304 CallOptions.setTypeListBeforeSoften(OpVT, RetVT);
4305 SDValue Chain = SDValue();
4307 std::tie(Result, Chain) =
4308 makeLibCall(DAG, LC, Op.getValueType(), Op0, CallOptions, DL, Chain);
4309 return Result;
4310}
4311
4312SDValue LoongArchTargetLowering::lowerBITCAST(SDValue Op,
4313 SelectionDAG &DAG) const {
4314
4315 SDLoc DL(Op);
4316 EVT VT = Op.getValueType();
4317 SDValue Op0 = Op.getOperand(0);
4318 EVT Op0VT = Op0.getValueType();
4319
4320 if (Op.getValueType() == MVT::f32 && Op0VT == MVT::i32 &&
4321 Subtarget.is64Bit() && Subtarget.hasBasicF()) {
4322 SDValue NewOp0 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op0);
4323 return DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64, DL, MVT::f32, NewOp0);
4324 }
4325 if (VT == MVT::f64 && Op0VT == MVT::i64 && !Subtarget.is64Bit()) {
4326 SDValue Lo, Hi;
4327 std::tie(Lo, Hi) = DAG.SplitScalar(Op0, DL, MVT::i32, MVT::i32);
4328 return DAG.getNode(LoongArchISD::BUILD_PAIR_F64, DL, MVT::f64, Lo, Hi);
4329 }
4330 return Op;
4331}
4332
4333SDValue LoongArchTargetLowering::lowerFP_TO_SINT(SDValue Op,
4334 SelectionDAG &DAG) const {
4335
4336 SDLoc DL(Op);
4337 SDValue Op0 = Op.getOperand(0);
4338
4339 if (Op0.getValueType() == MVT::f16)
4340 Op0 = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Op0);
4341
4342 if (Op.getValueSizeInBits() > 32 && Subtarget.hasBasicF() &&
4343 !Subtarget.hasBasicD()) {
4344 SDValue Dst = DAG.getNode(LoongArchISD::FTINT, DL, MVT::f32, Op0);
4345 return DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Dst);
4346 }
4347
4348 EVT FPTy = EVT::getFloatingPointVT(Op.getValueSizeInBits());
4349 SDValue Trunc = DAG.getNode(LoongArchISD::FTINT, DL, FPTy, Op0);
4350 return DAG.getNode(ISD::BITCAST, DL, Op.getValueType(), Trunc);
4351}
4352
4353SDValue LoongArchTargetLowering::lowerFP_TO_UINT(SDValue Op,
4354 SelectionDAG &DAG) const {
4355 if (!Subtarget.hasExtLSX())
4356 return SDValue();
4357
4358 SDLoc DL(Op);
4359 SDValue Src = Op.getOperand(0);
4360 EVT VT = Op.getValueType();
4361 EVT SrcVT = Src.getValueType();
4362
4363 if (VT != MVT::i64)
4364 return SDValue();
4365
4366 if (SrcVT != MVT::f32 && SrcVT != MVT::f64)
4367 return SDValue();
4368
4369 if (SrcVT == MVT::f32)
4370 Src = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f64, Src);
4371 Src = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2f64, Src);
4372 SDValue Conv = DAG.getNode(ISD::FP_TO_UINT, DL, MVT::v2i64, Src);
4373 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Conv,
4374 DAG.getIntPtrConstant(0, DL));
4375}
4376
4378 SelectionDAG &DAG, unsigned Flags) {
4379 return DAG.getTargetGlobalAddress(N->getGlobal(), DL, Ty, 0, Flags);
4380}
4381
4383 SelectionDAG &DAG, unsigned Flags) {
4384 return DAG.getTargetBlockAddress(N->getBlockAddress(), Ty, N->getOffset(),
4385 Flags);
4386}
4387
4389 SelectionDAG &DAG, unsigned Flags) {
4390 return DAG.getTargetConstantPool(N->getConstVal(), Ty, N->getAlign(),
4391 N->getOffset(), Flags);
4392}
4393
4395 SelectionDAG &DAG, unsigned Flags) {
4396 return DAG.getTargetJumpTable(N->getIndex(), Ty, Flags);
4397}
4398
4399template <class NodeTy>
4400SDValue LoongArchTargetLowering::getAddr(NodeTy *N, SelectionDAG &DAG,
4402 bool IsLocal) const {
4403 SDLoc DL(N);
4404 EVT Ty = getPointerTy(DAG.getDataLayout());
4405 SDValue Addr = getTargetNode(N, DL, Ty, DAG, 0);
4406 SDValue Load;
4407
4408 switch (M) {
4409 default:
4410 report_fatal_error("Unsupported code model");
4411
4412 case CodeModel::Large: {
4413 assert(Subtarget.is64Bit() && "Large code model requires LA64");
4414
4415 // This is not actually used, but is necessary for successfully matching
4416 // the PseudoLA_*_LARGE nodes.
4417 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4418 if (IsLocal) {
4419 // This generates the pattern (PseudoLA_PCREL_LARGE tmp sym), that
4420 // eventually becomes the desired 5-insn code sequence.
4421 Load = SDValue(DAG.getMachineNode(LoongArch::PseudoLA_PCREL_LARGE, DL, Ty,
4422 Tmp, Addr),
4423 0);
4424 } else {
4425 // This generates the pattern (PseudoLA_GOT_LARGE tmp sym), that
4426 // eventually becomes the desired 5-insn code sequence.
4427 Load = SDValue(
4428 DAG.getMachineNode(LoongArch::PseudoLA_GOT_LARGE, DL, Ty, Tmp, Addr),
4429 0);
4430 }
4431 break;
4432 }
4433
4434 case CodeModel::Small:
4435 case CodeModel::Medium:
4436 if (IsLocal) {
4437 // This generates the pattern (PseudoLA_PCREL sym), which
4438 //
4439 // for la32r expands to:
4440 // (addi.w (pcaddu12i %pcadd_hi20(sym)) %pcadd_lo12(.Lpcadd_hi)).
4441 //
4442 // for la32s and la64 expands to:
4443 // (addi.w/d (pcalau12i %pc_hi20(sym)) %pc_lo12(sym)).
4444 Load = SDValue(
4445 DAG.getMachineNode(LoongArch::PseudoLA_PCREL, DL, Ty, Addr), 0);
4446 } else {
4447 // This generates the pattern (PseudoLA_GOT sym), which
4448 //
4449 // for la32r expands to:
4450 // (ld.w (pcaddu12i %got_pcadd_hi20(sym)) %pcadd_lo12(.Lpcadd_hi)).
4451 //
4452 // for la32s and la64 expands to:
4453 // (ld.w/d (pcalau12i %got_pc_hi20(sym)) %got_pc_lo12(sym)).
4454 Load =
4455 SDValue(DAG.getMachineNode(LoongArch::PseudoLA_GOT, DL, Ty, Addr), 0);
4456 }
4457 }
4458
4459 if (!IsLocal) {
4460 // Mark the load instruction as invariant to enable hoisting in MachineLICM.
4462 MachineMemOperand *MemOp = MF.getMachineMemOperand(
4466 LLT(Ty.getSimpleVT()), Align(Ty.getFixedSizeInBits() / 8));
4467 DAG.setNodeMemRefs(cast<MachineSDNode>(Load.getNode()), {MemOp});
4468 }
4469
4470 return Load;
4471}
4472
4473SDValue LoongArchTargetLowering::lowerBlockAddress(SDValue Op,
4474 SelectionDAG &DAG) const {
4475 return getAddr(cast<BlockAddressSDNode>(Op), DAG,
4476 DAG.getTarget().getCodeModel());
4477}
4478
4479SDValue LoongArchTargetLowering::lowerJumpTable(SDValue Op,
4480 SelectionDAG &DAG) const {
4481 return getAddr(cast<JumpTableSDNode>(Op), DAG,
4482 DAG.getTarget().getCodeModel());
4483}
4484
4485SDValue LoongArchTargetLowering::lowerConstantPool(SDValue Op,
4486 SelectionDAG &DAG) const {
4487 return getAddr(cast<ConstantPoolSDNode>(Op), DAG,
4488 DAG.getTarget().getCodeModel());
4489}
4490
4491SDValue LoongArchTargetLowering::lowerGlobalAddress(SDValue Op,
4492 SelectionDAG &DAG) const {
4493 GlobalAddressSDNode *N = cast<GlobalAddressSDNode>(Op);
4494 assert(N->getOffset() == 0 && "unexpected offset in global node");
4495 auto CM = DAG.getTarget().getCodeModel();
4496 const GlobalValue *GV = N->getGlobal();
4497
4498 if (GV->isDSOLocal() && isa<GlobalVariable>(GV)) {
4499 if (auto GCM = dyn_cast<GlobalVariable>(GV)->getCodeModel())
4500 CM = *GCM;
4501 }
4502
4503 return getAddr(N, DAG, CM, GV->isDSOLocal());
4504}
4505
4506SDValue LoongArchTargetLowering::getStaticTLSAddr(GlobalAddressSDNode *N,
4507 SelectionDAG &DAG,
4508 unsigned Opc, bool UseGOT,
4509 bool Large) const {
4510 SDLoc DL(N);
4511 EVT Ty = getPointerTy(DAG.getDataLayout());
4512 MVT GRLenVT = Subtarget.getGRLenVT();
4513
4514 // This is not actually used, but is necessary for successfully matching the
4515 // PseudoLA_*_LARGE nodes.
4516 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4517 SDValue Addr = DAG.getTargetGlobalAddress(N->getGlobal(), DL, Ty, 0, 0);
4518
4519 // Only IE needs an extra argument for large code model.
4520 SDValue Offset = Opc == LoongArch::PseudoLA_TLS_IE_LARGE
4521 ? SDValue(DAG.getMachineNode(Opc, DL, Ty, Tmp, Addr), 0)
4522 : SDValue(DAG.getMachineNode(Opc, DL, Ty, Addr), 0);
4523
4524 // If it is LE for normal/medium code model, the add tp operation will occur
4525 // during the pseudo-instruction expansion.
4526 if (Opc == LoongArch::PseudoLA_TLS_LE && !Large)
4527 return Offset;
4528
4529 if (UseGOT) {
4530 // Mark the load instruction as invariant to enable hoisting in MachineLICM.
4532 MachineMemOperand *MemOp = MF.getMachineMemOperand(
4536 LLT(Ty.getSimpleVT()), Align(Ty.getFixedSizeInBits() / 8));
4537 DAG.setNodeMemRefs(cast<MachineSDNode>(Offset.getNode()), {MemOp});
4538 }
4539
4540 // Add the thread pointer.
4541 return DAG.getNode(ISD::ADD, DL, Ty, Offset,
4542 DAG.getRegister(LoongArch::R2, GRLenVT));
4543}
4544
4545SDValue LoongArchTargetLowering::getDynamicTLSAddr(GlobalAddressSDNode *N,
4546 SelectionDAG &DAG,
4547 unsigned Opc,
4548 bool Large) const {
4549 SDLoc DL(N);
4550 EVT Ty = getPointerTy(DAG.getDataLayout());
4551 IntegerType *CallTy = Type::getIntNTy(*DAG.getContext(), Ty.getSizeInBits());
4552
4553 // This is not actually used, but is necessary for successfully matching the
4554 // PseudoLA_*_LARGE nodes.
4555 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4556
4557 // Use a PC-relative addressing mode to access the dynamic GOT address.
4558 SDValue Addr = DAG.getTargetGlobalAddress(N->getGlobal(), DL, Ty, 0, 0);
4559 SDValue Load = Large ? SDValue(DAG.getMachineNode(Opc, DL, Ty, Tmp, Addr), 0)
4560 : SDValue(DAG.getMachineNode(Opc, DL, Ty, Addr), 0);
4561
4562 // Prepare argument list to generate call.
4564 Args.emplace_back(Load, CallTy);
4565
4566 // Setup call to __tls_get_addr.
4567 TargetLowering::CallLoweringInfo CLI(DAG);
4568 CLI.setDebugLoc(DL)
4569 .setChain(DAG.getEntryNode())
4570 .setLibCallee(CallingConv::C, CallTy,
4571 DAG.getExternalSymbol("__tls_get_addr", Ty),
4572 std::move(Args));
4573
4574 return LowerCallTo(CLI).first;
4575}
4576
4577SDValue LoongArchTargetLowering::getTLSDescAddr(GlobalAddressSDNode *N,
4578 SelectionDAG &DAG, unsigned Opc,
4579 bool Large) const {
4580 SDLoc DL(N);
4581 EVT Ty = getPointerTy(DAG.getDataLayout());
4582 const GlobalValue *GV = N->getGlobal();
4583
4584 // This is not actually used, but is necessary for successfully matching the
4585 // PseudoLA_*_LARGE nodes.
4586 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4587
4588 // Use a PC-relative addressing mode to access the global dynamic GOT address.
4589 // This generates the pattern (PseudoLA_TLS_DESC_PC{,LARGE} sym).
4590 SDValue Addr = DAG.getTargetGlobalAddress(GV, DL, Ty, 0, 0);
4591 return Large ? SDValue(DAG.getMachineNode(Opc, DL, Ty, Tmp, Addr), 0)
4592 : SDValue(DAG.getMachineNode(Opc, DL, Ty, Addr), 0);
4593}
4594
4595SDValue
4596LoongArchTargetLowering::lowerGlobalTLSAddress(SDValue Op,
4597 SelectionDAG &DAG) const {
4600 report_fatal_error("In GHC calling convention TLS is not supported");
4601
4602 bool Large = DAG.getTarget().getCodeModel() == CodeModel::Large;
4603 assert((!Large || Subtarget.is64Bit()) && "Large code model requires LA64");
4604
4605 GlobalAddressSDNode *N = cast<GlobalAddressSDNode>(Op);
4606 assert(N->getOffset() == 0 && "unexpected offset in global node");
4607
4608 if (DAG.getTarget().useEmulatedTLS())
4609 reportFatalUsageError("the emulated TLS is prohibited");
4610
4611 bool IsDesc = DAG.getTarget().useTLSDESC();
4612
4613 switch (getTargetMachine().getTLSModel(N->getGlobal())) {
4615 // In this model, application code calls the dynamic linker function
4616 // __tls_get_addr to locate TLS offsets into the dynamic thread vector at
4617 // runtime.
4618 if (!IsDesc)
4619 return getDynamicTLSAddr(N, DAG,
4620 Large ? LoongArch::PseudoLA_TLS_GD_LARGE
4621 : LoongArch::PseudoLA_TLS_GD,
4622 Large);
4623 break;
4625 // Same as GeneralDynamic, except for assembly modifiers and relocation
4626 // records.
4627 if (!IsDesc)
4628 return getDynamicTLSAddr(N, DAG,
4629 Large ? LoongArch::PseudoLA_TLS_LD_LARGE
4630 : LoongArch::PseudoLA_TLS_LD,
4631 Large);
4632 break;
4634 // This model uses the GOT to resolve TLS offsets.
4635 return getStaticTLSAddr(N, DAG,
4636 Large ? LoongArch::PseudoLA_TLS_IE_LARGE
4637 : LoongArch::PseudoLA_TLS_IE,
4638 /*UseGOT=*/true, Large);
4640 // This model is used when static linking as the TLS offsets are resolved
4641 // during program linking.
4642 //
4643 // This node doesn't need an extra argument for the large code model.
4644 return getStaticTLSAddr(N, DAG, LoongArch::PseudoLA_TLS_LE,
4645 /*UseGOT=*/false, Large);
4646 }
4647
4648 return getTLSDescAddr(N, DAG,
4649 Large ? LoongArch::PseudoLA_TLS_DESC_LARGE
4650 : LoongArch::PseudoLA_TLS_DESC,
4651 Large);
4652}
4653
4654template <unsigned N>
4656 SelectionDAG &DAG, bool IsSigned = false) {
4657 auto *CImm = cast<ConstantSDNode>(Op->getOperand(ImmOp));
4658 // Check the ImmArg.
4659 if ((IsSigned && !isInt<N>(CImm->getSExtValue())) ||
4660 (!IsSigned && !isUInt<N>(CImm->getZExtValue()))) {
4661 DAG.getContext()->emitError(Op->getOperationName(0) +
4662 ": argument out of range.");
4663 return DAG.getNode(ISD::UNDEF, SDLoc(Op), Op.getValueType());
4664 }
4665 return SDValue();
4666}
4667
4668SDValue
4669LoongArchTargetLowering::lowerINTRINSIC_WO_CHAIN(SDValue Op,
4670 SelectionDAG &DAG) const {
4671 switch (Op.getConstantOperandVal(0)) {
4672 default:
4673 return SDValue(); // Don't custom lower most intrinsics.
4674 case Intrinsic::thread_pointer: {
4675 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4676 return DAG.getRegister(LoongArch::R2, PtrVT);
4677 }
4678 case Intrinsic::loongarch_lsx_vpickve2gr_d:
4679 case Intrinsic::loongarch_lsx_vpickve2gr_du:
4680 case Intrinsic::loongarch_lsx_vreplvei_d:
4681 case Intrinsic::loongarch_lasx_xvrepl128vei_d:
4682 return checkIntrinsicImmArg<1>(Op, 2, DAG);
4683 case Intrinsic::loongarch_lsx_vreplvei_w:
4684 case Intrinsic::loongarch_lasx_xvrepl128vei_w:
4685 case Intrinsic::loongarch_lasx_xvpickve2gr_d:
4686 case Intrinsic::loongarch_lasx_xvpickve2gr_du:
4687 case Intrinsic::loongarch_lasx_xvpickve_d:
4688 case Intrinsic::loongarch_lasx_xvpickve_d_f:
4689 return checkIntrinsicImmArg<2>(Op, 2, DAG);
4690 case Intrinsic::loongarch_lasx_xvinsve0_d:
4691 return checkIntrinsicImmArg<2>(Op, 3, DAG);
4692 case Intrinsic::loongarch_lsx_vsat_b:
4693 case Intrinsic::loongarch_lsx_vsat_bu:
4694 case Intrinsic::loongarch_lsx_vrotri_b:
4695 case Intrinsic::loongarch_lsx_vsllwil_h_b:
4696 case Intrinsic::loongarch_lsx_vsllwil_hu_bu:
4697 case Intrinsic::loongarch_lsx_vsrlri_b:
4698 case Intrinsic::loongarch_lsx_vsrari_b:
4699 case Intrinsic::loongarch_lsx_vreplvei_h:
4700 case Intrinsic::loongarch_lasx_xvsat_b:
4701 case Intrinsic::loongarch_lasx_xvsat_bu:
4702 case Intrinsic::loongarch_lasx_xvrotri_b:
4703 case Intrinsic::loongarch_lasx_xvsllwil_h_b:
4704 case Intrinsic::loongarch_lasx_xvsllwil_hu_bu:
4705 case Intrinsic::loongarch_lasx_xvsrlri_b:
4706 case Intrinsic::loongarch_lasx_xvsrari_b:
4707 case Intrinsic::loongarch_lasx_xvrepl128vei_h:
4708 case Intrinsic::loongarch_lasx_xvpickve_w:
4709 case Intrinsic::loongarch_lasx_xvpickve_w_f:
4710 return checkIntrinsicImmArg<3>(Op, 2, DAG);
4711 case Intrinsic::loongarch_lasx_xvinsve0_w:
4712 return checkIntrinsicImmArg<3>(Op, 3, DAG);
4713 case Intrinsic::loongarch_lsx_vsat_h:
4714 case Intrinsic::loongarch_lsx_vsat_hu:
4715 case Intrinsic::loongarch_lsx_vrotri_h:
4716 case Intrinsic::loongarch_lsx_vsllwil_w_h:
4717 case Intrinsic::loongarch_lsx_vsllwil_wu_hu:
4718 case Intrinsic::loongarch_lsx_vsrlri_h:
4719 case Intrinsic::loongarch_lsx_vsrari_h:
4720 case Intrinsic::loongarch_lsx_vreplvei_b:
4721 case Intrinsic::loongarch_lasx_xvsat_h:
4722 case Intrinsic::loongarch_lasx_xvsat_hu:
4723 case Intrinsic::loongarch_lasx_xvrotri_h:
4724 case Intrinsic::loongarch_lasx_xvsllwil_w_h:
4725 case Intrinsic::loongarch_lasx_xvsllwil_wu_hu:
4726 case Intrinsic::loongarch_lasx_xvsrlri_h:
4727 case Intrinsic::loongarch_lasx_xvsrari_h:
4728 case Intrinsic::loongarch_lasx_xvrepl128vei_b:
4729 return checkIntrinsicImmArg<4>(Op, 2, DAG);
4730 case Intrinsic::loongarch_lsx_vsrlni_b_h:
4731 case Intrinsic::loongarch_lsx_vsrani_b_h:
4732 case Intrinsic::loongarch_lsx_vsrlrni_b_h:
4733 case Intrinsic::loongarch_lsx_vsrarni_b_h:
4734 case Intrinsic::loongarch_lsx_vssrlni_b_h:
4735 case Intrinsic::loongarch_lsx_vssrani_b_h:
4736 case Intrinsic::loongarch_lsx_vssrlni_bu_h:
4737 case Intrinsic::loongarch_lsx_vssrani_bu_h:
4738 case Intrinsic::loongarch_lsx_vssrlrni_b_h:
4739 case Intrinsic::loongarch_lsx_vssrarni_b_h:
4740 case Intrinsic::loongarch_lsx_vssrlrni_bu_h:
4741 case Intrinsic::loongarch_lsx_vssrarni_bu_h:
4742 case Intrinsic::loongarch_lasx_xvsrlni_b_h:
4743 case Intrinsic::loongarch_lasx_xvsrani_b_h:
4744 case Intrinsic::loongarch_lasx_xvsrlrni_b_h:
4745 case Intrinsic::loongarch_lasx_xvsrarni_b_h:
4746 case Intrinsic::loongarch_lasx_xvssrlni_b_h:
4747 case Intrinsic::loongarch_lasx_xvssrani_b_h:
4748 case Intrinsic::loongarch_lasx_xvssrlni_bu_h:
4749 case Intrinsic::loongarch_lasx_xvssrani_bu_h:
4750 case Intrinsic::loongarch_lasx_xvssrlrni_b_h:
4751 case Intrinsic::loongarch_lasx_xvssrarni_b_h:
4752 case Intrinsic::loongarch_lasx_xvssrlrni_bu_h:
4753 case Intrinsic::loongarch_lasx_xvssrarni_bu_h:
4754 return checkIntrinsicImmArg<4>(Op, 3, DAG);
4755 case Intrinsic::loongarch_lsx_vsat_w:
4756 case Intrinsic::loongarch_lsx_vsat_wu:
4757 case Intrinsic::loongarch_lsx_vrotri_w:
4758 case Intrinsic::loongarch_lsx_vsllwil_d_w:
4759 case Intrinsic::loongarch_lsx_vsllwil_du_wu:
4760 case Intrinsic::loongarch_lsx_vsrlri_w:
4761 case Intrinsic::loongarch_lsx_vsrari_w:
4762 case Intrinsic::loongarch_lsx_vslei_bu:
4763 case Intrinsic::loongarch_lsx_vslei_hu:
4764 case Intrinsic::loongarch_lsx_vslei_wu:
4765 case Intrinsic::loongarch_lsx_vslei_du:
4766 case Intrinsic::loongarch_lsx_vslti_bu:
4767 case Intrinsic::loongarch_lsx_vslti_hu:
4768 case Intrinsic::loongarch_lsx_vslti_wu:
4769 case Intrinsic::loongarch_lsx_vslti_du:
4770 case Intrinsic::loongarch_lsx_vbsll_v:
4771 case Intrinsic::loongarch_lsx_vbsrl_v:
4772 case Intrinsic::loongarch_lasx_xvsat_w:
4773 case Intrinsic::loongarch_lasx_xvsat_wu:
4774 case Intrinsic::loongarch_lasx_xvrotri_w:
4775 case Intrinsic::loongarch_lasx_xvsllwil_d_w:
4776 case Intrinsic::loongarch_lasx_xvsllwil_du_wu:
4777 case Intrinsic::loongarch_lasx_xvsrlri_w:
4778 case Intrinsic::loongarch_lasx_xvsrari_w:
4779 case Intrinsic::loongarch_lasx_xvslei_bu:
4780 case Intrinsic::loongarch_lasx_xvslei_hu:
4781 case Intrinsic::loongarch_lasx_xvslei_wu:
4782 case Intrinsic::loongarch_lasx_xvslei_du:
4783 case Intrinsic::loongarch_lasx_xvslti_bu:
4784 case Intrinsic::loongarch_lasx_xvslti_hu:
4785 case Intrinsic::loongarch_lasx_xvslti_wu:
4786 case Intrinsic::loongarch_lasx_xvslti_du:
4787 case Intrinsic::loongarch_lasx_xvbsll_v:
4788 case Intrinsic::loongarch_lasx_xvbsrl_v:
4789 return checkIntrinsicImmArg<5>(Op, 2, DAG);
4790 case Intrinsic::loongarch_lsx_vseqi_b:
4791 case Intrinsic::loongarch_lsx_vseqi_h:
4792 case Intrinsic::loongarch_lsx_vseqi_w:
4793 case Intrinsic::loongarch_lsx_vseqi_d:
4794 case Intrinsic::loongarch_lsx_vslei_b:
4795 case Intrinsic::loongarch_lsx_vslei_h:
4796 case Intrinsic::loongarch_lsx_vslei_w:
4797 case Intrinsic::loongarch_lsx_vslei_d:
4798 case Intrinsic::loongarch_lsx_vslti_b:
4799 case Intrinsic::loongarch_lsx_vslti_h:
4800 case Intrinsic::loongarch_lsx_vslti_w:
4801 case Intrinsic::loongarch_lsx_vslti_d:
4802 case Intrinsic::loongarch_lasx_xvseqi_b:
4803 case Intrinsic::loongarch_lasx_xvseqi_h:
4804 case Intrinsic::loongarch_lasx_xvseqi_w:
4805 case Intrinsic::loongarch_lasx_xvseqi_d:
4806 case Intrinsic::loongarch_lasx_xvslei_b:
4807 case Intrinsic::loongarch_lasx_xvslei_h:
4808 case Intrinsic::loongarch_lasx_xvslei_w:
4809 case Intrinsic::loongarch_lasx_xvslei_d:
4810 case Intrinsic::loongarch_lasx_xvslti_b:
4811 case Intrinsic::loongarch_lasx_xvslti_h:
4812 case Intrinsic::loongarch_lasx_xvslti_w:
4813 case Intrinsic::loongarch_lasx_xvslti_d:
4814 return checkIntrinsicImmArg<5>(Op, 2, DAG, /*IsSigned=*/true);
4815 case Intrinsic::loongarch_lsx_vsrlni_h_w:
4816 case Intrinsic::loongarch_lsx_vsrani_h_w:
4817 case Intrinsic::loongarch_lsx_vsrlrni_h_w:
4818 case Intrinsic::loongarch_lsx_vsrarni_h_w:
4819 case Intrinsic::loongarch_lsx_vssrlni_h_w:
4820 case Intrinsic::loongarch_lsx_vssrani_h_w:
4821 case Intrinsic::loongarch_lsx_vssrlni_hu_w:
4822 case Intrinsic::loongarch_lsx_vssrani_hu_w:
4823 case Intrinsic::loongarch_lsx_vssrlrni_h_w:
4824 case Intrinsic::loongarch_lsx_vssrarni_h_w:
4825 case Intrinsic::loongarch_lsx_vssrlrni_hu_w:
4826 case Intrinsic::loongarch_lsx_vssrarni_hu_w:
4827 case Intrinsic::loongarch_lsx_vfrstpi_b:
4828 case Intrinsic::loongarch_lsx_vfrstpi_h:
4829 case Intrinsic::loongarch_lasx_xvsrlni_h_w:
4830 case Intrinsic::loongarch_lasx_xvsrani_h_w:
4831 case Intrinsic::loongarch_lasx_xvsrlrni_h_w:
4832 case Intrinsic::loongarch_lasx_xvsrarni_h_w:
4833 case Intrinsic::loongarch_lasx_xvssrlni_h_w:
4834 case Intrinsic::loongarch_lasx_xvssrani_h_w:
4835 case Intrinsic::loongarch_lasx_xvssrlni_hu_w:
4836 case Intrinsic::loongarch_lasx_xvssrani_hu_w:
4837 case Intrinsic::loongarch_lasx_xvssrlrni_h_w:
4838 case Intrinsic::loongarch_lasx_xvssrarni_h_w:
4839 case Intrinsic::loongarch_lasx_xvssrlrni_hu_w:
4840 case Intrinsic::loongarch_lasx_xvssrarni_hu_w:
4841 case Intrinsic::loongarch_lasx_xvfrstpi_b:
4842 case Intrinsic::loongarch_lasx_xvfrstpi_h:
4843 return checkIntrinsicImmArg<5>(Op, 3, DAG);
4844 case Intrinsic::loongarch_lsx_vsat_d:
4845 case Intrinsic::loongarch_lsx_vsat_du:
4846 case Intrinsic::loongarch_lsx_vrotri_d:
4847 case Intrinsic::loongarch_lsx_vsrlri_d:
4848 case Intrinsic::loongarch_lsx_vsrari_d:
4849 case Intrinsic::loongarch_lasx_xvsat_d:
4850 case Intrinsic::loongarch_lasx_xvsat_du:
4851 case Intrinsic::loongarch_lasx_xvrotri_d:
4852 case Intrinsic::loongarch_lasx_xvsrlri_d:
4853 case Intrinsic::loongarch_lasx_xvsrari_d:
4854 return checkIntrinsicImmArg<6>(Op, 2, DAG);
4855 case Intrinsic::loongarch_lsx_vsrlni_w_d:
4856 case Intrinsic::loongarch_lsx_vsrani_w_d:
4857 case Intrinsic::loongarch_lsx_vsrlrni_w_d:
4858 case Intrinsic::loongarch_lsx_vsrarni_w_d:
4859 case Intrinsic::loongarch_lsx_vssrlni_w_d:
4860 case Intrinsic::loongarch_lsx_vssrani_w_d:
4861 case Intrinsic::loongarch_lsx_vssrlni_wu_d:
4862 case Intrinsic::loongarch_lsx_vssrani_wu_d:
4863 case Intrinsic::loongarch_lsx_vssrlrni_w_d:
4864 case Intrinsic::loongarch_lsx_vssrarni_w_d:
4865 case Intrinsic::loongarch_lsx_vssrlrni_wu_d:
4866 case Intrinsic::loongarch_lsx_vssrarni_wu_d:
4867 case Intrinsic::loongarch_lasx_xvsrlni_w_d:
4868 case Intrinsic::loongarch_lasx_xvsrani_w_d:
4869 case Intrinsic::loongarch_lasx_xvsrlrni_w_d:
4870 case Intrinsic::loongarch_lasx_xvsrarni_w_d:
4871 case Intrinsic::loongarch_lasx_xvssrlni_w_d:
4872 case Intrinsic::loongarch_lasx_xvssrani_w_d:
4873 case Intrinsic::loongarch_lasx_xvssrlni_wu_d:
4874 case Intrinsic::loongarch_lasx_xvssrani_wu_d:
4875 case Intrinsic::loongarch_lasx_xvssrlrni_w_d:
4876 case Intrinsic::loongarch_lasx_xvssrarni_w_d:
4877 case Intrinsic::loongarch_lasx_xvssrlrni_wu_d:
4878 case Intrinsic::loongarch_lasx_xvssrarni_wu_d:
4879 return checkIntrinsicImmArg<6>(Op, 3, DAG);
4880 case Intrinsic::loongarch_lsx_vsrlni_d_q:
4881 case Intrinsic::loongarch_lsx_vsrani_d_q:
4882 case Intrinsic::loongarch_lsx_vsrlrni_d_q:
4883 case Intrinsic::loongarch_lsx_vsrarni_d_q:
4884 case Intrinsic::loongarch_lsx_vssrlni_d_q:
4885 case Intrinsic::loongarch_lsx_vssrani_d_q:
4886 case Intrinsic::loongarch_lsx_vssrlni_du_q:
4887 case Intrinsic::loongarch_lsx_vssrani_du_q:
4888 case Intrinsic::loongarch_lsx_vssrlrni_d_q:
4889 case Intrinsic::loongarch_lsx_vssrarni_d_q:
4890 case Intrinsic::loongarch_lsx_vssrlrni_du_q:
4891 case Intrinsic::loongarch_lsx_vssrarni_du_q:
4892 case Intrinsic::loongarch_lasx_xvsrlni_d_q:
4893 case Intrinsic::loongarch_lasx_xvsrani_d_q:
4894 case Intrinsic::loongarch_lasx_xvsrlrni_d_q:
4895 case Intrinsic::loongarch_lasx_xvsrarni_d_q:
4896 case Intrinsic::loongarch_lasx_xvssrlni_d_q:
4897 case Intrinsic::loongarch_lasx_xvssrani_d_q:
4898 case Intrinsic::loongarch_lasx_xvssrlni_du_q:
4899 case Intrinsic::loongarch_lasx_xvssrani_du_q:
4900 case Intrinsic::loongarch_lasx_xvssrlrni_d_q:
4901 case Intrinsic::loongarch_lasx_xvssrarni_d_q:
4902 case Intrinsic::loongarch_lasx_xvssrlrni_du_q:
4903 case Intrinsic::loongarch_lasx_xvssrarni_du_q:
4904 return checkIntrinsicImmArg<7>(Op, 3, DAG);
4905 case Intrinsic::loongarch_lsx_vnori_b:
4906 case Intrinsic::loongarch_lsx_vshuf4i_b:
4907 case Intrinsic::loongarch_lsx_vshuf4i_h:
4908 case Intrinsic::loongarch_lsx_vshuf4i_w:
4909 case Intrinsic::loongarch_lasx_xvnori_b:
4910 case Intrinsic::loongarch_lasx_xvshuf4i_b:
4911 case Intrinsic::loongarch_lasx_xvshuf4i_h:
4912 case Intrinsic::loongarch_lasx_xvshuf4i_w:
4913 case Intrinsic::loongarch_lasx_xvpermi_d:
4914 return checkIntrinsicImmArg<8>(Op, 2, DAG);
4915 case Intrinsic::loongarch_lsx_vshuf4i_d:
4916 case Intrinsic::loongarch_lsx_vpermi_w:
4917 case Intrinsic::loongarch_lsx_vbitseli_b:
4918 case Intrinsic::loongarch_lsx_vextrins_b:
4919 case Intrinsic::loongarch_lsx_vextrins_h:
4920 case Intrinsic::loongarch_lsx_vextrins_w:
4921 case Intrinsic::loongarch_lsx_vextrins_d:
4922 case Intrinsic::loongarch_lasx_xvshuf4i_d:
4923 case Intrinsic::loongarch_lasx_xvpermi_w:
4924 case Intrinsic::loongarch_lasx_xvpermi_q:
4925 case Intrinsic::loongarch_lasx_xvbitseli_b:
4926 case Intrinsic::loongarch_lasx_xvextrins_b:
4927 case Intrinsic::loongarch_lasx_xvextrins_h:
4928 case Intrinsic::loongarch_lasx_xvextrins_w:
4929 case Intrinsic::loongarch_lasx_xvextrins_d:
4930 return checkIntrinsicImmArg<8>(Op, 3, DAG);
4931 case Intrinsic::loongarch_lsx_vrepli_b:
4932 case Intrinsic::loongarch_lsx_vrepli_h:
4933 case Intrinsic::loongarch_lsx_vrepli_w:
4934 case Intrinsic::loongarch_lsx_vrepli_d:
4935 case Intrinsic::loongarch_lasx_xvrepli_b:
4936 case Intrinsic::loongarch_lasx_xvrepli_h:
4937 case Intrinsic::loongarch_lasx_xvrepli_w:
4938 case Intrinsic::loongarch_lasx_xvrepli_d:
4939 return checkIntrinsicImmArg<10>(Op, 1, DAG, /*IsSigned=*/true);
4940 case Intrinsic::loongarch_lsx_vldi:
4941 case Intrinsic::loongarch_lasx_xvldi:
4942 return checkIntrinsicImmArg<13>(Op, 1, DAG, /*IsSigned=*/true);
4943 }
4944}
4945
4946// Helper function that emits error message for intrinsics with chain and return
4947// merge values of a UNDEF and the chain.
4949 StringRef ErrorMsg,
4950 SelectionDAG &DAG) {
4951 DAG.getContext()->emitError(Op->getOperationName(0) + ": " + ErrorMsg + ".");
4952 return DAG.getMergeValues({DAG.getUNDEF(Op.getValueType()), Op.getOperand(0)},
4953 SDLoc(Op));
4954}
4955
4956SDValue
4957LoongArchTargetLowering::lowerINTRINSIC_W_CHAIN(SDValue Op,
4958 SelectionDAG &DAG) const {
4959 SDLoc DL(Op);
4960 MVT GRLenVT = Subtarget.getGRLenVT();
4961 EVT VT = Op.getValueType();
4962 SDValue Chain = Op.getOperand(0);
4963 const StringRef ErrorMsgOOR = "argument out of range";
4964 const StringRef ErrorMsgReqLA64 = "requires loongarch64";
4965 const StringRef ErrorMsgReqF = "requires basic 'f' target feature";
4966
4967 switch (Op.getConstantOperandVal(1)) {
4968 default:
4969 return Op;
4970 case Intrinsic::loongarch_crc_w_b_w:
4971 case Intrinsic::loongarch_crc_w_h_w:
4972 case Intrinsic::loongarch_crc_w_w_w:
4973 case Intrinsic::loongarch_crc_w_d_w:
4974 case Intrinsic::loongarch_crcc_w_b_w:
4975 case Intrinsic::loongarch_crcc_w_h_w:
4976 case Intrinsic::loongarch_crcc_w_w_w:
4977 case Intrinsic::loongarch_crcc_w_d_w:
4978 return emitIntrinsicWithChainErrorMessage(Op, ErrorMsgReqLA64, DAG);
4979 case Intrinsic::loongarch_csrrd_w:
4980 case Intrinsic::loongarch_csrrd_d: {
4981 unsigned Imm = Op.getConstantOperandVal(2);
4982 return !isUInt<14>(Imm)
4983 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
4984 : DAG.getNode(LoongArchISD::CSRRD, DL, {GRLenVT, MVT::Other},
4985 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
4986 }
4987 case Intrinsic::loongarch_csrwr_w:
4988 case Intrinsic::loongarch_csrwr_d: {
4989 unsigned Imm = Op.getConstantOperandVal(3);
4990 return !isUInt<14>(Imm)
4991 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
4992 : DAG.getNode(LoongArchISD::CSRWR, DL, {GRLenVT, MVT::Other},
4993 {Chain, Op.getOperand(2),
4994 DAG.getConstant(Imm, DL, GRLenVT)});
4995 }
4996 case Intrinsic::loongarch_csrxchg_w:
4997 case Intrinsic::loongarch_csrxchg_d: {
4998 unsigned Imm = Op.getConstantOperandVal(4);
4999 return !isUInt<14>(Imm)
5000 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5001 : DAG.getNode(LoongArchISD::CSRXCHG, DL, {GRLenVT, MVT::Other},
5002 {Chain, Op.getOperand(2), Op.getOperand(3),
5003 DAG.getConstant(Imm, DL, GRLenVT)});
5004 }
5005 case Intrinsic::loongarch_iocsrrd_d: {
5006 return DAG.getNode(
5007 LoongArchISD::IOCSRRD_D, DL, {GRLenVT, MVT::Other},
5008 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op.getOperand(2))});
5009 }
5010#define IOCSRRD_CASE(NAME, NODE) \
5011 case Intrinsic::loongarch_##NAME: { \
5012 return DAG.getNode(LoongArchISD::NODE, DL, {GRLenVT, MVT::Other}, \
5013 {Chain, Op.getOperand(2)}); \
5014 }
5015 IOCSRRD_CASE(iocsrrd_b, IOCSRRD_B);
5016 IOCSRRD_CASE(iocsrrd_h, IOCSRRD_H);
5017 IOCSRRD_CASE(iocsrrd_w, IOCSRRD_W);
5018#undef IOCSRRD_CASE
5019 case Intrinsic::loongarch_cpucfg: {
5020 return DAG.getNode(LoongArchISD::CPUCFG, DL, {GRLenVT, MVT::Other},
5021 {Chain, Op.getOperand(2)});
5022 }
5023 case Intrinsic::loongarch_lddir_d: {
5024 unsigned Imm = Op.getConstantOperandVal(3);
5025 return !isUInt<8>(Imm)
5026 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5027 : Op;
5028 }
5029 case Intrinsic::loongarch_movfcsr2gr: {
5030 if (!Subtarget.hasBasicF())
5031 return emitIntrinsicWithChainErrorMessage(Op, ErrorMsgReqF, DAG);
5032 unsigned Imm = Op.getConstantOperandVal(2);
5033 return !isUInt<2>(Imm)
5034 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5035 : DAG.getNode(LoongArchISD::MOVFCSR2GR, DL, {VT, MVT::Other},
5036 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
5037 }
5038 case Intrinsic::loongarch_lsx_vld:
5039 case Intrinsic::loongarch_lsx_vldrepl_b:
5040 case Intrinsic::loongarch_lasx_xvld:
5041 case Intrinsic::loongarch_lasx_xvldrepl_b:
5042 return !isInt<12>(cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5043 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5044 : SDValue();
5045 case Intrinsic::loongarch_lsx_vldrepl_h:
5046 case Intrinsic::loongarch_lasx_xvldrepl_h:
5047 return !isShiftedInt<11, 1>(
5048 cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5050 Op, "argument out of range or not a multiple of 2", DAG)
5051 : SDValue();
5052 case Intrinsic::loongarch_lsx_vldrepl_w:
5053 case Intrinsic::loongarch_lasx_xvldrepl_w:
5054 return !isShiftedInt<10, 2>(
5055 cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5057 Op, "argument out of range or not a multiple of 4", DAG)
5058 : SDValue();
5059 case Intrinsic::loongarch_lsx_vldrepl_d:
5060 case Intrinsic::loongarch_lasx_xvldrepl_d:
5061 return !isShiftedInt<9, 3>(
5062 cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5064 Op, "argument out of range or not a multiple of 8", DAG)
5065 : SDValue();
5066 }
5067}
5068
5069// Helper function that emits error message for intrinsics with void return
5070// value and return the chain.
5072 SelectionDAG &DAG) {
5073
5074 DAG.getContext()->emitError(Op->getOperationName(0) + ": " + ErrorMsg + ".");
5075 return Op.getOperand(0);
5076}
5077
5078SDValue LoongArchTargetLowering::lowerINTRINSIC_VOID(SDValue Op,
5079 SelectionDAG &DAG) const {
5080 SDLoc DL(Op);
5081 MVT GRLenVT = Subtarget.getGRLenVT();
5082 SDValue Chain = Op.getOperand(0);
5083 uint64_t IntrinsicEnum = Op.getConstantOperandVal(1);
5084 SDValue Op2 = Op.getOperand(2);
5085 const StringRef ErrorMsgOOR = "argument out of range";
5086 const StringRef ErrorMsgReqLA64 = "requires loongarch64";
5087 const StringRef ErrorMsgReqLA32 = "requires loongarch32";
5088 const StringRef ErrorMsgReqF = "requires basic 'f' target feature";
5089
5090 switch (IntrinsicEnum) {
5091 default:
5092 // TODO: Add more Intrinsics.
5093 return SDValue();
5094 case Intrinsic::loongarch_cacop_d:
5095 case Intrinsic::loongarch_cacop_w: {
5096 if (IntrinsicEnum == Intrinsic::loongarch_cacop_d && !Subtarget.is64Bit())
5097 return emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG);
5098 if (IntrinsicEnum == Intrinsic::loongarch_cacop_w && Subtarget.is64Bit())
5099 return emitIntrinsicErrorMessage(Op, ErrorMsgReqLA32, DAG);
5100 // call void @llvm.loongarch.cacop.[d/w](uimm5, rj, simm12)
5101 unsigned Imm1 = Op2->getAsZExtVal();
5102 int Imm2 = cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue();
5103 if (!isUInt<5>(Imm1) || !isInt<12>(Imm2))
5104 return emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG);
5105 return Op;
5106 }
5107 case Intrinsic::loongarch_dbar: {
5108 unsigned Imm = Op2->getAsZExtVal();
5109 return !isUInt<15>(Imm)
5110 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5111 : DAG.getNode(LoongArchISD::DBAR, DL, MVT::Other, Chain,
5112 DAG.getConstant(Imm, DL, GRLenVT));
5113 }
5114 case Intrinsic::loongarch_ibar: {
5115 unsigned Imm = Op2->getAsZExtVal();
5116 return !isUInt<15>(Imm)
5117 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5118 : DAG.getNode(LoongArchISD::IBAR, DL, MVT::Other, Chain,
5119 DAG.getConstant(Imm, DL, GRLenVT));
5120 }
5121 case Intrinsic::loongarch_break: {
5122 unsigned Imm = Op2->getAsZExtVal();
5123 return !isUInt<15>(Imm)
5124 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5125 : DAG.getNode(LoongArchISD::BREAK, DL, MVT::Other, Chain,
5126 DAG.getConstant(Imm, DL, GRLenVT));
5127 }
5128 case Intrinsic::loongarch_movgr2fcsr: {
5129 if (!Subtarget.hasBasicF())
5130 return emitIntrinsicErrorMessage(Op, ErrorMsgReqF, DAG);
5131 unsigned Imm = Op2->getAsZExtVal();
5132 return !isUInt<2>(Imm)
5133 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5134 : DAG.getNode(LoongArchISD::MOVGR2FCSR, DL, MVT::Other, Chain,
5135 DAG.getConstant(Imm, DL, GRLenVT),
5136 DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT,
5137 Op.getOperand(3)));
5138 }
5139 case Intrinsic::loongarch_syscall: {
5140 unsigned Imm = Op2->getAsZExtVal();
5141 return !isUInt<15>(Imm)
5142 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5143 : DAG.getNode(LoongArchISD::SYSCALL, DL, MVT::Other, Chain,
5144 DAG.getConstant(Imm, DL, GRLenVT));
5145 }
5146#define IOCSRWR_CASE(NAME, NODE) \
5147 case Intrinsic::loongarch_##NAME: { \
5148 SDValue Op3 = Op.getOperand(3); \
5149 return Subtarget.is64Bit() \
5150 ? DAG.getNode(LoongArchISD::NODE, DL, MVT::Other, Chain, \
5151 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2), \
5152 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op3)) \
5153 : DAG.getNode(LoongArchISD::NODE, DL, MVT::Other, Chain, Op2, \
5154 Op3); \
5155 }
5156 IOCSRWR_CASE(iocsrwr_b, IOCSRWR_B);
5157 IOCSRWR_CASE(iocsrwr_h, IOCSRWR_H);
5158 IOCSRWR_CASE(iocsrwr_w, IOCSRWR_W);
5159#undef IOCSRWR_CASE
5160 case Intrinsic::loongarch_iocsrwr_d: {
5161 return !Subtarget.is64Bit()
5162 ? emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG)
5163 : DAG.getNode(LoongArchISD::IOCSRWR_D, DL, MVT::Other, Chain,
5164 Op2,
5165 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64,
5166 Op.getOperand(3)));
5167 }
5168#define ASRT_LE_GT_CASE(NAME) \
5169 case Intrinsic::loongarch_##NAME: { \
5170 return !Subtarget.is64Bit() \
5171 ? emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG) \
5172 : Op; \
5173 }
5174 ASRT_LE_GT_CASE(asrtle_d)
5175 ASRT_LE_GT_CASE(asrtgt_d)
5176#undef ASRT_LE_GT_CASE
5177 case Intrinsic::loongarch_ldpte_d: {
5178 unsigned Imm = Op.getConstantOperandVal(3);
5179 return !Subtarget.is64Bit()
5180 ? emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG)
5181 : !isUInt<8>(Imm) ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5182 : Op;
5183 }
5184 case Intrinsic::loongarch_lsx_vst:
5185 case Intrinsic::loongarch_lasx_xvst:
5186 return !isInt<12>(cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue())
5187 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5188 : SDValue();
5189 case Intrinsic::loongarch_lasx_xvstelm_b:
5190 return (!isInt<8>(cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5191 !isUInt<5>(Op.getConstantOperandVal(5)))
5192 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5193 : SDValue();
5194 case Intrinsic::loongarch_lsx_vstelm_b:
5195 return (!isInt<8>(cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5196 !isUInt<4>(Op.getConstantOperandVal(5)))
5197 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5198 : SDValue();
5199 case Intrinsic::loongarch_lasx_xvstelm_h:
5200 return (!isShiftedInt<8, 1>(
5201 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5202 !isUInt<4>(Op.getConstantOperandVal(5)))
5204 Op, "argument out of range or not a multiple of 2", DAG)
5205 : SDValue();
5206 case Intrinsic::loongarch_lsx_vstelm_h:
5207 return (!isShiftedInt<8, 1>(
5208 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5209 !isUInt<3>(Op.getConstantOperandVal(5)))
5211 Op, "argument out of range or not a multiple of 2", DAG)
5212 : SDValue();
5213 case Intrinsic::loongarch_lasx_xvstelm_w:
5214 return (!isShiftedInt<8, 2>(
5215 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5216 !isUInt<3>(Op.getConstantOperandVal(5)))
5218 Op, "argument out of range or not a multiple of 4", DAG)
5219 : SDValue();
5220 case Intrinsic::loongarch_lsx_vstelm_w:
5221 return (!isShiftedInt<8, 2>(
5222 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5223 !isUInt<2>(Op.getConstantOperandVal(5)))
5225 Op, "argument out of range or not a multiple of 4", DAG)
5226 : SDValue();
5227 case Intrinsic::loongarch_lasx_xvstelm_d:
5228 return (!isShiftedInt<8, 3>(
5229 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5230 !isUInt<2>(Op.getConstantOperandVal(5)))
5232 Op, "argument out of range or not a multiple of 8", DAG)
5233 : SDValue();
5234 case Intrinsic::loongarch_lsx_vstelm_d:
5235 return (!isShiftedInt<8, 3>(
5236 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5237 !isUInt<1>(Op.getConstantOperandVal(5)))
5239 Op, "argument out of range or not a multiple of 8", DAG)
5240 : SDValue();
5241 }
5242}
5243
5244SDValue LoongArchTargetLowering::lowerShiftLeftParts(SDValue Op,
5245 SelectionDAG &DAG) const {
5246 SDLoc DL(Op);
5247 SDValue Lo = Op.getOperand(0);
5248 SDValue Hi = Op.getOperand(1);
5249 SDValue Shamt = Op.getOperand(2);
5250 EVT VT = Lo.getValueType();
5251
5252 // if Shamt-GRLen < 0: // Shamt < GRLen
5253 // Lo = Lo << Shamt
5254 // Hi = (Hi << Shamt) | ((Lo >>u 1) >>u (GRLen-1 ^ Shamt))
5255 // else:
5256 // Lo = 0
5257 // Hi = Lo << (Shamt-GRLen)
5258
5259 SDValue Zero = DAG.getConstant(0, DL, VT);
5260 SDValue One = DAG.getConstant(1, DL, VT);
5261 SDValue MinusGRLen =
5262 DAG.getSignedConstant(-(int)Subtarget.getGRLen(), DL, VT);
5263 SDValue GRLenMinus1 = DAG.getConstant(Subtarget.getGRLen() - 1, DL, VT);
5264 SDValue ShamtMinusGRLen = DAG.getNode(ISD::ADD, DL, VT, Shamt, MinusGRLen);
5265 SDValue GRLenMinus1Shamt = DAG.getNode(ISD::XOR, DL, VT, Shamt, GRLenMinus1);
5266
5267 SDValue LoTrue = DAG.getNode(ISD::SHL, DL, VT, Lo, Shamt);
5268 SDValue ShiftRight1Lo = DAG.getNode(ISD::SRL, DL, VT, Lo, One);
5269 SDValue ShiftRightLo =
5270 DAG.getNode(ISD::SRL, DL, VT, ShiftRight1Lo, GRLenMinus1Shamt);
5271 SDValue ShiftLeftHi = DAG.getNode(ISD::SHL, DL, VT, Hi, Shamt);
5272 SDValue HiTrue = DAG.getNode(ISD::OR, DL, VT, ShiftLeftHi, ShiftRightLo);
5273 SDValue HiFalse = DAG.getNode(ISD::SHL, DL, VT, Lo, ShamtMinusGRLen);
5274
5275 SDValue CC = DAG.getSetCC(DL, VT, ShamtMinusGRLen, Zero, ISD::SETLT);
5276
5277 Lo = DAG.getNode(ISD::SELECT, DL, VT, CC, LoTrue, Zero);
5278 Hi = DAG.getNode(ISD::SELECT, DL, VT, CC, HiTrue, HiFalse);
5279
5280 SDValue Parts[2] = {Lo, Hi};
5281 return DAG.getMergeValues(Parts, DL);
5282}
5283
5284SDValue LoongArchTargetLowering::lowerShiftRightParts(SDValue Op,
5285 SelectionDAG &DAG,
5286 bool IsSRA) const {
5287 SDLoc DL(Op);
5288 SDValue Lo = Op.getOperand(0);
5289 SDValue Hi = Op.getOperand(1);
5290 SDValue Shamt = Op.getOperand(2);
5291 EVT VT = Lo.getValueType();
5292
5293 // SRA expansion:
5294 // if Shamt-GRLen < 0: // Shamt < GRLen
5295 // Lo = (Lo >>u Shamt) | ((Hi << 1) << (ShAmt ^ GRLen-1))
5296 // Hi = Hi >>s Shamt
5297 // else:
5298 // Lo = Hi >>s (Shamt-GRLen);
5299 // Hi = Hi >>s (GRLen-1)
5300 //
5301 // SRL expansion:
5302 // if Shamt-GRLen < 0: // Shamt < GRLen
5303 // Lo = (Lo >>u Shamt) | ((Hi << 1) << (ShAmt ^ GRLen-1))
5304 // Hi = Hi >>u Shamt
5305 // else:
5306 // Lo = Hi >>u (Shamt-GRLen);
5307 // Hi = 0;
5308
5309 unsigned ShiftRightOp = IsSRA ? ISD::SRA : ISD::SRL;
5310
5311 SDValue Zero = DAG.getConstant(0, DL, VT);
5312 SDValue One = DAG.getConstant(1, DL, VT);
5313 SDValue MinusGRLen =
5314 DAG.getSignedConstant(-(int)Subtarget.getGRLen(), DL, VT);
5315 SDValue GRLenMinus1 = DAG.getConstant(Subtarget.getGRLen() - 1, DL, VT);
5316 SDValue ShamtMinusGRLen = DAG.getNode(ISD::ADD, DL, VT, Shamt, MinusGRLen);
5317 SDValue GRLenMinus1Shamt = DAG.getNode(ISD::XOR, DL, VT, Shamt, GRLenMinus1);
5318
5319 SDValue ShiftRightLo = DAG.getNode(ISD::SRL, DL, VT, Lo, Shamt);
5320 SDValue ShiftLeftHi1 = DAG.getNode(ISD::SHL, DL, VT, Hi, One);
5321 SDValue ShiftLeftHi =
5322 DAG.getNode(ISD::SHL, DL, VT, ShiftLeftHi1, GRLenMinus1Shamt);
5323 SDValue LoTrue = DAG.getNode(ISD::OR, DL, VT, ShiftRightLo, ShiftLeftHi);
5324 SDValue HiTrue = DAG.getNode(ShiftRightOp, DL, VT, Hi, Shamt);
5325 SDValue LoFalse = DAG.getNode(ShiftRightOp, DL, VT, Hi, ShamtMinusGRLen);
5326 SDValue HiFalse =
5327 IsSRA ? DAG.getNode(ISD::SRA, DL, VT, Hi, GRLenMinus1) : Zero;
5328
5329 SDValue CC = DAG.getSetCC(DL, VT, ShamtMinusGRLen, Zero, ISD::SETLT);
5330
5331 Lo = DAG.getNode(ISD::SELECT, DL, VT, CC, LoTrue, LoFalse);
5332 Hi = DAG.getNode(ISD::SELECT, DL, VT, CC, HiTrue, HiFalse);
5333
5334 SDValue Parts[2] = {Lo, Hi};
5335 return DAG.getMergeValues(Parts, DL);
5336}
5337
5338// Returns the opcode of the target-specific SDNode that implements the 32-bit
5339// form of the given Opcode.
5340static unsigned getLoongArchWOpcode(unsigned Opcode) {
5341 switch (Opcode) {
5342 default:
5343 llvm_unreachable("Unexpected opcode");
5344 case ISD::SDIV:
5345 return LoongArchISD::DIV_W;
5346 case ISD::UDIV:
5347 return LoongArchISD::DIV_WU;
5348 case ISD::SREM:
5349 return LoongArchISD::MOD_W;
5350 case ISD::UREM:
5351 return LoongArchISD::MOD_WU;
5352 case ISD::SHL:
5353 return LoongArchISD::SLL_W;
5354 case ISD::SRA:
5355 return LoongArchISD::SRA_W;
5356 case ISD::SRL:
5357 return LoongArchISD::SRL_W;
5358 case ISD::ROTL:
5359 case ISD::ROTR:
5360 return LoongArchISD::ROTR_W;
5361 case ISD::CTTZ:
5362 return LoongArchISD::CTZ_W;
5363 case ISD::CTLZ:
5364 return LoongArchISD::CLZ_W;
5365 }
5366}
5367
5368// Converts the given i8/i16/i32 operation to a target-specific SelectionDAG
5369// node. Because i8/i16/i32 isn't a legal type for LA64, these operations would
5370// otherwise be promoted to i64, making it difficult to select the
5371// SLL_W/.../*W later one because the fact the operation was originally of
5372// type i8/i16/i32 is lost.
5374 unsigned ExtOpc = ISD::ANY_EXTEND) {
5375 SDLoc DL(N);
5376 unsigned WOpcode = getLoongArchWOpcode(N->getOpcode());
5377 SDValue NewOp0, NewRes;
5378
5379 switch (NumOp) {
5380 default:
5381 llvm_unreachable("Unexpected NumOp");
5382 case 1: {
5383 NewOp0 = DAG.getNode(ExtOpc, DL, MVT::i64, N->getOperand(0));
5384 NewRes = DAG.getNode(WOpcode, DL, MVT::i64, NewOp0);
5385 break;
5386 }
5387 case 2: {
5388 NewOp0 = DAG.getNode(ExtOpc, DL, MVT::i64, N->getOperand(0));
5389 SDValue NewOp1 = DAG.getNode(ExtOpc, DL, MVT::i64, N->getOperand(1));
5390 if (N->getOpcode() == ISD::ROTL) {
5391 SDValue TmpOp = DAG.getConstant(32, DL, MVT::i64);
5392 NewOp1 = DAG.getNode(ISD::SUB, DL, MVT::i64, TmpOp, NewOp1);
5393 }
5394 NewRes = DAG.getNode(WOpcode, DL, MVT::i64, NewOp0, NewOp1);
5395 break;
5396 }
5397 // TODO:Handle more NumOp.
5398 }
5399
5400 // ReplaceNodeResults requires we maintain the same type for the return
5401 // value.
5402 return DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), NewRes);
5403}
5404
5405// Converts the given 32-bit operation to a i64 operation with signed extension
5406// semantic to reduce the signed extension instructions.
5408 SDLoc DL(N);
5409 SDValue NewOp0 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(0));
5410 SDValue NewOp1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(1));
5411 SDValue NewWOp = DAG.getNode(N->getOpcode(), DL, MVT::i64, NewOp0, NewOp1);
5412 SDValue NewRes = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i64, NewWOp,
5413 DAG.getValueType(MVT::i32));
5414 return DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, NewRes);
5415}
5416
5417// Helper function that emits error message for intrinsics with/without chain
5418// and return a UNDEF or and the chain as the results.
5421 StringRef ErrorMsg, bool WithChain = true) {
5422 DAG.getContext()->emitError(N->getOperationName(0) + ": " + ErrorMsg + ".");
5423 Results.push_back(DAG.getUNDEF(N->getValueType(0)));
5424 if (!WithChain)
5425 return;
5426 Results.push_back(N->getOperand(0));
5427}
5428
5429template <unsigned N>
5430static void
5432 SelectionDAG &DAG, const LoongArchSubtarget &Subtarget,
5433 unsigned ResOp) {
5434 const StringRef ErrorMsgOOR = "argument out of range";
5435 unsigned Imm = Node->getConstantOperandVal(2);
5436 if (!isUInt<N>(Imm)) {
5438 /*WithChain=*/false);
5439 return;
5440 }
5441 SDLoc DL(Node);
5442 SDValue Vec = Node->getOperand(1);
5443
5444 SDValue PickElt =
5445 DAG.getNode(ResOp, DL, Subtarget.getGRLenVT(), Vec,
5446 DAG.getConstant(Imm, DL, Subtarget.getGRLenVT()),
5448 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, Node->getValueType(0),
5449 PickElt.getValue(0)));
5450}
5451
5454 SelectionDAG &DAG,
5455 const LoongArchSubtarget &Subtarget,
5456 unsigned ResOp) {
5457 SDLoc DL(N);
5458 SDValue Vec = N->getOperand(1);
5459
5460 SDValue CB = DAG.getNode(ResOp, DL, Subtarget.getGRLenVT(), Vec);
5461 Results.push_back(
5462 DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), CB.getValue(0)));
5463}
5464
5465static void
5467 SelectionDAG &DAG,
5468 const LoongArchSubtarget &Subtarget) {
5469 switch (N->getConstantOperandVal(0)) {
5470 default:
5471 llvm_unreachable("Unexpected Intrinsic.");
5472 case Intrinsic::loongarch_lsx_vpickve2gr_b:
5473 replaceVPICKVE2GRResults<4>(N, Results, DAG, Subtarget,
5474 LoongArchISD::VPICK_SEXT_ELT);
5475 break;
5476 case Intrinsic::loongarch_lsx_vpickve2gr_h:
5477 case Intrinsic::loongarch_lasx_xvpickve2gr_w:
5478 replaceVPICKVE2GRResults<3>(N, Results, DAG, Subtarget,
5479 LoongArchISD::VPICK_SEXT_ELT);
5480 break;
5481 case Intrinsic::loongarch_lsx_vpickve2gr_w:
5482 replaceVPICKVE2GRResults<2>(N, Results, DAG, Subtarget,
5483 LoongArchISD::VPICK_SEXT_ELT);
5484 break;
5485 case Intrinsic::loongarch_lsx_vpickve2gr_bu:
5486 replaceVPICKVE2GRResults<4>(N, Results, DAG, Subtarget,
5487 LoongArchISD::VPICK_ZEXT_ELT);
5488 break;
5489 case Intrinsic::loongarch_lsx_vpickve2gr_hu:
5490 case Intrinsic::loongarch_lasx_xvpickve2gr_wu:
5491 replaceVPICKVE2GRResults<3>(N, Results, DAG, Subtarget,
5492 LoongArchISD::VPICK_ZEXT_ELT);
5493 break;
5494 case Intrinsic::loongarch_lsx_vpickve2gr_wu:
5495 replaceVPICKVE2GRResults<2>(N, Results, DAG, Subtarget,
5496 LoongArchISD::VPICK_ZEXT_ELT);
5497 break;
5498 case Intrinsic::loongarch_lsx_bz_b:
5499 case Intrinsic::loongarch_lsx_bz_h:
5500 case Intrinsic::loongarch_lsx_bz_w:
5501 case Intrinsic::loongarch_lsx_bz_d:
5502 case Intrinsic::loongarch_lasx_xbz_b:
5503 case Intrinsic::loongarch_lasx_xbz_h:
5504 case Intrinsic::loongarch_lasx_xbz_w:
5505 case Intrinsic::loongarch_lasx_xbz_d:
5506 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5507 LoongArchISD::VALL_ZERO);
5508 break;
5509 case Intrinsic::loongarch_lsx_bz_v:
5510 case Intrinsic::loongarch_lasx_xbz_v:
5511 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5512 LoongArchISD::VANY_ZERO);
5513 break;
5514 case Intrinsic::loongarch_lsx_bnz_b:
5515 case Intrinsic::loongarch_lsx_bnz_h:
5516 case Intrinsic::loongarch_lsx_bnz_w:
5517 case Intrinsic::loongarch_lsx_bnz_d:
5518 case Intrinsic::loongarch_lasx_xbnz_b:
5519 case Intrinsic::loongarch_lasx_xbnz_h:
5520 case Intrinsic::loongarch_lasx_xbnz_w:
5521 case Intrinsic::loongarch_lasx_xbnz_d:
5522 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5523 LoongArchISD::VALL_NONZERO);
5524 break;
5525 case Intrinsic::loongarch_lsx_bnz_v:
5526 case Intrinsic::loongarch_lasx_xbnz_v:
5527 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5528 LoongArchISD::VANY_NONZERO);
5529 break;
5530 }
5531}
5532
5535 SelectionDAG &DAG) {
5536 assert(N->getValueType(0) == MVT::i128 &&
5537 "AtomicCmpSwap on types less than 128 should be legal");
5538 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand();
5539
5540 unsigned Opcode;
5541 switch (MemOp->getMergedOrdering()) {
5545 Opcode = LoongArch::PseudoCmpXchg128Acquire;
5546 break;
5549 Opcode = LoongArch::PseudoCmpXchg128;
5550 break;
5551 default:
5552 llvm_unreachable("Unexpected ordering!");
5553 }
5554
5555 SDLoc DL(N);
5556 auto CmpVal = DAG.SplitScalar(N->getOperand(2), DL, MVT::i64, MVT::i64);
5557 auto NewVal = DAG.SplitScalar(N->getOperand(3), DL, MVT::i64, MVT::i64);
5558 SDValue Ops[] = {N->getOperand(1), CmpVal.first, CmpVal.second,
5559 NewVal.first, NewVal.second, N->getOperand(0)};
5560
5561 SDNode *CmpSwap = DAG.getMachineNode(
5562 Opcode, SDLoc(N), DAG.getVTList(MVT::i64, MVT::i64, MVT::i64, MVT::Other),
5563 Ops);
5564 DAG.setNodeMemRefs(cast<MachineSDNode>(CmpSwap), {MemOp});
5565 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i128,
5566 SDValue(CmpSwap, 0), SDValue(CmpSwap, 1)));
5567 Results.push_back(SDValue(CmpSwap, 3));
5568}
5569
5572 SDLoc DL(N);
5573 EVT VT = N->getValueType(0);
5574 switch (N->getOpcode()) {
5575 default:
5576 llvm_unreachable("Don't know how to legalize this operation");
5577 case ISD::ADD:
5578 case ISD::SUB:
5579 assert(N->getValueType(0) == MVT::i32 && Subtarget.is64Bit() &&
5580 "Unexpected custom legalisation");
5581 Results.push_back(customLegalizeToWOpWithSExt(N, DAG));
5582 break;
5583 case ISD::SDIV:
5584 case ISD::UDIV:
5585 case ISD::SREM:
5586 case ISD::UREM:
5587 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5588 "Unexpected custom legalisation");
5589 Results.push_back(customLegalizeToWOp(N, DAG, 2,
5590 Subtarget.hasDiv32() && VT == MVT::i32
5592 : ISD::SIGN_EXTEND));
5593 break;
5594 case ISD::SHL:
5595 case ISD::SRA:
5596 case ISD::SRL:
5597 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5598 "Unexpected custom legalisation");
5599 if (N->getOperand(1).getOpcode() != ISD::Constant) {
5600 Results.push_back(customLegalizeToWOp(N, DAG, 2));
5601 break;
5602 }
5603 break;
5604 case ISD::ROTL:
5605 case ISD::ROTR:
5606 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5607 "Unexpected custom legalisation");
5608 Results.push_back(customLegalizeToWOp(N, DAG, 2));
5609 break;
5610 case ISD::LOAD: {
5611 // Use an f64 load and a scalar_to_vector for v2f32 loads. This avoids
5612 // scalarizing in 32-bit mode. In 64-bit mode this avoids a int->fp
5613 // cast since type legalization will try to use an i64 load.
5614 MVT VT = N->getSimpleValueType(0);
5615 assert(VT == MVT::v2f32 && Subtarget.hasExtLSX() &&
5616 "Unexpected custom legalisation");
5618 "Unexpected type action!");
5619 if (!ISD::isNON_EXTLoad(N))
5620 return;
5621 auto *Ld = cast<LoadSDNode>(N);
5622 SDValue Res = DAG.getLoad(MVT::f64, DL, Ld->getChain(), Ld->getBasePtr(),
5623 Ld->getPointerInfo(), Ld->getBaseAlign(),
5624 Ld->getMemOperand()->getFlags());
5625 SDValue Chain = Res.getValue(1);
5626 MVT VecVT = MVT::getVectorVT(MVT::f64, 2);
5627 Res = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VecVT, Res);
5628 EVT WideVT = getTypeToTransformTo(*DAG.getContext(), VT);
5629 Res = DAG.getBitcast(WideVT, Res);
5630 Results.push_back(Res);
5631 Results.push_back(Chain);
5632 break;
5633 }
5634 case ISD::FP_TO_SINT: {
5635 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5636 "Unexpected custom legalisation");
5637 SDValue Src = N->getOperand(0);
5638 EVT FVT = EVT::getFloatingPointVT(N->getValueSizeInBits(0));
5639 if (getTypeAction(*DAG.getContext(), Src.getValueType()) !=
5641 if (!isTypeLegal(Src.getValueType()))
5642 return;
5643 if (Src.getValueType() == MVT::f16)
5644 Src = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
5645 SDValue Dst = DAG.getNode(LoongArchISD::FTINT, DL, FVT, Src);
5646 Results.push_back(DAG.getNode(ISD::BITCAST, DL, VT, Dst));
5647 return;
5648 }
5649 // If the FP type needs to be softened, emit a library call using the 'si'
5650 // version. If we left it to default legalization we'd end up with 'di'.
5651 RTLIB::Libcall LC;
5652 LC = RTLIB::getFPTOSINT(Src.getValueType(), VT);
5653 MakeLibCallOptions CallOptions;
5654 EVT OpVT = Src.getValueType();
5655 CallOptions.setTypeListBeforeSoften(OpVT, VT);
5656 SDValue Chain = SDValue();
5657 SDValue Result;
5658 std::tie(Result, Chain) =
5659 makeLibCall(DAG, LC, VT, Src, CallOptions, DL, Chain);
5660 Results.push_back(Result);
5661 break;
5662 }
5663 case ISD::BITCAST: {
5664 SDValue Src = N->getOperand(0);
5665 EVT SrcVT = Src.getValueType();
5666 if (VT == MVT::i32 && SrcVT == MVT::f32 && Subtarget.is64Bit() &&
5667 Subtarget.hasBasicF()) {
5668 SDValue Dst =
5669 DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Src);
5670 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Dst));
5671 } else if (VT == MVT::i64 && SrcVT == MVT::f64 && !Subtarget.is64Bit()) {
5672 SDValue NewReg = DAG.getNode(LoongArchISD::SPLIT_PAIR_F64, DL,
5673 DAG.getVTList(MVT::i32, MVT::i32), Src);
5674 SDValue RetReg = DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64,
5675 NewReg.getValue(0), NewReg.getValue(1));
5676 Results.push_back(RetReg);
5677 }
5678 break;
5679 }
5680 case ISD::FP_TO_UINT: {
5681 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5682 "Unexpected custom legalisation");
5683 auto &TLI = DAG.getTargetLoweringInfo();
5684 SDValue Tmp1, Tmp2;
5685 TLI.expandFP_TO_UINT(N, Tmp1, Tmp2, DAG);
5686 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Tmp1));
5687 break;
5688 }
5689 case ISD::FP_ROUND: {
5690 assert(VT == MVT::v2f32 && Subtarget.hasExtLSX() &&
5691 "Unexpected custom legalisation");
5692 // On LSX platforms, rounding from v2f64 to v4f32 (after legalization from
5693 // v2f32) is scalarized. Add a customized v2f32 widening to convert it into
5694 // a target-specific LoongArchISD::VFCVT to optimize it.
5695 SDValue Op0 = N->getOperand(0);
5696 EVT OpVT = Op0.getValueType();
5697 if (OpVT == MVT::v2f64) {
5698 SDValue Undef = DAG.getUNDEF(OpVT);
5699 SDValue Dst =
5700 DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v4f32, Undef, Op0);
5701 Results.push_back(Dst);
5702 }
5703 break;
5704 }
5705 case ISD::BSWAP: {
5706 SDValue Src = N->getOperand(0);
5707 assert((VT == MVT::i16 || VT == MVT::i32) &&
5708 "Unexpected custom legalization");
5709 MVT GRLenVT = Subtarget.getGRLenVT();
5710 SDValue NewSrc = DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT, Src);
5711 SDValue Tmp;
5712 switch (VT.getSizeInBits()) {
5713 default:
5714 llvm_unreachable("Unexpected operand width");
5715 case 16:
5716 Tmp = DAG.getNode(LoongArchISD::REVB_2H, DL, GRLenVT, NewSrc);
5717 break;
5718 case 32:
5719 // Only LA64 will get to here due to the size mismatch between VT and
5720 // GRLenVT, LA32 lowering is directly defined in LoongArchInstrInfo.
5721 Tmp = DAG.getNode(LoongArchISD::REVB_2W, DL, GRLenVT, NewSrc);
5722 break;
5723 }
5724 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, Tmp));
5725 break;
5726 }
5727 case ISD::BITREVERSE: {
5728 SDValue Src = N->getOperand(0);
5729 assert((VT == MVT::i8 || (VT == MVT::i32 && Subtarget.is64Bit())) &&
5730 "Unexpected custom legalization");
5731 MVT GRLenVT = Subtarget.getGRLenVT();
5732 SDValue NewSrc = DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT, Src);
5733 SDValue Tmp;
5734 switch (VT.getSizeInBits()) {
5735 default:
5736 llvm_unreachable("Unexpected operand width");
5737 case 8:
5738 Tmp = DAG.getNode(LoongArchISD::BITREV_4B, DL, GRLenVT, NewSrc);
5739 break;
5740 case 32:
5741 Tmp = DAG.getNode(LoongArchISD::BITREV_W, DL, GRLenVT, NewSrc);
5742 break;
5743 }
5744 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, Tmp));
5745 break;
5746 }
5747 case ISD::CTLZ:
5748 case ISD::CTTZ: {
5749 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5750 "Unexpected custom legalisation");
5751 Results.push_back(customLegalizeToWOp(N, DAG, 1));
5752 break;
5753 }
5755 SDValue Chain = N->getOperand(0);
5756 SDValue Op2 = N->getOperand(2);
5757 MVT GRLenVT = Subtarget.getGRLenVT();
5758 const StringRef ErrorMsgOOR = "argument out of range";
5759 const StringRef ErrorMsgReqLA64 = "requires loongarch64";
5760 const StringRef ErrorMsgReqF = "requires basic 'f' target feature";
5761
5762 switch (N->getConstantOperandVal(1)) {
5763 default:
5764 llvm_unreachable("Unexpected Intrinsic.");
5765 case Intrinsic::loongarch_movfcsr2gr: {
5766 if (!Subtarget.hasBasicF()) {
5767 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgReqF);
5768 return;
5769 }
5770 unsigned Imm = Op2->getAsZExtVal();
5771 if (!isUInt<2>(Imm)) {
5772 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5773 return;
5774 }
5775 SDValue MOVFCSR2GRResults = DAG.getNode(
5776 LoongArchISD::MOVFCSR2GR, SDLoc(N), {MVT::i64, MVT::Other},
5777 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
5778 Results.push_back(
5779 DAG.getNode(ISD::TRUNCATE, DL, VT, MOVFCSR2GRResults.getValue(0)));
5780 Results.push_back(MOVFCSR2GRResults.getValue(1));
5781 break;
5782 }
5783#define CRC_CASE_EXT_BINARYOP(NAME, NODE) \
5784 case Intrinsic::loongarch_##NAME: { \
5785 SDValue NODE = DAG.getNode( \
5786 LoongArchISD::NODE, DL, {MVT::i64, MVT::Other}, \
5787 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2), \
5788 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(3))}); \
5789 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, NODE.getValue(0))); \
5790 Results.push_back(NODE.getValue(1)); \
5791 break; \
5792 }
5793 CRC_CASE_EXT_BINARYOP(crc_w_b_w, CRC_W_B_W)
5794 CRC_CASE_EXT_BINARYOP(crc_w_h_w, CRC_W_H_W)
5795 CRC_CASE_EXT_BINARYOP(crc_w_w_w, CRC_W_W_W)
5796 CRC_CASE_EXT_BINARYOP(crcc_w_b_w, CRCC_W_B_W)
5797 CRC_CASE_EXT_BINARYOP(crcc_w_h_w, CRCC_W_H_W)
5798 CRC_CASE_EXT_BINARYOP(crcc_w_w_w, CRCC_W_W_W)
5799#undef CRC_CASE_EXT_BINARYOP
5800
5801#define CRC_CASE_EXT_UNARYOP(NAME, NODE) \
5802 case Intrinsic::loongarch_##NAME: { \
5803 SDValue NODE = DAG.getNode( \
5804 LoongArchISD::NODE, DL, {MVT::i64, MVT::Other}, \
5805 {Chain, Op2, \
5806 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(3))}); \
5807 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, NODE.getValue(0))); \
5808 Results.push_back(NODE.getValue(1)); \
5809 break; \
5810 }
5811 CRC_CASE_EXT_UNARYOP(crc_w_d_w, CRC_W_D_W)
5812 CRC_CASE_EXT_UNARYOP(crcc_w_d_w, CRCC_W_D_W)
5813#undef CRC_CASE_EXT_UNARYOP
5814#define CSR_CASE(ID) \
5815 case Intrinsic::loongarch_##ID: { \
5816 if (!Subtarget.is64Bit()) \
5817 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgReqLA64); \
5818 break; \
5819 }
5820 CSR_CASE(csrrd_d);
5821 CSR_CASE(csrwr_d);
5822 CSR_CASE(csrxchg_d);
5823 CSR_CASE(iocsrrd_d);
5824#undef CSR_CASE
5825 case Intrinsic::loongarch_csrrd_w: {
5826 unsigned Imm = Op2->getAsZExtVal();
5827 if (!isUInt<14>(Imm)) {
5828 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5829 return;
5830 }
5831 SDValue CSRRDResults =
5832 DAG.getNode(LoongArchISD::CSRRD, DL, {GRLenVT, MVT::Other},
5833 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
5834 Results.push_back(
5835 DAG.getNode(ISD::TRUNCATE, DL, VT, CSRRDResults.getValue(0)));
5836 Results.push_back(CSRRDResults.getValue(1));
5837 break;
5838 }
5839 case Intrinsic::loongarch_csrwr_w: {
5840 unsigned Imm = N->getConstantOperandVal(3);
5841 if (!isUInt<14>(Imm)) {
5842 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5843 return;
5844 }
5845 SDValue CSRWRResults =
5846 DAG.getNode(LoongArchISD::CSRWR, DL, {GRLenVT, MVT::Other},
5847 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2),
5848 DAG.getConstant(Imm, DL, GRLenVT)});
5849 Results.push_back(
5850 DAG.getNode(ISD::TRUNCATE, DL, VT, CSRWRResults.getValue(0)));
5851 Results.push_back(CSRWRResults.getValue(1));
5852 break;
5853 }
5854 case Intrinsic::loongarch_csrxchg_w: {
5855 unsigned Imm = N->getConstantOperandVal(4);
5856 if (!isUInt<14>(Imm)) {
5857 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5858 return;
5859 }
5860 SDValue CSRXCHGResults = DAG.getNode(
5861 LoongArchISD::CSRXCHG, DL, {GRLenVT, MVT::Other},
5862 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2),
5863 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(3)),
5864 DAG.getConstant(Imm, DL, GRLenVT)});
5865 Results.push_back(
5866 DAG.getNode(ISD::TRUNCATE, DL, VT, CSRXCHGResults.getValue(0)));
5867 Results.push_back(CSRXCHGResults.getValue(1));
5868 break;
5869 }
5870#define IOCSRRD_CASE(NAME, NODE) \
5871 case Intrinsic::loongarch_##NAME: { \
5872 SDValue IOCSRRDResults = \
5873 DAG.getNode(LoongArchISD::NODE, DL, {MVT::i64, MVT::Other}, \
5874 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2)}); \
5875 Results.push_back( \
5876 DAG.getNode(ISD::TRUNCATE, DL, VT, IOCSRRDResults.getValue(0))); \
5877 Results.push_back(IOCSRRDResults.getValue(1)); \
5878 break; \
5879 }
5880 IOCSRRD_CASE(iocsrrd_b, IOCSRRD_B);
5881 IOCSRRD_CASE(iocsrrd_h, IOCSRRD_H);
5882 IOCSRRD_CASE(iocsrrd_w, IOCSRRD_W);
5883#undef IOCSRRD_CASE
5884 case Intrinsic::loongarch_cpucfg: {
5885 SDValue CPUCFGResults =
5886 DAG.getNode(LoongArchISD::CPUCFG, DL, {GRLenVT, MVT::Other},
5887 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2)});
5888 Results.push_back(
5889 DAG.getNode(ISD::TRUNCATE, DL, VT, CPUCFGResults.getValue(0)));
5890 Results.push_back(CPUCFGResults.getValue(1));
5891 break;
5892 }
5893 case Intrinsic::loongarch_lddir_d: {
5894 if (!Subtarget.is64Bit()) {
5895 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgReqLA64);
5896 return;
5897 }
5898 break;
5899 }
5900 }
5901 break;
5902 }
5903 case ISD::READ_REGISTER: {
5904 if (Subtarget.is64Bit())
5905 DAG.getContext()->emitError(
5906 "On LA64, only 64-bit registers can be read.");
5907 else
5908 DAG.getContext()->emitError(
5909 "On LA32, only 32-bit registers can be read.");
5910 Results.push_back(DAG.getUNDEF(VT));
5911 Results.push_back(N->getOperand(0));
5912 break;
5913 }
5915 replaceINTRINSIC_WO_CHAINResults(N, Results, DAG, Subtarget);
5916 break;
5917 }
5918 case ISD::LROUND: {
5919 SDValue Op0 = N->getOperand(0);
5920 EVT OpVT = Op0.getValueType();
5921 RTLIB::Libcall LC =
5922 OpVT == MVT::f64 ? RTLIB::LROUND_F64 : RTLIB::LROUND_F32;
5923 MakeLibCallOptions CallOptions;
5924 CallOptions.setTypeListBeforeSoften(OpVT, MVT::i64);
5925 SDValue Result = makeLibCall(DAG, LC, MVT::i64, Op0, CallOptions, DL).first;
5926 Result = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Result);
5927 Results.push_back(Result);
5928 break;
5929 }
5930 case ISD::ATOMIC_CMP_SWAP: {
5932 break;
5933 }
5934 case ISD::TRUNCATE: {
5935 MVT VT = N->getSimpleValueType(0);
5936 if (getTypeAction(*DAG.getContext(), VT) != TypeWidenVector)
5937 return;
5938
5939 MVT WidenVT = getTypeToTransformTo(*DAG.getContext(), VT).getSimpleVT();
5940 SDValue In = N->getOperand(0);
5941 EVT InVT = In.getValueType();
5942 EVT InEltVT = InVT.getVectorElementType();
5943 EVT EltVT = VT.getVectorElementType();
5944 unsigned MinElts = VT.getVectorNumElements();
5945 unsigned WidenNumElts = WidenVT.getVectorNumElements();
5946 unsigned InBits = InVT.getSizeInBits();
5947
5948 // v8i64 -> (v8i32) -> v8i8
5949 if (InVT == MVT::v8i64 && WidenVT.is128BitVector()) {
5950 InVT = MVT::getVectorVT(MVT::getIntegerVT(256 / MinElts), MinElts);
5951 In = DAG.getNode(N->getOpcode(), DL, InVT, In);
5952 InBits = 256;
5953 }
5954
5955 // v8i32 -> v8i8 / v4i64 -> v4i16 / v4i64 -> v4i8
5956 if ((InVT == MVT::v8i32 || InVT == MVT::v4i64) &&
5957 WidenVT.is128BitVector()) {
5958 InVT = MVT::getVectorVT(MVT::getIntegerVT(128 / MinElts), MinElts);
5959 In = DAG.getNode(N->getOpcode(), DL, InVT, In);
5960 InBits = 128;
5961 InEltVT = InVT.getVectorElementType();
5962 }
5963
5964 if ((128 % InBits) == 0 && WidenVT.is128BitVector()) {
5965 if ((InEltVT.getSizeInBits() % EltVT.getSizeInBits()) == 0) {
5966 int Scale = InEltVT.getSizeInBits() / EltVT.getSizeInBits();
5967 SmallVector<int, 16> TruncMask(WidenNumElts, -1);
5968 for (unsigned I = 0; I < MinElts; ++I)
5969 TruncMask[I] = Scale * I;
5970
5971 unsigned WidenNumElts = 128 / In.getScalarValueSizeInBits();
5972 MVT SVT = In.getSimpleValueType().getScalarType();
5973 MVT VT = MVT::getVectorVT(SVT, WidenNumElts);
5974 SDValue WidenIn =
5975 DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VT, DAG.getUNDEF(VT), In,
5976 DAG.getVectorIdxConstant(0, DL));
5977 assert(isTypeLegal(WidenVT) && isTypeLegal(WidenIn.getValueType()) &&
5978 "Illegal vector type in truncation");
5979 WidenIn = DAG.getBitcast(WidenVT, WidenIn);
5980 Results.push_back(
5981 DAG.getVectorShuffle(WidenVT, DL, WidenIn, WidenIn, TruncMask));
5982 return;
5983 }
5984 }
5985
5986 break;
5987 }
5988 case ISD::SIGN_EXTEND: {
5989 // LASX has native VEXT2XV_* for sign extension.
5990 if (!Subtarget.hasExtLSX() || Subtarget.hasExtLASX())
5991 return;
5992
5993 EVT DstVT = N->getValueType(0);
5994 SDValue Src = N->getOperand(0);
5995 MVT SrcVT = Src.getSimpleValueType();
5996
5997 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
5998 unsigned DstEltBits = DstVT.getScalarSizeInBits();
5999 unsigned NumElts = DstVT.getVectorNumElements();
6000
6001 if (SrcVT.getSizeInBits() > 128)
6002 return;
6003
6004 if (!DstVT.isVector() || DstVT.getSizeInBits() <= 128)
6005 return;
6006
6007 // Legalize and extend the src to 128-bit first.
6008 if (SrcVT.getSizeInBits() < 128) {
6009 unsigned WidenSrcElts = 128 / SrcEltBits;
6010 MVT WidenSrcVT = MVT::getVectorVT(SrcVT.getScalarType(), WidenSrcElts);
6011 Src = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, WidenSrcVT,
6012 DAG.getUNDEF(WidenSrcVT), Src,
6013 DAG.getVectorIdxConstant(0, DL));
6014 SrcVT = WidenSrcVT;
6015
6016 unsigned FirstStageEltBits = 128 / NumElts;
6017 MVT FirstStageEltVT = MVT::getIntegerVT(FirstStageEltBits);
6018 MVT FirstStageVT = MVT::getVectorVT(FirstStageEltVT, NumElts);
6019 Src = DAG.getNode(ISD::SIGN_EXTEND_VECTOR_INREG, DL, FirstStageVT, Src);
6020 SrcVT = FirstStageVT;
6021 SrcEltBits = FirstStageEltBits;
6022 }
6023
6025 Blocks.push_back(Src);
6026
6027 // Sign-extend the src by using SLTI + VILVL + VILVH recursively.
6028 while (SrcEltBits < DstEltBits) {
6029 unsigned NextEltBits = SrcEltBits * 2;
6030 MVT NextEltVT = MVT::getIntegerVT(NextEltBits);
6031 unsigned CurEltsPerBlock = SrcVT.getVectorNumElements();
6032 unsigned NextEltsPerBlock = CurEltsPerBlock / 2;
6033 MVT NextBlockVT = MVT::getVectorVT(NextEltVT, NextEltsPerBlock);
6034
6035 SmallVector<SDValue, 8> NextBlocks;
6036 NextBlocks.reserve(Blocks.size() * 2);
6037 for (SDValue Block : Blocks) {
6038 SDValue Zero = DAG.getConstant(0, DL, SrcVT);
6039 SDValue Mask = DAG.getNode(ISD::SETCC, DL, SrcVT, Block, Zero,
6040 DAG.getCondCode(ISD::SETLT));
6041 SDValue LoInterleaved =
6042 DAG.getNode(LoongArchISD::VILVL, DL, SrcVT, Mask, Block);
6043 SDValue HiInterleaved =
6044 DAG.getNode(LoongArchISD::VILVH, DL, SrcVT, Mask, Block);
6045
6046 NextBlocks.push_back(DAG.getBitcast(NextBlockVT, LoInterleaved));
6047 NextBlocks.push_back(DAG.getBitcast(NextBlockVT, HiInterleaved));
6048 }
6049
6050 Blocks = std::move(NextBlocks);
6051 SrcVT = NextBlockVT;
6052 SrcEltBits = NextEltBits;
6053 }
6054
6055 Results.push_back(DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Blocks));
6056 break;
6057 }
6058 case ISD::FP_EXTEND:
6059 // FP_EXTEND may reach here due to the Custom action for v2f32 results, but
6060 // no target-specific lowering is required. Leave it unchanged and rely on
6061 // the default type legalization.
6062 break;
6063 }
6064}
6065
6066/// Try to fold: (and (xor X, -1), Y) -> (vandn X, Y).
6068 SelectionDAG &DAG) {
6069 assert(N->getOpcode() == ISD::AND && "Unexpected opcode combine into ANDN");
6070
6071 MVT VT = N->getSimpleValueType(0);
6072 if (!VT.is128BitVector() && !VT.is256BitVector())
6073 return SDValue();
6074
6075 SDValue X, Y;
6076 SDValue N0 = N->getOperand(0);
6077 SDValue N1 = N->getOperand(1);
6078
6079 if (SDValue Not = isNOT(N0, DAG)) {
6080 X = Not;
6081 Y = N1;
6082 } else if (SDValue Not = isNOT(N1, DAG)) {
6083 X = Not;
6084 Y = N0;
6085 } else
6086 return SDValue();
6087
6088 X = DAG.getBitcast(VT, X);
6089 Y = DAG.getBitcast(VT, Y);
6090 return DAG.getNode(LoongArchISD::VANDN, DL, VT, X, Y);
6091}
6092
6093static bool isConstantSplatVector(SDValue N, APInt &SplatValue,
6094 unsigned MinSizeInBits) {
6097
6098 if (!Node)
6099 return false;
6100
6101 APInt SplatUndef;
6102 unsigned SplatBitSize;
6103 bool HasAnyUndefs;
6104
6105 return Node->isConstantSplat(SplatValue, SplatUndef, SplatBitSize,
6106 HasAnyUndefs, MinSizeInBits,
6107 /*IsBigEndian=*/false);
6108}
6109
6110static SDValue matchDeinterleaveBuildVector(SDValue N, unsigned &StartIndex) {
6111 auto *BV = dyn_cast<BuildVectorSDNode>(N);
6112 if (!BV)
6113 return SDValue();
6114
6115 SDValue Src;
6116 int Start = -1;
6117
6118 for (unsigned i = 0, NumElts = BV->getNumOperands(); i < NumElts; ++i) {
6119 SDValue Op = BV->getOperand(i);
6120 if (Op.isUndef())
6121 continue;
6122 if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT)
6123 return SDValue();
6124
6125 auto *IdxC = dyn_cast<ConstantSDNode>(Op.getOperand(1));
6126 if (!IdxC)
6127 return SDValue();
6128
6129 unsigned EltIdx = IdxC->getZExtValue();
6130 if (Start < 0)
6131 Start = (int)EltIdx - (int)(i * 2);
6132 if (Start < 0 || Start > 1 || EltIdx != (unsigned)(Start + (int)(i * 2)))
6133 return SDValue();
6134
6135 SDValue CurSrc = Op.getOperand(0);
6136 if (!Src)
6137 Src = CurSrc;
6138 else if (Src != CurSrc)
6139 return SDValue();
6140 }
6141
6142 if (!Src || Start < 0)
6143 return SDValue();
6144
6145 StartIndex = (unsigned)Start;
6146 return Src;
6147}
6148
6149static SDValue
6151 const LoongArchSubtarget &Subtarget) {
6152 if (!Subtarget.hasExtLSX())
6153 return SDValue();
6154
6155 unsigned Opc = N->getOpcode();
6156 assert((Opc == ISD::ADD || Opc == ISD::SUB) && "Unexpected opcode");
6157
6158 EVT VT = N->getValueType(0);
6159 SDLoc DL(N);
6160
6161 SDValue LHS = N->getOperand(0);
6162 SDValue RHS = N->getOperand(1);
6163
6164 bool isSigned;
6165 unsigned ExtOpc = LHS.getOpcode();
6166 if (ExtOpc == ISD::SIGN_EXTEND)
6167 isSigned = true;
6168 else if (ExtOpc == ISD::ZERO_EXTEND)
6169 isSigned = false;
6170 else
6171 return SDValue();
6172
6173 if (ExtOpc != RHS.getOpcode())
6174 return SDValue();
6175
6176 if (!LHS.hasOneUse() || !RHS.hasOneUse())
6177 return SDValue();
6178
6179 unsigned OddIdx, EvenIdx;
6180 SDValue LHSVec = matchDeinterleaveBuildVector(LHS.getOperand(0), OddIdx);
6181 SDValue RHSVec = matchDeinterleaveBuildVector(RHS.getOperand(0), EvenIdx);
6182
6183 if (!LHSVec || !RHSVec)
6184 return SDValue();
6185 if (OddIdx != 1 || EvenIdx != 0)
6186 return SDValue();
6187 if (LHSVec.getValueType() != RHSVec.getValueType())
6188 return SDValue();
6189
6190 EVT SrcVT = LHSVec.getValueType();
6191 EVT SrcEltVT = SrcVT.getVectorElementType();
6192 EVT DstEltVT = VT.getVectorElementType();
6193 auto &TLI = DAG.getTargetLoweringInfo();
6194
6195 if (!TLI.isTypeLegal(VT) || !TLI.isTypeLegal(SrcVT))
6196 return SDValue();
6197 if (!SrcVT.isVector() || !VT.isVector())
6198 return SDValue();
6199 if (SrcVT.getSizeInBits() != VT.getSizeInBits())
6200 return SDValue();
6201 if (DstEltVT.getSizeInBits() != SrcEltVT.getSizeInBits() * 2)
6202 return SDValue();
6203 if (!SrcEltVT.isInteger() || SrcEltVT.getSizeInBits() > 32)
6204 return SDValue();
6205
6206 unsigned TargetOpc;
6207 if (Opc == ISD::ADD)
6208 TargetOpc = isSigned ? LoongArchISD::VHADDW : LoongArchISD::VHADDW_U;
6209 else
6210 TargetOpc = isSigned ? LoongArchISD::VHSUBW : LoongArchISD::VHSUBW_U;
6211
6212 return DAG.getNode(TargetOpc, DL, VT, LHSVec, RHSVec);
6213}
6214
6217 const LoongArchSubtarget &Subtarget) {
6218 if (SDValue V = performHorizWideningCombine(N, DAG, Subtarget))
6219 return V;
6220
6221 if (DCI.isBeforeLegalizeOps())
6222 return SDValue();
6223
6224 EVT VT = N->getValueType(0);
6225 if (!VT.isVector())
6226 return SDValue();
6227
6228 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT))
6229 return SDValue();
6230
6231 EVT EltVT = VT.getVectorElementType();
6232 if (!EltVT.isInteger())
6233 return SDValue();
6234
6235 // match:
6236 //
6237 // add
6238 // (and
6239 // (srl X, shift-1) / X
6240 // 1)
6241 // (srl/sra X, shift)
6242
6243 SDValue Add0 = N->getOperand(0);
6244 SDValue Add1 = N->getOperand(1);
6245 SDValue And;
6246 SDValue Shr;
6247
6248 if (Add0.getOpcode() == ISD::AND) {
6249 And = Add0;
6250 Shr = Add1;
6251 } else if (Add1.getOpcode() == ISD::AND) {
6252 And = Add1;
6253 Shr = Add0;
6254 } else {
6255 return SDValue();
6256 }
6257
6258 // match:
6259 //
6260 // srl/sra X, shift
6261
6262 if (Shr.getOpcode() != ISD::SRL && Shr.getOpcode() != ISD::SRA)
6263 return SDValue();
6264
6265 SDValue X = Shr.getOperand(0);
6266 SDValue Shift = Shr.getOperand(1);
6267 APInt ShiftVal;
6268
6269 if (!isConstantSplatVector(Shift, ShiftVal, EltVT.getSizeInBits()))
6270 return SDValue();
6271
6272 if (ShiftVal == 0)
6273 return SDValue();
6274
6275 // match:
6276 //
6277 // and
6278 // (srl X, shift-1) / X
6279 // 1
6280
6281 SDValue One = And.getOperand(1);
6282 APInt SplatVal;
6283
6284 if (!isConstantSplatVector(One, SplatVal, EltVT.getSizeInBits()))
6285 return SDValue();
6286
6287 if (SplatVal != 1)
6288 return SDValue();
6289
6290 if (And.getOperand(0) == X) {
6291 // match:
6292 //
6293 // shift == 1
6294
6295 if (ShiftVal != 1)
6296 return SDValue();
6297 } else {
6298 // match:
6299 //
6300 // srl X, shift-1
6301
6302 SDValue Srl = And.getOperand(0);
6303
6304 if (Srl.getOpcode() != ISD::SRL)
6305 return SDValue();
6306
6307 if (Srl.getOperand(0) != X)
6308 return SDValue();
6309
6310 // match:
6311 //
6312 // shift-1
6313
6314 SDValue ShiftMinus1 = Srl.getOperand(1);
6315
6316 if (!isConstantSplatVector(ShiftMinus1, SplatVal, EltVT.getSizeInBits()))
6317 return SDValue();
6318
6319 if (ShiftVal != (SplatVal + 1))
6320 return SDValue();
6321 }
6322
6323 // We matched a rounded right shift pattern and can lower it
6324 // to a single vector rounded shift instruction.
6325
6326 SDLoc DL(N);
6327 return DAG.getNode(Shr.getOpcode() == ISD::SRL ? LoongArchISD::VSRLR
6328 : LoongArchISD::VSRAR,
6329 DL, VT, X, Shift);
6330}
6331
6334 const LoongArchSubtarget &Subtarget) {
6335 if (DCI.isBeforeLegalizeOps())
6336 return SDValue();
6337
6338 SDValue FirstOperand = N->getOperand(0);
6339 SDValue SecondOperand = N->getOperand(1);
6340 unsigned FirstOperandOpc = FirstOperand.getOpcode();
6341 EVT ValTy = N->getValueType(0);
6342 SDLoc DL(N);
6343 uint64_t lsb, msb;
6344 unsigned SMIdx, SMLen;
6345 ConstantSDNode *CN;
6346 SDValue NewOperand;
6347 MVT GRLenVT = Subtarget.getGRLenVT();
6348
6349 if (SDValue R = combineAndNotIntoVANDN(N, DL, DAG))
6350 return R;
6351
6352 // BSTRPICK requires the 32S feature.
6353 if (!Subtarget.has32S())
6354 return SDValue();
6355
6356 // Op's second operand must be a shifted mask.
6357 if (!(CN = dyn_cast<ConstantSDNode>(SecondOperand)) ||
6358 !isShiftedMask_64(CN->getZExtValue(), SMIdx, SMLen))
6359 return SDValue();
6360
6361 if (FirstOperandOpc == ISD::SRA || FirstOperandOpc == ISD::SRL) {
6362 // Pattern match BSTRPICK.
6363 // $dst = and ((sra or srl) $src , lsb), (2**len - 1)
6364 // => BSTRPICK $dst, $src, msb, lsb
6365 // where msb = lsb + len - 1
6366
6367 // The second operand of the shift must be an immediate.
6368 if (!(CN = dyn_cast<ConstantSDNode>(FirstOperand.getOperand(1))))
6369 return SDValue();
6370
6371 lsb = CN->getZExtValue();
6372
6373 // Return if the shifted mask does not start at bit 0 or the sum of its
6374 // length and lsb exceeds the word's size.
6375 if (SMIdx != 0 || lsb + SMLen > ValTy.getSizeInBits())
6376 return SDValue();
6377
6378 NewOperand = FirstOperand.getOperand(0);
6379 } else {
6380 // Pattern match BSTRPICK.
6381 // $dst = and $src, (2**len- 1) , if len > 12
6382 // => BSTRPICK $dst, $src, msb, lsb
6383 // where lsb = 0 and msb = len - 1
6384
6385 // If the mask is <= 0xfff, andi can be used instead.
6386 if (CN->getZExtValue() <= 0xfff)
6387 return SDValue();
6388
6389 // Return if the MSB exceeds.
6390 if (SMIdx + SMLen > ValTy.getSizeInBits())
6391 return SDValue();
6392
6393 if (SMIdx > 0) {
6394 // Omit if the constant has more than 2 uses. This a conservative
6395 // decision. Whether it is a win depends on the HW microarchitecture.
6396 // However it should always be better for 1 and 2 uses.
6397 if (CN->use_size() > 2)
6398 return SDValue();
6399 // Return if the constant can be composed by a single LU12I.W.
6400 if ((CN->getZExtValue() & 0xfff) == 0)
6401 return SDValue();
6402 // Return if the constand can be composed by a single ADDI with
6403 // the zero register.
6404 if (CN->getSExtValue() >= -2048 && CN->getSExtValue() < 0)
6405 return SDValue();
6406 }
6407
6408 lsb = SMIdx;
6409 NewOperand = FirstOperand;
6410 }
6411
6412 msb = lsb + SMLen - 1;
6413 SDValue NR0 = DAG.getNode(LoongArchISD::BSTRPICK, DL, ValTy, NewOperand,
6414 DAG.getConstant(msb, DL, GRLenVT),
6415 DAG.getConstant(lsb, DL, GRLenVT));
6416 if (FirstOperandOpc == ISD::SRA || FirstOperandOpc == ISD::SRL || lsb == 0)
6417 return NR0;
6418 // Try to optimize to
6419 // bstrpick $Rd, $Rs, msb, lsb
6420 // slli $Rd, $Rd, lsb
6421 return DAG.getNode(ISD::SHL, DL, ValTy, NR0,
6422 DAG.getConstant(lsb, DL, GRLenVT));
6423}
6424
6425// Return the original source vector if N consists of the half
6426// of each 128-bit lane.
6429
6430 EVT DstVT = N.getValueType();
6431 if (!DstVT.isVector())
6432 return SDValue();
6433
6434 unsigned NumElts = DstVT.getVectorNumElements();
6435
6436 // LSX canonical form:
6437 if (N.getOpcode() == ISD::EXTRACT_SUBVECTOR) {
6438 SDValue Src = N.getOperand(0);
6439 EVT SrcVT = Src.getValueType();
6440
6441 if (!SrcVT.isVector() || !SrcVT.is128BitVector())
6442 return SDValue();
6443 if (SrcVT.getSizeInBits() != DstVT.getSizeInBits() * 2)
6444 return SDValue();
6445 if (SrcVT.getVectorNumElements() != NumElts * 2)
6446 return SDValue();
6447 if (N.getConstantOperandVal(1) != (isLow ? 0 : NumElts))
6448 return SDValue();
6449
6450 return Src;
6451 }
6452
6453 // LASX canonical form:
6454 auto *BV = dyn_cast<BuildVectorSDNode>(N);
6455 if (!BV)
6456 return SDValue();
6457
6458 if (NumElts % 2 != 0)
6459 return SDValue();
6460
6461 SDValue Src;
6462 EVT SrcVT;
6463
6464 for (unsigned I = 0; I != NumElts; ++I) {
6465 SDValue Elt = BV->getOperand(I);
6466 if (Elt.isUndef())
6467 continue;
6469 return SDValue();
6470
6471 SDValue ThisSrc = Elt.getOperand(0);
6472 SDValue Idx = Elt.getOperand(1);
6473 auto *CI = dyn_cast<ConstantSDNode>(Idx);
6474 if (!CI)
6475 return SDValue();
6476
6477 if (!Src) {
6478 Src = ThisSrc;
6479 SrcVT = Src.getValueType();
6480 if (!SrcVT.isVector())
6481 return SDValue();
6482
6483 if (!SrcVT.is256BitVector())
6484 return SDValue();
6485 if (SrcVT.getSizeInBits() != DstVT.getSizeInBits() * 2)
6486 return SDValue();
6487 if (SrcVT.getVectorNumElements() != NumElts * 2)
6488 return SDValue();
6489 } else if (ThisSrc != Src) {
6490 return SDValue();
6491 }
6492
6493 unsigned Half = NumElts / 2;
6494 unsigned ExpectedIdx = (I < Half) ? I : (I + Half);
6495 ExpectedIdx += isLow ? 0 : Half;
6496
6497 if (CI->getZExtValue() != ExpectedIdx)
6498 return SDValue();
6499 }
6500
6501 return Src;
6502}
6503
6506 const LoongArchSubtarget &Subtarget) {
6507 assert(N->getOpcode() == ISD::SHL && "Unexpected opcode");
6508
6509 EVT VT = N->getValueType(0);
6510 SDLoc DL(N);
6511
6512 SDValue LHS = N->getOperand(0);
6513 SDValue RHS = N->getOperand(1);
6514
6515 bool isSigned;
6516 unsigned ExtOpc = LHS.getOpcode();
6517 if (ExtOpc == ISD::SIGN_EXTEND)
6518 isSigned = true;
6519 else if (ExtOpc == ISD::ZERO_EXTEND)
6520 isSigned = false;
6521 else
6522 return SDValue();
6523
6524 if (!LHS.hasOneUse())
6525 return SDValue();
6526
6527 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT) ||
6528 N->getValueSizeInBits(0) != LHS->getOperand(0).getValueSizeInBits() * 2)
6529 return SDValue();
6530
6531 SDValue Vec = matchHalfOf128BitLanes(LHS.getOperand(0), /*isLow=*/true);
6532 if (!Vec)
6533 return SDValue();
6534
6535 EVT SrcVT = Vec.getValueType();
6536 EVT SrcEltVT = SrcVT.getVectorElementType();
6537 EVT DstEltVT = VT.getVectorElementType();
6538 APInt Imm;
6539 if (!isConstantSplatVector(RHS, Imm, DstEltVT.getSizeInBits()))
6540 return SDValue();
6541 if (!Imm.ult(SrcEltVT.getSizeInBits()))
6542 return SDValue();
6543
6544 unsigned Opc = isSigned ? LoongArchISD::VSLLWIL : LoongArchISD::VSLLWIL_U;
6545 SDValue Sht = DAG.getConstant(Imm.getZExtValue(), DL, Subtarget.getGRLenVT());
6546 return DAG.getNode(Opc, DL, VT, Vec, Sht);
6547}
6548
6551 const LoongArchSubtarget &Subtarget) {
6552 // BSTRPICK requires the 32S feature.
6553 if (!Subtarget.has32S())
6554 return SDValue();
6555
6556 if (DCI.isBeforeLegalizeOps())
6557 return SDValue();
6558
6559 // $dst = srl (and $src, Mask), Shamt
6560 // =>
6561 // BSTRPICK $dst, $src, MaskIdx+MaskLen-1, Shamt
6562 // when Mask is a shifted mask, and MaskIdx <= Shamt <= MaskIdx+MaskLen-1
6563 //
6564
6565 SDValue FirstOperand = N->getOperand(0);
6566 ConstantSDNode *CN;
6567 EVT ValTy = N->getValueType(0);
6568 SDLoc DL(N);
6569 MVT GRLenVT = Subtarget.getGRLenVT();
6570 unsigned MaskIdx, MaskLen;
6571 uint64_t Shamt;
6572
6573 // The first operand must be an AND and the second operand of the AND must be
6574 // a shifted mask.
6575 if (FirstOperand.getOpcode() != ISD::AND ||
6576 !(CN = dyn_cast<ConstantSDNode>(FirstOperand.getOperand(1))) ||
6577 !isShiftedMask_64(CN->getZExtValue(), MaskIdx, MaskLen))
6578 return SDValue();
6579
6580 // The second operand (shift amount) must be an immediate.
6581 if (!(CN = dyn_cast<ConstantSDNode>(N->getOperand(1))))
6582 return SDValue();
6583
6584 Shamt = CN->getZExtValue();
6585 if (MaskIdx <= Shamt && Shamt <= MaskIdx + MaskLen - 1)
6586 return DAG.getNode(LoongArchISD::BSTRPICK, DL, ValTy,
6587 FirstOperand->getOperand(0),
6588 DAG.getConstant(MaskIdx + MaskLen - 1, DL, GRLenVT),
6589 DAG.getConstant(Shamt, DL, GRLenVT));
6590
6591 return SDValue();
6592}
6593
6596 const LoongArchSubtarget &Subtarget) {
6597 if (SDValue V = performHorizWideningCombine(N, DAG, Subtarget))
6598 return V;
6599
6600 return SDValue();
6601}
6602
6603// Helper to peek through bitops/trunc/setcc to determine size of source vector.
6604// Allows BITCASTCombine to determine what size vector generated a <X x i1>.
6605static bool checkBitcastSrcVectorSize(SDValue Src, unsigned Size,
6606 unsigned Depth) {
6607 // Limit recursion.
6609 return false;
6610 switch (Src.getOpcode()) {
6611 case ISD::SETCC:
6612 case ISD::TRUNCATE:
6613 return Src.getOperand(0).getValueSizeInBits() == Size;
6614 case ISD::FREEZE:
6615 return checkBitcastSrcVectorSize(Src.getOperand(0), Size, Depth + 1);
6616 case ISD::AND:
6617 case ISD::XOR:
6618 case ISD::OR:
6619 return checkBitcastSrcVectorSize(Src.getOperand(0), Size, Depth + 1) &&
6620 checkBitcastSrcVectorSize(Src.getOperand(1), Size, Depth + 1);
6621 case ISD::SELECT:
6622 case ISD::VSELECT:
6623 return Src.getOperand(0).getScalarValueSizeInBits() == 1 &&
6624 checkBitcastSrcVectorSize(Src.getOperand(1), Size, Depth + 1) &&
6625 checkBitcastSrcVectorSize(Src.getOperand(2), Size, Depth + 1);
6626 case ISD::BUILD_VECTOR:
6627 return ISD::isBuildVectorAllZeros(Src.getNode()) ||
6628 ISD::isBuildVectorAllOnes(Src.getNode());
6629 }
6630 return false;
6631}
6632
6633// Helper to push sign extension of vXi1 SETCC result through bitops.
6635 SDValue Src, const SDLoc &DL) {
6636 switch (Src.getOpcode()) {
6637 case ISD::SETCC:
6638 case ISD::FREEZE:
6639 case ISD::TRUNCATE:
6640 case ISD::BUILD_VECTOR:
6641 return DAG.getNode(ISD::SIGN_EXTEND, DL, SExtVT, Src);
6642 case ISD::AND:
6643 case ISD::XOR:
6644 case ISD::OR:
6645 return DAG.getNode(
6646 Src.getOpcode(), DL, SExtVT,
6647 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(0), DL),
6648 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(1), DL));
6649 case ISD::SELECT:
6650 case ISD::VSELECT:
6651 return DAG.getSelect(
6652 DL, SExtVT, Src.getOperand(0),
6653 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(1), DL),
6654 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(2), DL));
6655 }
6656 llvm_unreachable("Unexpected node type for vXi1 sign extension");
6657}
6658
6659static SDValue
6662 const LoongArchSubtarget &Subtarget) {
6663 SDLoc DL(N);
6664 EVT VT = N->getValueType(0);
6665 SDValue Src = N->getOperand(0);
6666 EVT SrcVT = Src.getValueType();
6667
6668 if (Src.getOpcode() != ISD::SETCC || !Src.hasOneUse())
6669 return SDValue();
6670
6671 bool UseLASX;
6672 unsigned Opc = ISD::DELETED_NODE;
6673 EVT CmpVT = Src.getOperand(0).getValueType();
6674 EVT EltVT = CmpVT.getVectorElementType();
6675
6676 if (Subtarget.hasExtLSX() && CmpVT.getSizeInBits() == 128)
6677 UseLASX = false;
6678 else if (Subtarget.has32S() && Subtarget.hasExtLASX() &&
6679 CmpVT.getSizeInBits() == 256)
6680 UseLASX = true;
6681 else
6682 return SDValue();
6683
6684 SDValue SrcN1 = Src.getOperand(1);
6685 switch (cast<CondCodeSDNode>(Src.getOperand(2))->get()) {
6686 default:
6687 break;
6688 case ISD::SETEQ:
6689 // x == 0 => not (vmsknez.b x)
6690 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) && EltVT == MVT::i8)
6691 Opc = UseLASX ? LoongArchISD::XVMSKEQZ : LoongArchISD::VMSKEQZ;
6692 break;
6693 case ISD::SETGT:
6694 // x > -1 => vmskgez.b x
6695 if (ISD::isBuildVectorAllOnes(SrcN1.getNode()) && EltVT == MVT::i8)
6696 Opc = UseLASX ? LoongArchISD::XVMSKGEZ : LoongArchISD::VMSKGEZ;
6697 break;
6698 case ISD::SETGE:
6699 // x >= 0 => vmskgez.b x
6700 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) && EltVT == MVT::i8)
6701 Opc = UseLASX ? LoongArchISD::XVMSKGEZ : LoongArchISD::VMSKGEZ;
6702 break;
6703 case ISD::SETLT:
6704 // x < 0 => vmskltz.{b,h,w,d} x
6705 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) &&
6706 (EltVT == MVT::i8 || EltVT == MVT::i16 || EltVT == MVT::i32 ||
6707 EltVT == MVT::i64))
6708 Opc = UseLASX ? LoongArchISD::XVMSKLTZ : LoongArchISD::VMSKLTZ;
6709 break;
6710 case ISD::SETLE:
6711 // x <= -1 => vmskltz.{b,h,w,d} x
6712 if (ISD::isBuildVectorAllOnes(SrcN1.getNode()) &&
6713 (EltVT == MVT::i8 || EltVT == MVT::i16 || EltVT == MVT::i32 ||
6714 EltVT == MVT::i64))
6715 Opc = UseLASX ? LoongArchISD::XVMSKLTZ : LoongArchISD::VMSKLTZ;
6716 break;
6717 case ISD::SETNE:
6718 // x != 0 => vmsknez.b x
6719 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) && EltVT == MVT::i8)
6720 Opc = UseLASX ? LoongArchISD::XVMSKNEZ : LoongArchISD::VMSKNEZ;
6721 break;
6722 }
6723
6724 if (Opc == ISD::DELETED_NODE)
6725 return SDValue();
6726
6727 SDValue V = DAG.getNode(Opc, DL, Subtarget.getGRLenVT(), Src.getOperand(0));
6729 V = DAG.getZExtOrTrunc(V, DL, T);
6730 return DAG.getBitcast(VT, V);
6731}
6732
6735 const LoongArchSubtarget &Subtarget) {
6736 SDLoc DL(N);
6737 EVT VT = N->getValueType(0);
6738 SDValue Src = N->getOperand(0);
6739 EVT SrcVT = Src.getValueType();
6740 MVT GRLenVT = Subtarget.getGRLenVT();
6741
6742 if (!DCI.isBeforeLegalizeOps())
6743 return SDValue();
6744
6745 if (!SrcVT.isSimple() || SrcVT.getScalarType() != MVT::i1)
6746 return SDValue();
6747
6748 // Combine SETCC and BITCAST into [X]VMSK{LT,GE,NE} when possible
6749 SDValue Res = performSETCC_BITCASTCombine(N, DAG, DCI, Subtarget);
6750 if (Res)
6751 return Res;
6752
6753 // Generate vXi1 using [X]VMSKLTZ
6754 MVT SExtVT;
6755 unsigned Opc;
6756 bool UseLASX = false;
6757 bool PropagateSExt = false;
6758
6759 if (Src.getOpcode() == ISD::SETCC && Src.hasOneUse()) {
6760 EVT CmpVT = Src.getOperand(0).getValueType();
6761 if (CmpVT.getSizeInBits() > 256)
6762 return SDValue();
6763 }
6764
6765 switch (SrcVT.getSimpleVT().SimpleTy) {
6766 default:
6767 return SDValue();
6768 case MVT::v2i1:
6769 SExtVT = MVT::v2i64;
6770 break;
6771 case MVT::v4i1:
6772 SExtVT = MVT::v4i32;
6773 if (Subtarget.hasExtLASX() && checkBitcastSrcVectorSize(Src, 256, 0)) {
6774 SExtVT = MVT::v4i64;
6775 UseLASX = true;
6776 PropagateSExt = true;
6777 }
6778 break;
6779 case MVT::v8i1:
6780 SExtVT = MVT::v8i16;
6781 if (Subtarget.hasExtLASX() && checkBitcastSrcVectorSize(Src, 256, 0)) {
6782 SExtVT = MVT::v8i32;
6783 UseLASX = true;
6784 PropagateSExt = true;
6785 }
6786 break;
6787 case MVT::v16i1:
6788 SExtVT = MVT::v16i8;
6789 if (Subtarget.hasExtLASX() && checkBitcastSrcVectorSize(Src, 256, 0)) {
6790 SExtVT = MVT::v16i16;
6791 UseLASX = true;
6792 PropagateSExt = true;
6793 }
6794 break;
6795 case MVT::v32i1:
6796 SExtVT = MVT::v32i8;
6797 UseLASX = true;
6798 break;
6799 };
6800 Src = PropagateSExt ? signExtendBitcastSrcVector(DAG, SExtVT, Src, DL)
6801 : DAG.getNode(ISD::SIGN_EXTEND, DL, SExtVT, Src);
6802
6803 SDValue V;
6804 if (!Subtarget.has32S() || !Subtarget.hasExtLASX()) {
6805 if (Src.getSimpleValueType() == MVT::v32i8) {
6806 SDValue Lo, Hi;
6807 std::tie(Lo, Hi) = DAG.SplitVector(Src, DL);
6808 Lo = DAG.getNode(LoongArchISD::VMSKLTZ, DL, GRLenVT, Lo);
6809 Hi = DAG.getNode(LoongArchISD::VMSKLTZ, DL, GRLenVT, Hi);
6810 Hi = DAG.getNode(ISD::SHL, DL, GRLenVT, Hi,
6811 DAG.getShiftAmountConstant(16, GRLenVT, DL));
6812 V = DAG.getNode(ISD::OR, DL, GRLenVT, Lo, Hi);
6813 } else if (UseLASX) {
6814 return SDValue();
6815 }
6816 }
6817
6818 if (!V) {
6819 Opc = UseLASX ? LoongArchISD::XVMSKLTZ : LoongArchISD::VMSKLTZ;
6820 V = DAG.getNode(Opc, DL, GRLenVT, Src);
6821 }
6822
6824 V = DAG.getZExtOrTrunc(V, DL, T);
6825 return DAG.getBitcast(VT, V);
6826}
6827
6830 const LoongArchSubtarget &Subtarget) {
6831 MVT GRLenVT = Subtarget.getGRLenVT();
6832 EVT ValTy = N->getValueType(0);
6833 SDValue N0 = N->getOperand(0), N1 = N->getOperand(1);
6834 ConstantSDNode *CN0, *CN1;
6835 SDLoc DL(N);
6836 unsigned ValBits = ValTy.getSizeInBits();
6837 unsigned MaskIdx0, MaskLen0, MaskIdx1, MaskLen1;
6838 unsigned Shamt;
6839 bool SwapAndRetried = false;
6840
6841 // BSTRPICK requires the 32S feature.
6842 if (!Subtarget.has32S())
6843 return SDValue();
6844
6845 if (DCI.isBeforeLegalizeOps())
6846 return SDValue();
6847
6848 if (ValBits != 32 && ValBits != 64)
6849 return SDValue();
6850
6851Retry:
6852 // 1st pattern to match BSTRINS:
6853 // R = or (and X, mask0), (and (shl Y, lsb), mask1)
6854 // where mask1 = (2**size - 1) << lsb, mask0 = ~mask1
6855 // =>
6856 // R = BSTRINS X, Y, msb, lsb (where msb = lsb + size - 1)
6857 if (N0.getOpcode() == ISD::AND &&
6858 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6859 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6860 N1.getOpcode() == ISD::AND && N1.getOperand(0).getOpcode() == ISD::SHL &&
6861 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6862 isShiftedMask_64(CN1->getZExtValue(), MaskIdx1, MaskLen1) &&
6863 MaskIdx0 == MaskIdx1 && MaskLen0 == MaskLen1 &&
6864 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
6865 (Shamt = CN1->getZExtValue()) == MaskIdx0 &&
6866 (MaskIdx0 + MaskLen0 <= ValBits)) {
6867 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 1\n");
6868 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6869 N1.getOperand(0).getOperand(0),
6870 DAG.getConstant((MaskIdx0 + MaskLen0 - 1), DL, GRLenVT),
6871 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6872 }
6873
6874 // 2nd pattern to match BSTRINS:
6875 // R = or (and X, mask0), (shl (and Y, mask1), lsb)
6876 // where mask1 = (2**size - 1), mask0 = ~(mask1 << lsb)
6877 // =>
6878 // R = BSTRINS X, Y, msb, lsb (where msb = lsb + size - 1)
6879 if (N0.getOpcode() == ISD::AND &&
6880 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6881 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6882 N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::AND &&
6883 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6884 (Shamt = CN1->getZExtValue()) == MaskIdx0 &&
6885 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
6886 isShiftedMask_64(CN1->getZExtValue(), MaskIdx1, MaskLen1) &&
6887 MaskLen0 == MaskLen1 && MaskIdx1 == 0 &&
6888 (MaskIdx0 + MaskLen0 <= ValBits)) {
6889 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 2\n");
6890 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6891 N1.getOperand(0).getOperand(0),
6892 DAG.getConstant((MaskIdx0 + MaskLen0 - 1), DL, GRLenVT),
6893 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6894 }
6895
6896 // 3rd pattern to match BSTRINS:
6897 // R = or (and X, mask0), (and Y, mask1)
6898 // where ~mask0 = (2**size - 1) << lsb, mask0 & mask1 = 0
6899 // =>
6900 // R = BSTRINS X, (shr (and Y, mask1), lsb), msb, lsb
6901 // where msb = lsb + size - 1
6902 if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND &&
6903 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6904 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6905 (MaskIdx0 + MaskLen0 <= 64) &&
6906 (CN1 = dyn_cast<ConstantSDNode>(N1->getOperand(1))) &&
6907 (CN1->getSExtValue() & CN0->getSExtValue()) == 0) {
6908 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 3\n");
6909 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6910 DAG.getNode(ISD::SRL, DL, N1->getValueType(0), N1,
6911 DAG.getConstant(MaskIdx0, DL, GRLenVT)),
6912 DAG.getConstant(ValBits == 32
6913 ? (MaskIdx0 + (MaskLen0 & 31) - 1)
6914 : (MaskIdx0 + MaskLen0 - 1),
6915 DL, GRLenVT),
6916 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6917 }
6918
6919 // 4th pattern to match BSTRINS:
6920 // R = or (and X, mask), (shl Y, shamt)
6921 // where mask = (2**shamt - 1)
6922 // =>
6923 // R = BSTRINS X, Y, ValBits - 1, shamt
6924 // where ValBits = 32 or 64
6925 if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::SHL &&
6926 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6927 isShiftedMask_64(CN0->getZExtValue(), MaskIdx0, MaskLen0) &&
6928 MaskIdx0 == 0 && (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6929 (Shamt = CN1->getZExtValue()) == MaskLen0 &&
6930 (MaskIdx0 + MaskLen0 <= ValBits)) {
6931 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 4\n");
6932 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6933 N1.getOperand(0),
6934 DAG.getConstant((ValBits - 1), DL, GRLenVT),
6935 DAG.getConstant(Shamt, DL, GRLenVT));
6936 }
6937
6938 // 5th pattern to match BSTRINS:
6939 // R = or (and X, mask), const
6940 // where ~mask = (2**size - 1) << lsb, mask & const = 0
6941 // =>
6942 // R = BSTRINS X, (const >> lsb), msb, lsb
6943 // where msb = lsb + size - 1
6944 if (N0.getOpcode() == ISD::AND &&
6945 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6946 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6947 (CN1 = dyn_cast<ConstantSDNode>(N1)) &&
6948 (CN1->getSExtValue() & CN0->getSExtValue()) == 0) {
6949 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 5\n");
6950 return DAG.getNode(
6951 LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6952 DAG.getSignedConstant(CN1->getSExtValue() >> MaskIdx0, DL, ValTy),
6953 DAG.getConstant(ValBits == 32 ? (MaskIdx0 + (MaskLen0 & 31) - 1)
6954 : (MaskIdx0 + MaskLen0 - 1),
6955 DL, GRLenVT),
6956 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6957 }
6958
6959 // 6th pattern.
6960 // a = b | ((c & mask) << shamt), where all positions in b to be overwritten
6961 // by the incoming bits are known to be zero.
6962 // =>
6963 // a = BSTRINS b, c, shamt + MaskLen - 1, shamt
6964 //
6965 // Note that the 1st pattern is a special situation of the 6th, i.e. the 6th
6966 // pattern is more common than the 1st. So we put the 1st before the 6th in
6967 // order to match as many nodes as possible.
6968 ConstantSDNode *CNMask, *CNShamt;
6969 unsigned MaskIdx, MaskLen;
6970 if (N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::AND &&
6971 (CNMask = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
6972 isShiftedMask_64(CNMask->getZExtValue(), MaskIdx, MaskLen) &&
6973 MaskIdx == 0 && (CNShamt = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6974 CNShamt->getZExtValue() + MaskLen <= ValBits) {
6975 Shamt = CNShamt->getZExtValue();
6976 APInt ShMask(ValBits, CNMask->getZExtValue() << Shamt);
6977 if (ShMask.isSubsetOf(DAG.computeKnownBits(N0).Zero)) {
6978 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 6\n");
6979 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0,
6980 N1.getOperand(0).getOperand(0),
6981 DAG.getConstant(Shamt + MaskLen - 1, DL, GRLenVT),
6982 DAG.getConstant(Shamt, DL, GRLenVT));
6983 }
6984 }
6985
6986 // 7th pattern.
6987 // a = b | ((c << shamt) & shifted_mask), where all positions in b to be
6988 // overwritten by the incoming bits are known to be zero.
6989 // =>
6990 // a = BSTRINS b, c, MaskIdx + MaskLen - 1, MaskIdx
6991 //
6992 // Similarly, the 7th pattern is more common than the 2nd. So we put the 2nd
6993 // before the 7th in order to match as many nodes as possible.
6994 if (N1.getOpcode() == ISD::AND &&
6995 (CNMask = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6996 isShiftedMask_64(CNMask->getZExtValue(), MaskIdx, MaskLen) &&
6997 N1.getOperand(0).getOpcode() == ISD::SHL &&
6998 (CNShamt = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
6999 CNShamt->getZExtValue() == MaskIdx) {
7000 APInt ShMask(ValBits, CNMask->getZExtValue());
7001 if (ShMask.isSubsetOf(DAG.computeKnownBits(N0).Zero)) {
7002 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 7\n");
7003 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0,
7004 N1.getOperand(0).getOperand(0),
7005 DAG.getConstant(MaskIdx + MaskLen - 1, DL, GRLenVT),
7006 DAG.getConstant(MaskIdx, DL, GRLenVT));
7007 }
7008 }
7009
7010 // (or a, b) and (or b, a) are equivalent, so swap the operands and retry.
7011 if (!SwapAndRetried) {
7012 std::swap(N0, N1);
7013 SwapAndRetried = true;
7014 goto Retry;
7015 }
7016
7017 SwapAndRetried = false;
7018Retry2:
7019 // 8th pattern.
7020 // a = b | (c & shifted_mask), where all positions in b to be overwritten by
7021 // the incoming bits are known to be zero.
7022 // =>
7023 // a = BSTRINS b, c >> MaskIdx, MaskIdx + MaskLen - 1, MaskIdx
7024 //
7025 // Similarly, the 8th pattern is more common than the 4th and 5th patterns. So
7026 // we put it here in order to match as many nodes as possible or generate less
7027 // instructions.
7028 if (N1.getOpcode() == ISD::AND &&
7029 (CNMask = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
7030 isShiftedMask_64(CNMask->getZExtValue(), MaskIdx, MaskLen)) {
7031 APInt ShMask(ValBits, CNMask->getZExtValue());
7032 if (ShMask.isSubsetOf(DAG.computeKnownBits(N0).Zero)) {
7033 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 8\n");
7034 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0,
7035 DAG.getNode(ISD::SRL, DL, N1->getValueType(0),
7036 N1->getOperand(0),
7037 DAG.getConstant(MaskIdx, DL, GRLenVT)),
7038 DAG.getConstant(MaskIdx + MaskLen - 1, DL, GRLenVT),
7039 DAG.getConstant(MaskIdx, DL, GRLenVT));
7040 }
7041 }
7042 // Swap N0/N1 and retry.
7043 if (!SwapAndRetried) {
7044 std::swap(N0, N1);
7045 SwapAndRetried = true;
7046 goto Retry2;
7047 }
7048
7049 return SDValue();
7050}
7051
7052static bool checkValueWidth(SDValue V, ISD::LoadExtType &ExtType) {
7053 ExtType = ISD::NON_EXTLOAD;
7054
7055 switch (V.getNode()->getOpcode()) {
7056 case ISD::LOAD: {
7057 LoadSDNode *LoadNode = cast<LoadSDNode>(V.getNode());
7058 if ((LoadNode->getMemoryVT() == MVT::i8) ||
7059 (LoadNode->getMemoryVT() == MVT::i16)) {
7060 ExtType = LoadNode->getExtensionType();
7061 return true;
7062 }
7063 return false;
7064 }
7065 case ISD::AssertSext: {
7066 VTSDNode *TypeNode = cast<VTSDNode>(V.getNode()->getOperand(1));
7067 if ((TypeNode->getVT() == MVT::i8) || (TypeNode->getVT() == MVT::i16)) {
7068 ExtType = ISD::SEXTLOAD;
7069 return true;
7070 }
7071 return false;
7072 }
7073 case ISD::AssertZext: {
7074 VTSDNode *TypeNode = cast<VTSDNode>(V.getNode()->getOperand(1));
7075 if ((TypeNode->getVT() == MVT::i8) || (TypeNode->getVT() == MVT::i16)) {
7076 ExtType = ISD::ZEXTLOAD;
7077 return true;
7078 }
7079 return false;
7080 }
7081 default:
7082 return false;
7083 }
7084
7085 return false;
7086}
7087
7088// Eliminate redundant truncation and zero-extension nodes.
7089// * Case 1:
7090// +------------+ +------------+ +------------+
7091// | Input1 | | Input2 | | CC |
7092// +------------+ +------------+ +------------+
7093// | | |
7094// V V +----+
7095// +------------+ +------------+ |
7096// | TRUNCATE | | TRUNCATE | |
7097// +------------+ +------------+ |
7098// | | |
7099// V V |
7100// +------------+ +------------+ |
7101// | ZERO_EXT | | ZERO_EXT | |
7102// +------------+ +------------+ |
7103// | | |
7104// | +-------------+ |
7105// V V | |
7106// +----------------+ | |
7107// | AND | | |
7108// +----------------+ | |
7109// | | |
7110// +---------------+ | |
7111// | | |
7112// V V V
7113// +-------------+
7114// | CMP |
7115// +-------------+
7116// * Case 2:
7117// +------------+ +------------+ +-------------+ +------------+ +------------+
7118// | Input1 | | Input2 | | Constant -1 | | Constant 0 | | CC |
7119// +------------+ +------------+ +-------------+ +------------+ +------------+
7120// | | | | |
7121// V | | | |
7122// +------------+ | | | |
7123// | XOR |<---------------------+ | |
7124// +------------+ | | |
7125// | | | |
7126// V V +---------------+ |
7127// +------------+ +------------+ | |
7128// | TRUNCATE | | TRUNCATE | | +-------------------------+
7129// +------------+ +------------+ | |
7130// | | | |
7131// V V | |
7132// +------------+ +------------+ | |
7133// | ZERO_EXT | | ZERO_EXT | | |
7134// +------------+ +------------+ | |
7135// | | | |
7136// V V | |
7137// +----------------+ | |
7138// | AND | | |
7139// +----------------+ | |
7140// | | |
7141// +---------------+ | |
7142// | | |
7143// V V V
7144// +-------------+
7145// | CMP |
7146// +-------------+
7149 const LoongArchSubtarget &Subtarget) {
7150 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
7151
7152 SDNode *AndNode = N->getOperand(0).getNode();
7153 if (AndNode->getOpcode() != ISD::AND)
7154 return SDValue();
7155
7156 SDValue AndInputValue2 = AndNode->getOperand(1);
7157 if (AndInputValue2.getOpcode() != ISD::ZERO_EXTEND)
7158 return SDValue();
7159
7160 SDValue CmpInputValue = N->getOperand(1);
7161 SDValue AndInputValue1 = AndNode->getOperand(0);
7162 if (AndInputValue1.getOpcode() == ISD::XOR) {
7163 if (CC != ISD::SETEQ && CC != ISD::SETNE)
7164 return SDValue();
7165 ConstantSDNode *CN = dyn_cast<ConstantSDNode>(AndInputValue1.getOperand(1));
7166 if (!CN || !CN->isAllOnes())
7167 return SDValue();
7168 CN = dyn_cast<ConstantSDNode>(CmpInputValue);
7169 if (!CN || !CN->isZero())
7170 return SDValue();
7171 AndInputValue1 = AndInputValue1.getOperand(0);
7172 if (AndInputValue1.getOpcode() != ISD::ZERO_EXTEND)
7173 return SDValue();
7174 } else if (AndInputValue1.getOpcode() == ISD::ZERO_EXTEND) {
7175 if (AndInputValue2 != CmpInputValue)
7176 return SDValue();
7177 } else {
7178 return SDValue();
7179 }
7180
7181 SDValue TruncValue1 = AndInputValue1.getNode()->getOperand(0);
7182 if (TruncValue1.getOpcode() != ISD::TRUNCATE)
7183 return SDValue();
7184
7185 SDValue TruncValue2 = AndInputValue2.getNode()->getOperand(0);
7186 if (TruncValue2.getOpcode() != ISD::TRUNCATE)
7187 return SDValue();
7188
7189 SDValue TruncInputValue1 = TruncValue1.getNode()->getOperand(0);
7190 SDValue TruncInputValue2 = TruncValue2.getNode()->getOperand(0);
7191 ISD::LoadExtType ExtType1;
7192 ISD::LoadExtType ExtType2;
7193
7194 if (!checkValueWidth(TruncInputValue1, ExtType1) ||
7195 !checkValueWidth(TruncInputValue2, ExtType2))
7196 return SDValue();
7197
7198 if (TruncInputValue1->getValueType(0) != TruncInputValue2->getValueType(0) ||
7199 AndNode->getValueType(0) != TruncInputValue1->getValueType(0))
7200 return SDValue();
7201
7202 if ((ExtType2 != ISD::ZEXTLOAD) &&
7203 ((ExtType2 != ISD::SEXTLOAD) && (ExtType1 != ISD::SEXTLOAD)))
7204 return SDValue();
7205
7206 // These truncation and zero-extension nodes are not necessary, remove them.
7207 SDValue NewAnd = DAG.getNode(ISD::AND, SDLoc(N), AndNode->getValueType(0),
7208 TruncInputValue1, TruncInputValue2);
7209 SDValue NewSetCC =
7210 DAG.getSetCC(SDLoc(N), N->getValueType(0), NewAnd, TruncInputValue2, CC);
7211 DAG.ReplaceAllUsesWith(N, NewSetCC.getNode());
7212 return SDValue(N, 0);
7213}
7214
7215// Combine (loongarch_bitrev_w (loongarch_revb_2w X)) to loongarch_bitrev_4b.
7218 const LoongArchSubtarget &Subtarget) {
7219 if (DCI.isBeforeLegalizeOps())
7220 return SDValue();
7221
7222 SDValue Src = N->getOperand(0);
7223 if (Src.getOpcode() != LoongArchISD::REVB_2W)
7224 return SDValue();
7225
7226 return DAG.getNode(LoongArchISD::BITREV_4B, SDLoc(N), N->getValueType(0),
7227 Src.getOperand(0));
7228}
7229
7230// Perform common combines for BR_CC and SELECT_CC conditions.
7231static bool combine_CC(SDValue &LHS, SDValue &RHS, SDValue &CC, const SDLoc &DL,
7232 SelectionDAG &DAG, const LoongArchSubtarget &Subtarget) {
7233 ISD::CondCode CCVal = cast<CondCodeSDNode>(CC)->get();
7234
7235 // As far as arithmetic right shift always saves the sign,
7236 // shift can be omitted.
7237 // Fold setlt (sra X, N), 0 -> setlt X, 0 and
7238 // setge (sra X, N), 0 -> setge X, 0
7239 if (isNullConstant(RHS) && (CCVal == ISD::SETGE || CCVal == ISD::SETLT) &&
7240 LHS.getOpcode() == ISD::SRA) {
7241 LHS = LHS.getOperand(0);
7242 return true;
7243 }
7244
7245 if (!ISD::isIntEqualitySetCC(CCVal))
7246 return false;
7247
7248 // Fold ((setlt X, Y), 0, ne) -> (X, Y, lt)
7249 // Sometimes the setcc is introduced after br_cc/select_cc has been formed.
7250 if (LHS.getOpcode() == ISD::SETCC && isNullConstant(RHS) &&
7251 LHS.getOperand(0).getValueType() == Subtarget.getGRLenVT()) {
7252 // If we're looking for eq 0 instead of ne 0, we need to invert the
7253 // condition.
7254 bool Invert = CCVal == ISD::SETEQ;
7255 CCVal = cast<CondCodeSDNode>(LHS.getOperand(2))->get();
7256 if (Invert)
7257 CCVal = ISD::getSetCCInverse(CCVal, LHS.getValueType());
7258
7259 RHS = LHS.getOperand(1);
7260 LHS = LHS.getOperand(0);
7261 translateSetCCForBranch(DL, LHS, RHS, CCVal, DAG);
7262
7263 CC = DAG.getCondCode(CCVal);
7264 return true;
7265 }
7266
7267 // Fold ((srl (and X, 1<<C), C), 0, eq/ne) -> ((shl X, GRLen-1-C), 0, ge/lt)
7268 if (isNullConstant(RHS) && LHS.getOpcode() == ISD::SRL && LHS.hasOneUse() &&
7269 LHS.getOperand(1).getOpcode() == ISD::Constant) {
7270 SDValue LHS0 = LHS.getOperand(0);
7271 if (LHS0.getOpcode() == ISD::AND &&
7272 LHS0.getOperand(1).getOpcode() == ISD::Constant) {
7273 uint64_t Mask = LHS0.getConstantOperandVal(1);
7274 uint64_t ShAmt = LHS.getConstantOperandVal(1);
7275 if (isPowerOf2_64(Mask) && Log2_64(Mask) == ShAmt) {
7276 CCVal = CCVal == ISD::SETEQ ? ISD::SETGE : ISD::SETLT;
7277 CC = DAG.getCondCode(CCVal);
7278
7279 ShAmt = LHS.getValueSizeInBits() - 1 - ShAmt;
7280 LHS = LHS0.getOperand(0);
7281 if (ShAmt != 0)
7282 LHS =
7283 DAG.getNode(ISD::SHL, DL, LHS.getValueType(), LHS0.getOperand(0),
7284 DAG.getConstant(ShAmt, DL, LHS.getValueType()));
7285 return true;
7286 }
7287 }
7288 }
7289
7290 // (X, 1, setne) -> (X, 0, seteq) if we can prove X is 0/1.
7291 // This can occur when legalizing some floating point comparisons.
7292 APInt Mask = APInt::getBitsSetFrom(LHS.getValueSizeInBits(), 1);
7293 if (isOneConstant(RHS) && DAG.MaskedValueIsZero(LHS, Mask)) {
7294 CCVal = ISD::getSetCCInverse(CCVal, LHS.getValueType());
7295 CC = DAG.getCondCode(CCVal);
7296 RHS = DAG.getConstant(0, DL, LHS.getValueType());
7297 return true;
7298 }
7299
7300 // Fold ((shl (extract_vector_elt X, I), GRLen - EleBits)), 0, eq/ne) ->
7301 // ((extract_vector_elt X, I), 0, eq/ne)
7302 if (isNullConstant(RHS) && (CCVal == ISD::SETEQ || CCVal == ISD::SETNE) &&
7303 LHS.getOpcode() == ISD::SHL && LHS.hasOneUse() &&
7304 isa<ConstantSDNode>(LHS.getOperand(1))) {
7305 SDValue Ext = LHS.getOperand(0);
7306 unsigned Sht = LHS.getConstantOperandVal(1);
7307 if (Ext.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
7308 SDValue Vec = Ext.getOperand(0);
7309 unsigned EleBits = Vec.getScalarValueSizeInBits();
7310 if ((EleBits + Sht) == Subtarget.getGRLen()) {
7311 LHS = Ext;
7312 return true;
7313 }
7314 }
7315 }
7316
7317 return false;
7318}
7319
7322 const LoongArchSubtarget &Subtarget) {
7323 SDValue LHS = N->getOperand(1);
7324 SDValue RHS = N->getOperand(2);
7325 SDValue CC = N->getOperand(3);
7326 SDLoc DL(N);
7327
7328 if (combine_CC(LHS, RHS, CC, DL, DAG, Subtarget))
7329 return DAG.getNode(LoongArchISD::BR_CC, DL, N->getValueType(0),
7330 N->getOperand(0), LHS, RHS, CC, N->getOperand(4));
7331
7332 return SDValue();
7333}
7334
7337 const LoongArchSubtarget &Subtarget) {
7338 // Transform
7339 SDValue LHS = N->getOperand(0);
7340 SDValue RHS = N->getOperand(1);
7341 SDValue CC = N->getOperand(2);
7342 ISD::CondCode CCVal = cast<CondCodeSDNode>(CC)->get();
7343 SDValue TrueV = N->getOperand(3);
7344 SDValue FalseV = N->getOperand(4);
7345 SDLoc DL(N);
7346 EVT VT = N->getValueType(0);
7347
7348 // If the True and False values are the same, we don't need a select_cc.
7349 if (TrueV == FalseV)
7350 return TrueV;
7351
7352 // (select (x < 0), y, z) -> x >> (GRLEN - 1) & (y - z) + z
7353 // (select (x >= 0), y, z) -> x >> (GRLEN - 1) & (z - y) + y
7354 if (isa<ConstantSDNode>(TrueV) && isa<ConstantSDNode>(FalseV) &&
7356 (CCVal == ISD::CondCode::SETLT || CCVal == ISD::CondCode::SETGE)) {
7357 if (CCVal == ISD::CondCode::SETGE)
7358 std::swap(TrueV, FalseV);
7359
7360 int64_t TrueSImm = cast<ConstantSDNode>(TrueV)->getSExtValue();
7361 int64_t FalseSImm = cast<ConstantSDNode>(FalseV)->getSExtValue();
7362 // Only handle simm12, if it is not in this range, it can be considered as
7363 // register.
7364 if (isInt<12>(TrueSImm) && isInt<12>(FalseSImm) &&
7365 isInt<12>(TrueSImm - FalseSImm)) {
7366 SDValue SRA =
7367 DAG.getNode(ISD::SRA, DL, VT, LHS,
7368 DAG.getConstant(Subtarget.getGRLen() - 1, DL, VT));
7369 SDValue AND =
7370 DAG.getNode(ISD::AND, DL, VT, SRA,
7371 DAG.getSignedConstant(TrueSImm - FalseSImm, DL, VT));
7372 return DAG.getNode(ISD::ADD, DL, VT, AND, FalseV);
7373 }
7374
7375 if (CCVal == ISD::CondCode::SETGE)
7376 std::swap(TrueV, FalseV);
7377 }
7378
7379 if (combine_CC(LHS, RHS, CC, DL, DAG, Subtarget))
7380 return DAG.getNode(LoongArchISD::SELECT_CC, DL, N->getValueType(0),
7381 {LHS, RHS, CC, TrueV, FalseV});
7382
7383 return SDValue();
7384}
7385
7386template <unsigned N>
7388 SelectionDAG &DAG,
7389 const LoongArchSubtarget &Subtarget,
7390 bool IsSigned = false) {
7391 SDLoc DL(Node);
7392 auto *CImm = cast<ConstantSDNode>(Node->getOperand(ImmOp));
7393 // Check the ImmArg.
7394 if ((IsSigned && !isInt<N>(CImm->getSExtValue())) ||
7395 (!IsSigned && !isUInt<N>(CImm->getZExtValue()))) {
7396 DAG.getContext()->emitError(Node->getOperationName(0) +
7397 ": argument out of range.");
7398 return DAG.getNode(ISD::UNDEF, DL, Subtarget.getGRLenVT());
7399 }
7400 return DAG.getConstant(CImm->getZExtValue(), DL, Subtarget.getGRLenVT());
7401}
7402
7403template <unsigned N>
7404static SDValue lowerVectorSplatImm(SDNode *Node, unsigned ImmOp,
7405 SelectionDAG &DAG, bool IsSigned = false) {
7406 SDLoc DL(Node);
7407 EVT ResTy = Node->getValueType(0);
7408 auto *CImm = cast<ConstantSDNode>(Node->getOperand(ImmOp));
7409
7410 // Check the ImmArg.
7411 if ((IsSigned && !isInt<N>(CImm->getSExtValue())) ||
7412 (!IsSigned && !isUInt<N>(CImm->getZExtValue()))) {
7413 DAG.getContext()->emitError(Node->getOperationName(0) +
7414 ": argument out of range.");
7415 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7416 }
7417 return DAG.getConstant(
7419 IsSigned ? CImm->getSExtValue() : CImm->getZExtValue(), IsSigned),
7420 DL, ResTy);
7421}
7422
7424 SDLoc DL(Node);
7425 EVT ResTy = Node->getValueType(0);
7426 SDValue Vec = Node->getOperand(2);
7427 SDValue Mask = DAG.getConstant(Vec.getScalarValueSizeInBits() - 1, DL, ResTy);
7428 return DAG.getNode(ISD::AND, DL, ResTy, Vec, Mask);
7429}
7430
7432 SDLoc DL(Node);
7433 EVT ResTy = Node->getValueType(0);
7434 SDValue One = DAG.getConstant(1, DL, ResTy);
7435 SDValue Bit =
7436 DAG.getNode(ISD::SHL, DL, ResTy, One, truncateVecElts(Node, DAG));
7437
7438 return DAG.getNode(ISD::AND, DL, ResTy, Node->getOperand(1),
7439 DAG.getNOT(DL, Bit, ResTy));
7440}
7441
7442template <unsigned N>
7444 SDLoc DL(Node);
7445 EVT ResTy = Node->getValueType(0);
7446 auto *CImm = cast<ConstantSDNode>(Node->getOperand(2));
7447 // Check the unsigned ImmArg.
7448 if (!isUInt<N>(CImm->getZExtValue())) {
7449 DAG.getContext()->emitError(Node->getOperationName(0) +
7450 ": argument out of range.");
7451 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7452 }
7453
7454 APInt BitImm = APInt(ResTy.getScalarSizeInBits(), 1) << CImm->getAPIntValue();
7455 SDValue Mask = DAG.getConstant(~BitImm, DL, ResTy);
7456
7457 return DAG.getNode(ISD::AND, DL, ResTy, Node->getOperand(1), Mask);
7458}
7459
7460template <unsigned N>
7462 SDLoc DL(Node);
7463 EVT ResTy = Node->getValueType(0);
7464 auto *CImm = cast<ConstantSDNode>(Node->getOperand(2));
7465 // Check the unsigned ImmArg.
7466 if (!isUInt<N>(CImm->getZExtValue())) {
7467 DAG.getContext()->emitError(Node->getOperationName(0) +
7468 ": argument out of range.");
7469 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7470 }
7471
7472 APInt Imm = APInt(ResTy.getScalarSizeInBits(), 1) << CImm->getAPIntValue();
7473 SDValue BitImm = DAG.getConstant(Imm, DL, ResTy);
7474 return DAG.getNode(ISD::OR, DL, ResTy, Node->getOperand(1), BitImm);
7475}
7476
7477template <unsigned N>
7479 SDLoc DL(Node);
7480 EVT ResTy = Node->getValueType(0);
7481 auto *CImm = cast<ConstantSDNode>(Node->getOperand(2));
7482 // Check the unsigned ImmArg.
7483 if (!isUInt<N>(CImm->getZExtValue())) {
7484 DAG.getContext()->emitError(Node->getOperationName(0) +
7485 ": argument out of range.");
7486 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7487 }
7488
7489 APInt Imm = APInt(ResTy.getScalarSizeInBits(), 1) << CImm->getAPIntValue();
7490 SDValue BitImm = DAG.getConstant(Imm, DL, ResTy);
7491 return DAG.getNode(ISD::XOR, DL, ResTy, Node->getOperand(1), BitImm);
7492}
7493
7494template <unsigned W>
7496 unsigned ResOp) {
7497 unsigned Imm = N->getConstantOperandVal(2);
7498 if (!isUInt<W>(Imm)) {
7499 const StringRef ErrorMsg = "argument out of range";
7500 DAG.getContext()->emitError(N->getOperationName(0) + ": " + ErrorMsg + ".");
7501 return DAG.getUNDEF(N->getValueType(0));
7502 }
7503 SDLoc DL(N);
7504 SDValue Vec = N->getOperand(1);
7505 SDValue Idx = DAG.getConstant(Imm, DL, MVT::i32);
7507 return DAG.getNode(ResOp, DL, N->getValueType(0), Vec, Idx, EltVT);
7508}
7509
7510static SDValue
7513 const LoongArchSubtarget &Subtarget) {
7514 SDLoc DL(N);
7515 switch (N->getConstantOperandVal(0)) {
7516 default:
7517 break;
7518 case Intrinsic::loongarch_lsx_vadd_b:
7519 case Intrinsic::loongarch_lsx_vadd_h:
7520 case Intrinsic::loongarch_lsx_vadd_w:
7521 case Intrinsic::loongarch_lsx_vadd_d:
7522 case Intrinsic::loongarch_lasx_xvadd_b:
7523 case Intrinsic::loongarch_lasx_xvadd_h:
7524 case Intrinsic::loongarch_lasx_xvadd_w:
7525 case Intrinsic::loongarch_lasx_xvadd_d:
7526 return DAG.getNode(ISD::ADD, DL, N->getValueType(0), N->getOperand(1),
7527 N->getOperand(2));
7528 case Intrinsic::loongarch_lsx_vaddi_bu:
7529 case Intrinsic::loongarch_lsx_vaddi_hu:
7530 case Intrinsic::loongarch_lsx_vaddi_wu:
7531 case Intrinsic::loongarch_lsx_vaddi_du:
7532 case Intrinsic::loongarch_lasx_xvaddi_bu:
7533 case Intrinsic::loongarch_lasx_xvaddi_hu:
7534 case Intrinsic::loongarch_lasx_xvaddi_wu:
7535 case Intrinsic::loongarch_lasx_xvaddi_du:
7536 return DAG.getNode(ISD::ADD, DL, N->getValueType(0), N->getOperand(1),
7537 lowerVectorSplatImm<5>(N, 2, DAG));
7538 case Intrinsic::loongarch_lsx_vsub_b:
7539 case Intrinsic::loongarch_lsx_vsub_h:
7540 case Intrinsic::loongarch_lsx_vsub_w:
7541 case Intrinsic::loongarch_lsx_vsub_d:
7542 case Intrinsic::loongarch_lasx_xvsub_b:
7543 case Intrinsic::loongarch_lasx_xvsub_h:
7544 case Intrinsic::loongarch_lasx_xvsub_w:
7545 case Intrinsic::loongarch_lasx_xvsub_d:
7546 return DAG.getNode(ISD::SUB, DL, N->getValueType(0), N->getOperand(1),
7547 N->getOperand(2));
7548 case Intrinsic::loongarch_lsx_vsubi_bu:
7549 case Intrinsic::loongarch_lsx_vsubi_hu:
7550 case Intrinsic::loongarch_lsx_vsubi_wu:
7551 case Intrinsic::loongarch_lsx_vsubi_du:
7552 case Intrinsic::loongarch_lasx_xvsubi_bu:
7553 case Intrinsic::loongarch_lasx_xvsubi_hu:
7554 case Intrinsic::loongarch_lasx_xvsubi_wu:
7555 case Intrinsic::loongarch_lasx_xvsubi_du:
7556 return DAG.getNode(ISD::SUB, DL, N->getValueType(0), N->getOperand(1),
7557 lowerVectorSplatImm<5>(N, 2, DAG));
7558 case Intrinsic::loongarch_lsx_vneg_b:
7559 case Intrinsic::loongarch_lsx_vneg_h:
7560 case Intrinsic::loongarch_lsx_vneg_w:
7561 case Intrinsic::loongarch_lsx_vneg_d:
7562 case Intrinsic::loongarch_lasx_xvneg_b:
7563 case Intrinsic::loongarch_lasx_xvneg_h:
7564 case Intrinsic::loongarch_lasx_xvneg_w:
7565 case Intrinsic::loongarch_lasx_xvneg_d:
7566 return DAG.getNode(
7567 ISD::SUB, DL, N->getValueType(0),
7568 DAG.getConstant(
7569 APInt(N->getValueType(0).getScalarType().getSizeInBits(), 0,
7570 /*isSigned=*/true),
7571 SDLoc(N), N->getValueType(0)),
7572 N->getOperand(1));
7573 case Intrinsic::loongarch_lsx_vmax_b:
7574 case Intrinsic::loongarch_lsx_vmax_h:
7575 case Intrinsic::loongarch_lsx_vmax_w:
7576 case Intrinsic::loongarch_lsx_vmax_d:
7577 case Intrinsic::loongarch_lasx_xvmax_b:
7578 case Intrinsic::loongarch_lasx_xvmax_h:
7579 case Intrinsic::loongarch_lasx_xvmax_w:
7580 case Intrinsic::loongarch_lasx_xvmax_d:
7581 return DAG.getNode(ISD::SMAX, DL, N->getValueType(0), N->getOperand(1),
7582 N->getOperand(2));
7583 case Intrinsic::loongarch_lsx_vmax_bu:
7584 case Intrinsic::loongarch_lsx_vmax_hu:
7585 case Intrinsic::loongarch_lsx_vmax_wu:
7586 case Intrinsic::loongarch_lsx_vmax_du:
7587 case Intrinsic::loongarch_lasx_xvmax_bu:
7588 case Intrinsic::loongarch_lasx_xvmax_hu:
7589 case Intrinsic::loongarch_lasx_xvmax_wu:
7590 case Intrinsic::loongarch_lasx_xvmax_du:
7591 return DAG.getNode(ISD::UMAX, DL, N->getValueType(0), N->getOperand(1),
7592 N->getOperand(2));
7593 case Intrinsic::loongarch_lsx_vmaxi_b:
7594 case Intrinsic::loongarch_lsx_vmaxi_h:
7595 case Intrinsic::loongarch_lsx_vmaxi_w:
7596 case Intrinsic::loongarch_lsx_vmaxi_d:
7597 case Intrinsic::loongarch_lasx_xvmaxi_b:
7598 case Intrinsic::loongarch_lasx_xvmaxi_h:
7599 case Intrinsic::loongarch_lasx_xvmaxi_w:
7600 case Intrinsic::loongarch_lasx_xvmaxi_d:
7601 return DAG.getNode(ISD::SMAX, DL, N->getValueType(0), N->getOperand(1),
7602 lowerVectorSplatImm<5>(N, 2, DAG, /*IsSigned=*/true));
7603 case Intrinsic::loongarch_lsx_vmaxi_bu:
7604 case Intrinsic::loongarch_lsx_vmaxi_hu:
7605 case Intrinsic::loongarch_lsx_vmaxi_wu:
7606 case Intrinsic::loongarch_lsx_vmaxi_du:
7607 case Intrinsic::loongarch_lasx_xvmaxi_bu:
7608 case Intrinsic::loongarch_lasx_xvmaxi_hu:
7609 case Intrinsic::loongarch_lasx_xvmaxi_wu:
7610 case Intrinsic::loongarch_lasx_xvmaxi_du:
7611 return DAG.getNode(ISD::UMAX, DL, N->getValueType(0), N->getOperand(1),
7612 lowerVectorSplatImm<5>(N, 2, DAG));
7613 case Intrinsic::loongarch_lsx_vmin_b:
7614 case Intrinsic::loongarch_lsx_vmin_h:
7615 case Intrinsic::loongarch_lsx_vmin_w:
7616 case Intrinsic::loongarch_lsx_vmin_d:
7617 case Intrinsic::loongarch_lasx_xvmin_b:
7618 case Intrinsic::loongarch_lasx_xvmin_h:
7619 case Intrinsic::loongarch_lasx_xvmin_w:
7620 case Intrinsic::loongarch_lasx_xvmin_d:
7621 return DAG.getNode(ISD::SMIN, DL, N->getValueType(0), N->getOperand(1),
7622 N->getOperand(2));
7623 case Intrinsic::loongarch_lsx_vmin_bu:
7624 case Intrinsic::loongarch_lsx_vmin_hu:
7625 case Intrinsic::loongarch_lsx_vmin_wu:
7626 case Intrinsic::loongarch_lsx_vmin_du:
7627 case Intrinsic::loongarch_lasx_xvmin_bu:
7628 case Intrinsic::loongarch_lasx_xvmin_hu:
7629 case Intrinsic::loongarch_lasx_xvmin_wu:
7630 case Intrinsic::loongarch_lasx_xvmin_du:
7631 return DAG.getNode(ISD::UMIN, DL, N->getValueType(0), N->getOperand(1),
7632 N->getOperand(2));
7633 case Intrinsic::loongarch_lsx_vmini_b:
7634 case Intrinsic::loongarch_lsx_vmini_h:
7635 case Intrinsic::loongarch_lsx_vmini_w:
7636 case Intrinsic::loongarch_lsx_vmini_d:
7637 case Intrinsic::loongarch_lasx_xvmini_b:
7638 case Intrinsic::loongarch_lasx_xvmini_h:
7639 case Intrinsic::loongarch_lasx_xvmini_w:
7640 case Intrinsic::loongarch_lasx_xvmini_d:
7641 return DAG.getNode(ISD::SMIN, DL, N->getValueType(0), N->getOperand(1),
7642 lowerVectorSplatImm<5>(N, 2, DAG, /*IsSigned=*/true));
7643 case Intrinsic::loongarch_lsx_vmini_bu:
7644 case Intrinsic::loongarch_lsx_vmini_hu:
7645 case Intrinsic::loongarch_lsx_vmini_wu:
7646 case Intrinsic::loongarch_lsx_vmini_du:
7647 case Intrinsic::loongarch_lasx_xvmini_bu:
7648 case Intrinsic::loongarch_lasx_xvmini_hu:
7649 case Intrinsic::loongarch_lasx_xvmini_wu:
7650 case Intrinsic::loongarch_lasx_xvmini_du:
7651 return DAG.getNode(ISD::UMIN, DL, N->getValueType(0), N->getOperand(1),
7652 lowerVectorSplatImm<5>(N, 2, DAG));
7653 case Intrinsic::loongarch_lsx_vmul_b:
7654 case Intrinsic::loongarch_lsx_vmul_h:
7655 case Intrinsic::loongarch_lsx_vmul_w:
7656 case Intrinsic::loongarch_lsx_vmul_d:
7657 case Intrinsic::loongarch_lasx_xvmul_b:
7658 case Intrinsic::loongarch_lasx_xvmul_h:
7659 case Intrinsic::loongarch_lasx_xvmul_w:
7660 case Intrinsic::loongarch_lasx_xvmul_d:
7661 return DAG.getNode(ISD::MUL, DL, N->getValueType(0), N->getOperand(1),
7662 N->getOperand(2));
7663 case Intrinsic::loongarch_lsx_vmadd_b:
7664 case Intrinsic::loongarch_lsx_vmadd_h:
7665 case Intrinsic::loongarch_lsx_vmadd_w:
7666 case Intrinsic::loongarch_lsx_vmadd_d:
7667 case Intrinsic::loongarch_lasx_xvmadd_b:
7668 case Intrinsic::loongarch_lasx_xvmadd_h:
7669 case Intrinsic::loongarch_lasx_xvmadd_w:
7670 case Intrinsic::loongarch_lasx_xvmadd_d: {
7671 EVT ResTy = N->getValueType(0);
7672 return DAG.getNode(ISD::ADD, SDLoc(N), ResTy, N->getOperand(1),
7673 DAG.getNode(ISD::MUL, SDLoc(N), ResTy, N->getOperand(2),
7674 N->getOperand(3)));
7675 }
7676 case Intrinsic::loongarch_lsx_vmsub_b:
7677 case Intrinsic::loongarch_lsx_vmsub_h:
7678 case Intrinsic::loongarch_lsx_vmsub_w:
7679 case Intrinsic::loongarch_lsx_vmsub_d:
7680 case Intrinsic::loongarch_lasx_xvmsub_b:
7681 case Intrinsic::loongarch_lasx_xvmsub_h:
7682 case Intrinsic::loongarch_lasx_xvmsub_w:
7683 case Intrinsic::loongarch_lasx_xvmsub_d: {
7684 EVT ResTy = N->getValueType(0);
7685 return DAG.getNode(ISD::SUB, SDLoc(N), ResTy, N->getOperand(1),
7686 DAG.getNode(ISD::MUL, SDLoc(N), ResTy, N->getOperand(2),
7687 N->getOperand(3)));
7688 }
7689 case Intrinsic::loongarch_lsx_vdiv_b:
7690 case Intrinsic::loongarch_lsx_vdiv_h:
7691 case Intrinsic::loongarch_lsx_vdiv_w:
7692 case Intrinsic::loongarch_lsx_vdiv_d:
7693 case Intrinsic::loongarch_lasx_xvdiv_b:
7694 case Intrinsic::loongarch_lasx_xvdiv_h:
7695 case Intrinsic::loongarch_lasx_xvdiv_w:
7696 case Intrinsic::loongarch_lasx_xvdiv_d:
7697 return DAG.getNode(ISD::SDIV, DL, N->getValueType(0), N->getOperand(1),
7698 N->getOperand(2));
7699 case Intrinsic::loongarch_lsx_vdiv_bu:
7700 case Intrinsic::loongarch_lsx_vdiv_hu:
7701 case Intrinsic::loongarch_lsx_vdiv_wu:
7702 case Intrinsic::loongarch_lsx_vdiv_du:
7703 case Intrinsic::loongarch_lasx_xvdiv_bu:
7704 case Intrinsic::loongarch_lasx_xvdiv_hu:
7705 case Intrinsic::loongarch_lasx_xvdiv_wu:
7706 case Intrinsic::loongarch_lasx_xvdiv_du:
7707 return DAG.getNode(ISD::UDIV, DL, N->getValueType(0), N->getOperand(1),
7708 N->getOperand(2));
7709 case Intrinsic::loongarch_lsx_vmod_b:
7710 case Intrinsic::loongarch_lsx_vmod_h:
7711 case Intrinsic::loongarch_lsx_vmod_w:
7712 case Intrinsic::loongarch_lsx_vmod_d:
7713 case Intrinsic::loongarch_lasx_xvmod_b:
7714 case Intrinsic::loongarch_lasx_xvmod_h:
7715 case Intrinsic::loongarch_lasx_xvmod_w:
7716 case Intrinsic::loongarch_lasx_xvmod_d:
7717 return DAG.getNode(ISD::SREM, DL, N->getValueType(0), N->getOperand(1),
7718 N->getOperand(2));
7719 case Intrinsic::loongarch_lsx_vmod_bu:
7720 case Intrinsic::loongarch_lsx_vmod_hu:
7721 case Intrinsic::loongarch_lsx_vmod_wu:
7722 case Intrinsic::loongarch_lsx_vmod_du:
7723 case Intrinsic::loongarch_lasx_xvmod_bu:
7724 case Intrinsic::loongarch_lasx_xvmod_hu:
7725 case Intrinsic::loongarch_lasx_xvmod_wu:
7726 case Intrinsic::loongarch_lasx_xvmod_du:
7727 return DAG.getNode(ISD::UREM, DL, N->getValueType(0), N->getOperand(1),
7728 N->getOperand(2));
7729 case Intrinsic::loongarch_lsx_vand_v:
7730 case Intrinsic::loongarch_lasx_xvand_v:
7731 return DAG.getNode(ISD::AND, DL, N->getValueType(0), N->getOperand(1),
7732 N->getOperand(2));
7733 case Intrinsic::loongarch_lsx_vor_v:
7734 case Intrinsic::loongarch_lasx_xvor_v:
7735 return DAG.getNode(ISD::OR, DL, N->getValueType(0), N->getOperand(1),
7736 N->getOperand(2));
7737 case Intrinsic::loongarch_lsx_vxor_v:
7738 case Intrinsic::loongarch_lasx_xvxor_v:
7739 return DAG.getNode(ISD::XOR, DL, N->getValueType(0), N->getOperand(1),
7740 N->getOperand(2));
7741 case Intrinsic::loongarch_lsx_vnor_v:
7742 case Intrinsic::loongarch_lasx_xvnor_v: {
7743 SDValue Res = DAG.getNode(ISD::OR, DL, N->getValueType(0), N->getOperand(1),
7744 N->getOperand(2));
7745 return DAG.getNOT(DL, Res, Res->getValueType(0));
7746 }
7747 case Intrinsic::loongarch_lsx_vandi_b:
7748 case Intrinsic::loongarch_lasx_xvandi_b:
7749 return DAG.getNode(ISD::AND, DL, N->getValueType(0), N->getOperand(1),
7750 lowerVectorSplatImm<8>(N, 2, DAG));
7751 case Intrinsic::loongarch_lsx_vori_b:
7752 case Intrinsic::loongarch_lasx_xvori_b:
7753 return DAG.getNode(ISD::OR, DL, N->getValueType(0), N->getOperand(1),
7754 lowerVectorSplatImm<8>(N, 2, DAG));
7755 case Intrinsic::loongarch_lsx_vxori_b:
7756 case Intrinsic::loongarch_lasx_xvxori_b:
7757 return DAG.getNode(ISD::XOR, DL, N->getValueType(0), N->getOperand(1),
7758 lowerVectorSplatImm<8>(N, 2, DAG));
7759 case Intrinsic::loongarch_lsx_vsll_b:
7760 case Intrinsic::loongarch_lsx_vsll_h:
7761 case Intrinsic::loongarch_lsx_vsll_w:
7762 case Intrinsic::loongarch_lsx_vsll_d:
7763 case Intrinsic::loongarch_lasx_xvsll_b:
7764 case Intrinsic::loongarch_lasx_xvsll_h:
7765 case Intrinsic::loongarch_lasx_xvsll_w:
7766 case Intrinsic::loongarch_lasx_xvsll_d:
7767 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7768 truncateVecElts(N, DAG));
7769 case Intrinsic::loongarch_lsx_vslli_b:
7770 case Intrinsic::loongarch_lasx_xvslli_b:
7771 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7772 lowerVectorSplatImm<3>(N, 2, DAG));
7773 case Intrinsic::loongarch_lsx_vslli_h:
7774 case Intrinsic::loongarch_lasx_xvslli_h:
7775 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7776 lowerVectorSplatImm<4>(N, 2, DAG));
7777 case Intrinsic::loongarch_lsx_vslli_w:
7778 case Intrinsic::loongarch_lasx_xvslli_w:
7779 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7780 lowerVectorSplatImm<5>(N, 2, DAG));
7781 case Intrinsic::loongarch_lsx_vslli_d:
7782 case Intrinsic::loongarch_lasx_xvslli_d:
7783 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7784 lowerVectorSplatImm<6>(N, 2, DAG));
7785 case Intrinsic::loongarch_lsx_vsrl_b:
7786 case Intrinsic::loongarch_lsx_vsrl_h:
7787 case Intrinsic::loongarch_lsx_vsrl_w:
7788 case Intrinsic::loongarch_lsx_vsrl_d:
7789 case Intrinsic::loongarch_lasx_xvsrl_b:
7790 case Intrinsic::loongarch_lasx_xvsrl_h:
7791 case Intrinsic::loongarch_lasx_xvsrl_w:
7792 case Intrinsic::loongarch_lasx_xvsrl_d:
7793 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
7794 truncateVecElts(N, DAG));
7795 case Intrinsic::loongarch_lsx_vsrli_b:
7796 case Intrinsic::loongarch_lasx_xvsrli_b:
7797 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
7798 lowerVectorSplatImm<3>(N, 2, DAG));
7799 case Intrinsic::loongarch_lsx_vsrli_h:
7800 case Intrinsic::loongarch_lasx_xvsrli_h:
7801 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
7802 lowerVectorSplatImm<4>(N, 2, DAG));
7803 case Intrinsic::loongarch_lsx_vsrli_w:
7804 case Intrinsic::loongarch_lasx_xvsrli_w:
7805 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
7806 lowerVectorSplatImm<5>(N, 2, DAG));
7807 case Intrinsic::loongarch_lsx_vsrli_d:
7808 case Intrinsic::loongarch_lasx_xvsrli_d:
7809 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
7810 lowerVectorSplatImm<6>(N, 2, DAG));
7811 case Intrinsic::loongarch_lsx_vsra_b:
7812 case Intrinsic::loongarch_lsx_vsra_h:
7813 case Intrinsic::loongarch_lsx_vsra_w:
7814 case Intrinsic::loongarch_lsx_vsra_d:
7815 case Intrinsic::loongarch_lasx_xvsra_b:
7816 case Intrinsic::loongarch_lasx_xvsra_h:
7817 case Intrinsic::loongarch_lasx_xvsra_w:
7818 case Intrinsic::loongarch_lasx_xvsra_d:
7819 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
7820 truncateVecElts(N, DAG));
7821 case Intrinsic::loongarch_lsx_vsrai_b:
7822 case Intrinsic::loongarch_lasx_xvsrai_b:
7823 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
7824 lowerVectorSplatImm<3>(N, 2, DAG));
7825 case Intrinsic::loongarch_lsx_vsrai_h:
7826 case Intrinsic::loongarch_lasx_xvsrai_h:
7827 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
7828 lowerVectorSplatImm<4>(N, 2, DAG));
7829 case Intrinsic::loongarch_lsx_vsrai_w:
7830 case Intrinsic::loongarch_lasx_xvsrai_w:
7831 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
7832 lowerVectorSplatImm<5>(N, 2, DAG));
7833 case Intrinsic::loongarch_lsx_vsrai_d:
7834 case Intrinsic::loongarch_lasx_xvsrai_d:
7835 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
7836 lowerVectorSplatImm<6>(N, 2, DAG));
7837 case Intrinsic::loongarch_lsx_vclz_b:
7838 case Intrinsic::loongarch_lsx_vclz_h:
7839 case Intrinsic::loongarch_lsx_vclz_w:
7840 case Intrinsic::loongarch_lsx_vclz_d:
7841 case Intrinsic::loongarch_lasx_xvclz_b:
7842 case Intrinsic::loongarch_lasx_xvclz_h:
7843 case Intrinsic::loongarch_lasx_xvclz_w:
7844 case Intrinsic::loongarch_lasx_xvclz_d:
7845 return DAG.getNode(ISD::CTLZ, DL, N->getValueType(0), N->getOperand(1));
7846 case Intrinsic::loongarch_lsx_vpcnt_b:
7847 case Intrinsic::loongarch_lsx_vpcnt_h:
7848 case Intrinsic::loongarch_lsx_vpcnt_w:
7849 case Intrinsic::loongarch_lsx_vpcnt_d:
7850 case Intrinsic::loongarch_lasx_xvpcnt_b:
7851 case Intrinsic::loongarch_lasx_xvpcnt_h:
7852 case Intrinsic::loongarch_lasx_xvpcnt_w:
7853 case Intrinsic::loongarch_lasx_xvpcnt_d:
7854 return DAG.getNode(ISD::CTPOP, DL, N->getValueType(0), N->getOperand(1));
7855 case Intrinsic::loongarch_lsx_vbitclr_b:
7856 case Intrinsic::loongarch_lsx_vbitclr_h:
7857 case Intrinsic::loongarch_lsx_vbitclr_w:
7858 case Intrinsic::loongarch_lsx_vbitclr_d:
7859 case Intrinsic::loongarch_lasx_xvbitclr_b:
7860 case Intrinsic::loongarch_lasx_xvbitclr_h:
7861 case Intrinsic::loongarch_lasx_xvbitclr_w:
7862 case Intrinsic::loongarch_lasx_xvbitclr_d:
7863 return lowerVectorBitClear(N, DAG);
7864 case Intrinsic::loongarch_lsx_vbitclri_b:
7865 case Intrinsic::loongarch_lasx_xvbitclri_b:
7866 return lowerVectorBitClearImm<3>(N, DAG);
7867 case Intrinsic::loongarch_lsx_vbitclri_h:
7868 case Intrinsic::loongarch_lasx_xvbitclri_h:
7869 return lowerVectorBitClearImm<4>(N, DAG);
7870 case Intrinsic::loongarch_lsx_vbitclri_w:
7871 case Intrinsic::loongarch_lasx_xvbitclri_w:
7872 return lowerVectorBitClearImm<5>(N, DAG);
7873 case Intrinsic::loongarch_lsx_vbitclri_d:
7874 case Intrinsic::loongarch_lasx_xvbitclri_d:
7875 return lowerVectorBitClearImm<6>(N, DAG);
7876 case Intrinsic::loongarch_lsx_vbitset_b:
7877 case Intrinsic::loongarch_lsx_vbitset_h:
7878 case Intrinsic::loongarch_lsx_vbitset_w:
7879 case Intrinsic::loongarch_lsx_vbitset_d:
7880 case Intrinsic::loongarch_lasx_xvbitset_b:
7881 case Intrinsic::loongarch_lasx_xvbitset_h:
7882 case Intrinsic::loongarch_lasx_xvbitset_w:
7883 case Intrinsic::loongarch_lasx_xvbitset_d: {
7884 EVT VecTy = N->getValueType(0);
7885 SDValue One = DAG.getConstant(1, DL, VecTy);
7886 return DAG.getNode(
7887 ISD::OR, DL, VecTy, N->getOperand(1),
7888 DAG.getNode(ISD::SHL, DL, VecTy, One, truncateVecElts(N, DAG)));
7889 }
7890 case Intrinsic::loongarch_lsx_vbitseti_b:
7891 case Intrinsic::loongarch_lasx_xvbitseti_b:
7892 return lowerVectorBitSetImm<3>(N, DAG);
7893 case Intrinsic::loongarch_lsx_vbitseti_h:
7894 case Intrinsic::loongarch_lasx_xvbitseti_h:
7895 return lowerVectorBitSetImm<4>(N, DAG);
7896 case Intrinsic::loongarch_lsx_vbitseti_w:
7897 case Intrinsic::loongarch_lasx_xvbitseti_w:
7898 return lowerVectorBitSetImm<5>(N, DAG);
7899 case Intrinsic::loongarch_lsx_vbitseti_d:
7900 case Intrinsic::loongarch_lasx_xvbitseti_d:
7901 return lowerVectorBitSetImm<6>(N, DAG);
7902 case Intrinsic::loongarch_lsx_vbitrev_b:
7903 case Intrinsic::loongarch_lsx_vbitrev_h:
7904 case Intrinsic::loongarch_lsx_vbitrev_w:
7905 case Intrinsic::loongarch_lsx_vbitrev_d:
7906 case Intrinsic::loongarch_lasx_xvbitrev_b:
7907 case Intrinsic::loongarch_lasx_xvbitrev_h:
7908 case Intrinsic::loongarch_lasx_xvbitrev_w:
7909 case Intrinsic::loongarch_lasx_xvbitrev_d: {
7910 EVT VecTy = N->getValueType(0);
7911 SDValue One = DAG.getConstant(1, DL, VecTy);
7912 return DAG.getNode(
7913 ISD::XOR, DL, VecTy, N->getOperand(1),
7914 DAG.getNode(ISD::SHL, DL, VecTy, One, truncateVecElts(N, DAG)));
7915 }
7916 case Intrinsic::loongarch_lsx_vbitrevi_b:
7917 case Intrinsic::loongarch_lasx_xvbitrevi_b:
7918 return lowerVectorBitRevImm<3>(N, DAG);
7919 case Intrinsic::loongarch_lsx_vbitrevi_h:
7920 case Intrinsic::loongarch_lasx_xvbitrevi_h:
7921 return lowerVectorBitRevImm<4>(N, DAG);
7922 case Intrinsic::loongarch_lsx_vbitrevi_w:
7923 case Intrinsic::loongarch_lasx_xvbitrevi_w:
7924 return lowerVectorBitRevImm<5>(N, DAG);
7925 case Intrinsic::loongarch_lsx_vbitrevi_d:
7926 case Intrinsic::loongarch_lasx_xvbitrevi_d:
7927 return lowerVectorBitRevImm<6>(N, DAG);
7928 case Intrinsic::loongarch_lsx_vfadd_s:
7929 case Intrinsic::loongarch_lsx_vfadd_d:
7930 case Intrinsic::loongarch_lasx_xvfadd_s:
7931 case Intrinsic::loongarch_lasx_xvfadd_d:
7932 return DAG.getNode(ISD::FADD, DL, N->getValueType(0), N->getOperand(1),
7933 N->getOperand(2));
7934 case Intrinsic::loongarch_lsx_vfsub_s:
7935 case Intrinsic::loongarch_lsx_vfsub_d:
7936 case Intrinsic::loongarch_lasx_xvfsub_s:
7937 case Intrinsic::loongarch_lasx_xvfsub_d:
7938 return DAG.getNode(ISD::FSUB, DL, N->getValueType(0), N->getOperand(1),
7939 N->getOperand(2));
7940 case Intrinsic::loongarch_lsx_vfmul_s:
7941 case Intrinsic::loongarch_lsx_vfmul_d:
7942 case Intrinsic::loongarch_lasx_xvfmul_s:
7943 case Intrinsic::loongarch_lasx_xvfmul_d:
7944 return DAG.getNode(ISD::FMUL, DL, N->getValueType(0), N->getOperand(1),
7945 N->getOperand(2));
7946 case Intrinsic::loongarch_lsx_vfdiv_s:
7947 case Intrinsic::loongarch_lsx_vfdiv_d:
7948 case Intrinsic::loongarch_lasx_xvfdiv_s:
7949 case Intrinsic::loongarch_lasx_xvfdiv_d:
7950 return DAG.getNode(ISD::FDIV, DL, N->getValueType(0), N->getOperand(1),
7951 N->getOperand(2));
7952 case Intrinsic::loongarch_lsx_vfmadd_s:
7953 case Intrinsic::loongarch_lsx_vfmadd_d:
7954 case Intrinsic::loongarch_lasx_xvfmadd_s:
7955 case Intrinsic::loongarch_lasx_xvfmadd_d:
7956 return DAG.getNode(ISD::FMA, DL, N->getValueType(0), N->getOperand(1),
7957 N->getOperand(2), N->getOperand(3));
7958 case Intrinsic::loongarch_lsx_vinsgr2vr_b:
7959 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
7960 N->getOperand(1), N->getOperand(2),
7961 legalizeIntrinsicImmArg<4>(N, 3, DAG, Subtarget));
7962 case Intrinsic::loongarch_lsx_vinsgr2vr_h:
7963 case Intrinsic::loongarch_lasx_xvinsgr2vr_w:
7964 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
7965 N->getOperand(1), N->getOperand(2),
7966 legalizeIntrinsicImmArg<3>(N, 3, DAG, Subtarget));
7967 case Intrinsic::loongarch_lsx_vinsgr2vr_w:
7968 case Intrinsic::loongarch_lasx_xvinsgr2vr_d:
7969 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
7970 N->getOperand(1), N->getOperand(2),
7971 legalizeIntrinsicImmArg<2>(N, 3, DAG, Subtarget));
7972 case Intrinsic::loongarch_lsx_vinsgr2vr_d:
7973 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
7974 N->getOperand(1), N->getOperand(2),
7975 legalizeIntrinsicImmArg<1>(N, 3, DAG, Subtarget));
7976 case Intrinsic::loongarch_lsx_vreplgr2vr_b:
7977 case Intrinsic::loongarch_lsx_vreplgr2vr_h:
7978 case Intrinsic::loongarch_lsx_vreplgr2vr_w:
7979 case Intrinsic::loongarch_lsx_vreplgr2vr_d:
7980 case Intrinsic::loongarch_lasx_xvreplgr2vr_b:
7981 case Intrinsic::loongarch_lasx_xvreplgr2vr_h:
7982 case Intrinsic::loongarch_lasx_xvreplgr2vr_w:
7983 case Intrinsic::loongarch_lasx_xvreplgr2vr_d:
7984 return DAG.getNode(LoongArchISD::VREPLGR2VR, DL, N->getValueType(0),
7985 DAG.getNode(ISD::ANY_EXTEND, DL, Subtarget.getGRLenVT(),
7986 N->getOperand(1)));
7987 case Intrinsic::loongarch_lsx_vreplve_b:
7988 case Intrinsic::loongarch_lsx_vreplve_h:
7989 case Intrinsic::loongarch_lsx_vreplve_w:
7990 case Intrinsic::loongarch_lsx_vreplve_d:
7991 case Intrinsic::loongarch_lasx_xvreplve_b:
7992 case Intrinsic::loongarch_lasx_xvreplve_h:
7993 case Intrinsic::loongarch_lasx_xvreplve_w:
7994 case Intrinsic::loongarch_lasx_xvreplve_d:
7995 return DAG.getNode(LoongArchISD::VREPLVE, DL, N->getValueType(0),
7996 N->getOperand(1),
7997 DAG.getNode(ISD::ANY_EXTEND, DL, Subtarget.getGRLenVT(),
7998 N->getOperand(2)));
7999 case Intrinsic::loongarch_lsx_vpickve2gr_b:
8000 if (!Subtarget.is64Bit())
8001 return lowerVectorPickVE2GR<4>(N, DAG, LoongArchISD::VPICK_SEXT_ELT);
8002 break;
8003 case Intrinsic::loongarch_lsx_vpickve2gr_h:
8004 case Intrinsic::loongarch_lasx_xvpickve2gr_w:
8005 if (!Subtarget.is64Bit())
8006 return lowerVectorPickVE2GR<3>(N, DAG, LoongArchISD::VPICK_SEXT_ELT);
8007 break;
8008 case Intrinsic::loongarch_lsx_vpickve2gr_w:
8009 if (!Subtarget.is64Bit())
8010 return lowerVectorPickVE2GR<2>(N, DAG, LoongArchISD::VPICK_SEXT_ELT);
8011 break;
8012 case Intrinsic::loongarch_lsx_vpickve2gr_bu:
8013 if (!Subtarget.is64Bit())
8014 return lowerVectorPickVE2GR<4>(N, DAG, LoongArchISD::VPICK_ZEXT_ELT);
8015 break;
8016 case Intrinsic::loongarch_lsx_vpickve2gr_hu:
8017 case Intrinsic::loongarch_lasx_xvpickve2gr_wu:
8018 if (!Subtarget.is64Bit())
8019 return lowerVectorPickVE2GR<3>(N, DAG, LoongArchISD::VPICK_ZEXT_ELT);
8020 break;
8021 case Intrinsic::loongarch_lsx_vpickve2gr_wu:
8022 if (!Subtarget.is64Bit())
8023 return lowerVectorPickVE2GR<2>(N, DAG, LoongArchISD::VPICK_ZEXT_ELT);
8024 break;
8025 case Intrinsic::loongarch_lsx_bz_b:
8026 case Intrinsic::loongarch_lsx_bz_h:
8027 case Intrinsic::loongarch_lsx_bz_w:
8028 case Intrinsic::loongarch_lsx_bz_d:
8029 case Intrinsic::loongarch_lasx_xbz_b:
8030 case Intrinsic::loongarch_lasx_xbz_h:
8031 case Intrinsic::loongarch_lasx_xbz_w:
8032 case Intrinsic::loongarch_lasx_xbz_d:
8033 if (!Subtarget.is64Bit())
8034 return DAG.getNode(LoongArchISD::VALL_ZERO, DL, N->getValueType(0),
8035 N->getOperand(1));
8036 break;
8037 case Intrinsic::loongarch_lsx_bz_v:
8038 case Intrinsic::loongarch_lasx_xbz_v:
8039 if (!Subtarget.is64Bit())
8040 return DAG.getNode(LoongArchISD::VANY_ZERO, DL, N->getValueType(0),
8041 N->getOperand(1));
8042 break;
8043 case Intrinsic::loongarch_lsx_bnz_b:
8044 case Intrinsic::loongarch_lsx_bnz_h:
8045 case Intrinsic::loongarch_lsx_bnz_w:
8046 case Intrinsic::loongarch_lsx_bnz_d:
8047 case Intrinsic::loongarch_lasx_xbnz_b:
8048 case Intrinsic::loongarch_lasx_xbnz_h:
8049 case Intrinsic::loongarch_lasx_xbnz_w:
8050 case Intrinsic::loongarch_lasx_xbnz_d:
8051 if (!Subtarget.is64Bit())
8052 return DAG.getNode(LoongArchISD::VALL_NONZERO, DL, N->getValueType(0),
8053 N->getOperand(1));
8054 break;
8055 case Intrinsic::loongarch_lsx_bnz_v:
8056 case Intrinsic::loongarch_lasx_xbnz_v:
8057 if (!Subtarget.is64Bit())
8058 return DAG.getNode(LoongArchISD::VANY_NONZERO, DL, N->getValueType(0),
8059 N->getOperand(1));
8060 break;
8061 case Intrinsic::loongarch_lasx_concat_128_s:
8062 case Intrinsic::loongarch_lasx_concat_128_d:
8063 case Intrinsic::loongarch_lasx_concat_128:
8064 return DAG.getNode(ISD::CONCAT_VECTORS, DL, N->getValueType(0),
8065 N->getOperand(1), N->getOperand(2));
8066 }
8067 return SDValue();
8068}
8069
8072 const LoongArchSubtarget &Subtarget) {
8073 // If the input to MOVGR2FR_W_LA64 is just MOVFR2GR_S_LA64 the the
8074 // conversion is unnecessary and can be replaced with the
8075 // MOVFR2GR_S_LA64 operand.
8076 SDValue Op0 = N->getOperand(0);
8077 if (Op0.getOpcode() == LoongArchISD::MOVFR2GR_S_LA64)
8078 return Op0.getOperand(0);
8079 return SDValue();
8080}
8081
8084 const LoongArchSubtarget &Subtarget) {
8085 // If the input to MOVFR2GR_S_LA64 is just MOVGR2FR_W_LA64 then the
8086 // conversion is unnecessary and can be replaced with the MOVGR2FR_W_LA64
8087 // operand.
8088 SDValue Op0 = N->getOperand(0);
8089 if (Op0->getOpcode() == LoongArchISD::MOVGR2FR_W_LA64) {
8090 assert(Op0.getOperand(0).getValueType() == N->getSimpleValueType(0) &&
8091 "Unexpected value type!");
8092 return Op0.getOperand(0);
8093 }
8094 return SDValue();
8095}
8096
8097static SDValue
8100 MVT VT = N->getSimpleValueType(0);
8101 unsigned NumBits = VT.getScalarSizeInBits();
8102
8103 // Simplify the inputs.
8104 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
8105 APInt DemandedMask(APInt::getAllOnes(NumBits));
8106 if (TLI.SimplifyDemandedBits(SDValue(N, 0), DemandedMask, DCI))
8107 return SDValue(N, 0);
8108
8109 return SDValue();
8110}
8111
8112static SDValue
8115 const LoongArchSubtarget &Subtarget) {
8116 SDValue Op0 = N->getOperand(0);
8117 SDLoc DL(N);
8118
8119 // If the input to SplitPairF64 is just BuildPairF64 then the operation is
8120 // redundant. Instead, use BuildPairF64's operands directly.
8121 if (Op0->getOpcode() == LoongArchISD::BUILD_PAIR_F64)
8122 return DCI.CombineTo(N, Op0.getOperand(0), Op0.getOperand(1));
8123
8124 if (Op0->isUndef()) {
8125 SDValue Lo = DAG.getUNDEF(MVT::i32);
8126 SDValue Hi = DAG.getUNDEF(MVT::i32);
8127 return DCI.CombineTo(N, Lo, Hi);
8128 }
8129
8130 // It's cheaper to materialise two 32-bit integers than to load a double
8131 // from the constant pool and transfer it to integer registers through the
8132 // stack.
8134 APInt V = C->getValueAPF().bitcastToAPInt();
8135 SDValue Lo = DAG.getConstant(V.trunc(32), DL, MVT::i32);
8136 SDValue Hi = DAG.getConstant(V.lshr(32).trunc(32), DL, MVT::i32);
8137 return DCI.CombineTo(N, Lo, Hi);
8138 }
8139
8140 return SDValue();
8141}
8142
8143/// Do target-specific dag combines on LoongArchISD::VANDN nodes.
8146 const LoongArchSubtarget &Subtarget) {
8147 SDValue N0 = N->getOperand(0);
8148 SDValue N1 = N->getOperand(1);
8149 MVT VT = N->getSimpleValueType(0);
8150 SDLoc DL(N);
8151
8152 // VANDN(undef, x) -> 0
8153 // VANDN(x, undef) -> 0
8154 if (N0.isUndef() || N1.isUndef())
8155 return DAG.getConstant(0, DL, VT);
8156
8157 // VANDN(0, x) -> x
8159 return N1;
8160
8161 // VANDN(x, 0) -> 0
8163 return DAG.getConstant(0, DL, VT);
8164
8165 // VANDN(x, -1) -> NOT(x) -> XOR(x, -1)
8167 return DAG.getNOT(DL, N0, VT);
8168
8169 // Turn VANDN back to AND if input is inverted.
8170 if (SDValue Not = isNOT(N0, DAG))
8171 return DAG.getNode(ISD::AND, DL, VT, DAG.getBitcast(VT, Not), N1);
8172
8173 // Folds for better commutativity:
8174 if (N1->hasOneUse()) {
8175 // VANDN(x,NOT(y)) -> AND(NOT(x),NOT(y)) -> NOT(OR(X,Y)).
8176 if (SDValue Not = isNOT(N1, DAG))
8177 return DAG.getNOT(
8178 DL, DAG.getNode(ISD::OR, DL, VT, N0, DAG.getBitcast(VT, Not)), VT);
8179
8180 // VANDN(x, SplatVector(Imm)) -> AND(NOT(x), NOT(SplatVector(~Imm)))
8181 // -> NOT(OR(x, SplatVector(-Imm))
8182 // Combination is performed only when VT is v16i8/v32i8, using `vnori.b` to
8183 // gain benefits.
8184 if (!DCI.isBeforeLegalizeOps() && (VT == MVT::v16i8 || VT == MVT::v32i8) &&
8185 N1.getOpcode() == ISD::BUILD_VECTOR) {
8186 if (SDValue SplatValue =
8187 cast<BuildVectorSDNode>(N1.getNode())->getSplatValue()) {
8188 if (!N1->isOnlyUserOf(SplatValue.getNode()))
8189 return SDValue();
8190
8191 if (auto *C = dyn_cast<ConstantSDNode>(SplatValue)) {
8192 uint8_t NCVal = static_cast<uint8_t>(~(C->getSExtValue()));
8193 SDValue Not =
8194 DAG.getSplat(VT, DL, DAG.getTargetConstant(NCVal, DL, MVT::i8));
8195 return DAG.getNOT(
8196 DL, DAG.getNode(ISD::OR, DL, VT, N0, DAG.getBitcast(VT, Not)),
8197 VT);
8198 }
8199 }
8200 }
8201 }
8202
8203 return SDValue();
8204}
8205
8206static SDValue ExtendSrcToDst(SDNode *N, SelectionDAG &DAG, unsigned ExtendOp) {
8207 SDLoc DL(N);
8208 EVT VT = N->getValueType(0);
8209 SDValue Src = N->getOperand(0);
8210 EVT SrcVT = Src.getValueType();
8211
8212 unsigned DstElts = VT.getVectorNumElements();
8213 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
8214 unsigned DstEltBits = VT.getScalarSizeInBits();
8215
8216 if (SrcEltBits >= DstEltBits)
8217 return SDValue();
8218
8219 MVT WidenEltVT = MVT::getIntegerVT(DstEltBits);
8220 MVT WidenSrcVT = MVT::getVectorVT(WidenEltVT, DstElts);
8221
8222 SDValue Extend = DAG.getNode(ExtendOp, DL, WidenSrcVT, Src);
8223 return DAG.getNode(N->getOpcode(), DL, VT, Extend);
8224}
8225
8226// Merge two 64 to 32 convert instructions into one,
8227// e.g.
8228// vffint.s.l $vr0, $vr1, $vr2
8229// will convert 4 si64 into 4 float at once.
8230// or
8231// vftintrz.w.d $vr0, $vr1, $vr2
8232// which will convert 4 double into 4 si32 at once.
8233// also deal with their 256-bits LASX version.
8234static SDValue MergeBlocksConvert(SDNode *N, SelectionDAG &DAG, unsigned Opcode,
8235 unsigned BlockBits) {
8236 SDLoc DL(N);
8237 MVT DstVT = N->getSimpleValueType(0);
8238 SDValue Src = N->getOperand(0);
8239 MVT SrcVT = Src.getSimpleValueType();
8240 unsigned SrcBits = SrcVT.getSizeInBits();
8241
8243 unsigned BlockNumElts = BlockBits / SrcVT.getScalarSizeInBits();
8244 MVT BlockVT = MVT::getVectorVT(SrcVT.getScalarType(), BlockNumElts);
8245 if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
8246 Src.getOperand(0).getValueType() == BlockVT) {
8247 for (unsigned i = 0; i < Src.getNumOperands(); ++i)
8248 Blocks.push_back(Src.getOperand(i));
8249 } else if (SrcBits > BlockBits) {
8250 // Wider than one register: extract each BlockBits-wide sub-vector.
8251 for (unsigned i = 0; i < SrcBits / BlockBits; ++i)
8252 Blocks.push_back(
8253 DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, BlockVT, Src,
8254 DAG.getVectorIdxConstant(i * BlockNumElts, DL)));
8255 } else {
8256 BlockBits = SrcBits;
8257 Blocks.push_back(Src);
8258 }
8259
8260 MVT NativeVecVT = MVT::getVectorVT(DstVT.getScalarType(),
8261 BlockBits / DstVT.getScalarSizeInBits());
8263 for (unsigned i = 0; i < Blocks.size(); i += 2) {
8264 SDValue Lo = Blocks[i];
8265 SDValue Hi = Blocks.size() > 1 ? Blocks[i + 1] : Lo;
8266 SDValue Res = DAG.getNode(Opcode, DL, NativeVecVT, Hi, Lo);
8267
8268 if (BlockBits == 256) {
8269 SDValue Undef = DAG.getUNDEF(NativeVecVT);
8270 SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
8271 Res = DAG.getVectorShuffle(NativeVecVT, DL, Res, Undef, Mask);
8272 Res = DAG.getBitcast(NativeVecVT, Res);
8273 }
8274
8275 Parts.push_back(Res);
8276 }
8277
8278 if (Blocks.size() == 1)
8279 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, DstVT, Parts[0],
8280 DAG.getVectorIdxConstant(0, DL));
8281 return DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Parts);
8282}
8283
8286 const LoongArchSubtarget &Subtarget) {
8287 SDLoc DL(N);
8288 EVT VT = N->getValueType(0);
8289 SDValue Src = N->getOperand(0);
8290 EVT SrcVT = Src.getValueType();
8291
8292 if (VT.isVector()) {
8293 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
8294 unsigned DstEltBits = VT.getScalarSizeInBits();
8295 unsigned NumElts = VT.getVectorNumElements();
8296 unsigned BlockBits = Subtarget.hasExtLASX() ? 256 : 128;
8297
8298 // Sign-extend src to avoid scalarization.
8299 if (SrcEltBits <= DstEltBits)
8300 return ExtendSrcToDst(N, DAG, ISD::SIGN_EXTEND);
8301
8302 if (SrcEltBits != 64 || DstEltBits != 32 || !isPowerOf2_32(NumElts))
8303 return SDValue();
8304
8305 if (!SrcVT.isSimple() || !VT.isSimple())
8306 return SDValue();
8307
8308 // Combine [x]vffint.s.l for vector si64 to float conversion.
8309 return MergeBlocksConvert(N, DAG, LoongArchISD::VFFINT, BlockBits);
8310 }
8311
8312 if (VT != MVT::f32 && VT != MVT::f64)
8313 return SDValue();
8314 if (VT == MVT::f32 && !Subtarget.hasBasicF())
8315 return SDValue();
8316 if (VT == MVT::f64 && !Subtarget.hasBasicD())
8317 return SDValue();
8318
8319 // Only optimize when the source and destination types have the same width.
8320 if (VT.getSizeInBits() != N->getOperand(0).getValueSizeInBits())
8321 return SDValue();
8322
8323 // If the result of an integer load is only used by an integer-to-float
8324 // conversion, use a fp load instead. This eliminates an integer-to-float-move
8325 // (movgr2fr) instruction.
8326 if (ISD::isNormalLoad(Src.getNode()) && Src.hasOneUse() &&
8327 // Do not change the width of a volatile load. This condition check is
8328 // inspired by AArch64.
8329 !cast<LoadSDNode>(Src)->isVolatile()) {
8330 LoadSDNode *LN0 = cast<LoadSDNode>(Src);
8331 SDValue Load = DAG.getLoad(VT, DL, LN0->getChain(), LN0->getBasePtr(),
8332 LN0->getPointerInfo(), LN0->getAlign(),
8333 LN0->getMemOperand()->getFlags());
8334
8335 // Make sure successors of the original load stay after it by updating them
8336 // to use the new Chain.
8337 DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), Load.getValue(1));
8338 return DAG.getNode(LoongArchISD::SITOF, SDLoc(N), VT, Load);
8339 }
8340
8341 return SDValue();
8342}
8343
8346 const LoongArchSubtarget &Subtarget) {
8347 SDLoc DL(N);
8348 EVT VT = N->getValueType(0);
8349
8350 // Zero-extend src to avoid scalarization.
8351 if (VT.isVector())
8352 return ExtendSrcToDst(N, DAG, ISD::ZERO_EXTEND);
8353
8354 return SDValue();
8355}
8356
8357// Using [X]VFTINTRZ_W_D for double to signed 32-bit integer conversion.
8358// For example:
8359// v4i32 = fp_to_sint (concat_vectors v2f64, v2f64)
8360// Can be combined into:
8361// v4i32 = VFTINTRZ_W_D v2f64. v2f64
8364 const LoongArchSubtarget &Subtarget) {
8365 if (!Subtarget.hasExtLSX())
8366 return SDValue();
8367
8368 SDLoc DL(N);
8369 EVT DstVT = N->getValueType(0);
8370 SDValue Src = N->getOperand(0);
8371 EVT SrcVT = Src.getValueType();
8372 bool IsSigned = N->getOpcode() == ISD::FP_TO_SINT;
8373
8374 if (!DstVT.isVector() || !DstVT.isSimple() || !SrcVT.isSimple())
8375 return SDValue();
8376
8377 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
8378 unsigned SrcBits = SrcVT.getSizeInBits();
8379 unsigned DstEltBits = DstVT.getScalarSizeInBits();
8380 unsigned NumElts = DstVT.getVectorNumElements();
8381 unsigned BlockBits = Subtarget.hasExtLASX() ? 256 : 128;
8382
8383 if (!isPowerOf2_32(NumElts) || !isPowerOf2_32(DstEltBits))
8384 return SDValue();
8385
8386 if (SrcBits % BlockBits != 0 && SrcBits != 128)
8387 return SDValue();
8388
8389 if (DstEltBits < 32) {
8390 MVT PromoteVT = MVT::getVectorVT(MVT::getIntegerVT(32), NumElts);
8391 SDValue Conv = DAG.getNode(N->getOpcode(), DL, PromoteVT, Src);
8392 return DAG.getNode(ISD::TRUNCATE, DL, DstVT, Conv);
8393 }
8394
8395 if (SrcEltBits != 64 || DstEltBits != 32)
8396 return SDValue();
8397
8398 if (!IsSigned) {
8399 // LASX already has pattern for double convert to uint32.
8400 if (Subtarget.hasExtLASX())
8401 return SDValue();
8402 MVT TmpVT = MVT::getVectorVT(MVT::i64, NumElts);
8403 SDValue Tmp = DAG.getNode(ISD::FP_TO_SINT, DL, TmpVT, Src);
8404 return DAG.getNode(ISD::TRUNCATE, DL, DstVT, Tmp);
8405 }
8406
8407 return MergeBlocksConvert(N, DAG, LoongArchISD::VFTINTRZ, BlockBits);
8408}
8409
8410// Try to widen AND, OR and XOR nodes to VT in order to remove casts around
8411// logical operations, like in the example below.
8412// or (and (truncate x, truncate y)),
8413// (xor (truncate z, build_vector (constants)))
8414// Given a target type \p VT, we generate
8415// or (and x, y), (xor z, zext(build_vector (constants)))
8416// given x, y and z are of type \p VT. We can do so, if operands are either
8417// truncates from VT types, the second operand is a vector of constants, can
8418// be recursively promoted or is an existing extension we can extend further.
8420 SelectionDAG &DAG,
8421 const LoongArchSubtarget &Subtarget,
8422 unsigned Depth) {
8423 // Limit recursion to avoid excessive compile times.
8425 return SDValue();
8426
8427 if (!ISD::isBitwiseLogicOp(N.getOpcode()))
8428 return SDValue();
8429
8430 SDValue N0 = N.getOperand(0);
8431 SDValue N1 = N.getOperand(1);
8432
8433 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
8434 if (!TLI.isOperationLegalOrPromote(N.getOpcode(), VT))
8435 return SDValue();
8436
8437 if (SDValue NN0 =
8438 PromoteMaskArithmetic(N0, DL, VT, DAG, Subtarget, Depth + 1))
8439 N0 = NN0;
8440