Source file src/cmd/compile/internal/ssa/_gen/ARM64Ops.go

     1  // Copyright 2016 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package main
     6  
     7  import "strings"
     8  
     9  // Notes:
    10  //  - Integer types live in the low portion of registers. Upper portions are junk.
    11  //  - Boolean types use the low-order byte of a register. 0=false, 1=true.
    12  //    Upper bytes are junk.
    13  //  - *const instructions may use a constant larger than the instruction can encode.
    14  //    In this case the assembler expands to multiple instructions and uses tmp
    15  //    register (R27).
    16  //  - All 32-bit Ops will zero the upper 32 bits of the destination register.
    17  
    18  // Suffixes encode the bit width of various instructions.
    19  // D (double word) = 64 bit
    20  // W (word)        = 32 bit
    21  // H (half word)   = 16 bit
    22  // HU              = 16 bit unsigned
    23  // B (byte)        = 8 bit
    24  // BU              = 8 bit unsigned
    25  // S (single)      = 32 bit float
    26  // D (double)      = 64 bit float
    27  
    28  // Note: registers not used in regalloc are not included in this list,
    29  // so that regmask stays within int64
    30  // Be careful when hand coding regmasks.
    31  var regNamesARM64 = []string{
    32  	"R0",
    33  	"R1",
    34  	"R2",
    35  	"R3",
    36  	"R4",
    37  	"R5",
    38  	"R6",
    39  	"R7",
    40  	"R8",
    41  	"R9",
    42  	"R10",
    43  	"R11",
    44  	"R12",
    45  	"R13",
    46  	"R14",
    47  	"R15",
    48  	"R16",
    49  	"R17",
    50  	// R18 = platform register, not used
    51  	"R19",
    52  	"R20",
    53  	"R21",
    54  	"R22",
    55  	"R23",
    56  	"R24",
    57  	"R25",
    58  	"R26",
    59  	// R27 = REGTMP not used in regalloc
    60  	"g",    // aka R28
    61  	"R29",  // frame pointer, not used
    62  	"R30",  // aka REGLINK
    63  	"ZERO", // zero register (aka R31)
    64  	"SP",   // stack pointer (aka R31)
    65  
    66  	// Note: both ZERO and SP are register number 31!
    67  	// What r31 means in a particular instruction depends on
    68  	// the instruction.  Generally, for arguments of instructions
    69  	// which are addresses to load or store from, r31 means SP.
    70  	// In other instructions, r31 means ZERO. But there are
    71  	// exceptions.
    72  	// See https://stackoverflow.com/questions/61532867
    73  	// This does not have much of an effect here, as the
    74  	// cmd/internal/obj/arm64 interface treats them as two
    75  	// different registers and picks the right instruction
    76  	// that encodes what r31 means. But see issue 71651.
    77  
    78  	"F0",
    79  	"F1",
    80  	"F2",
    81  	"F3",
    82  	"F4",
    83  	"F5",
    84  	"F6",
    85  	"F7",
    86  	"F8",
    87  	"F9",
    88  	"F10",
    89  	"F11",
    90  	"F12",
    91  	"F13",
    92  	"F14",
    93  	"F15",
    94  	"F16",
    95  	"F17",
    96  	"F18",
    97  	"F19",
    98  	"F20",
    99  	"F21",
   100  	"F22",
   101  	"F23",
   102  	"F24",
   103  	"F25",
   104  	"F26",
   105  	"F27",
   106  	"F28",
   107  	"F29",
   108  	"F30",
   109  	"F31",
   110  
   111  	"P0",
   112  	"P1",
   113  	"P2",
   114  	"P3",
   115  	"P4",
   116  	"P5",
   117  	"P6",
   118  	"P7",
   119  	"P8",
   120  	"P9",
   121  	"P10",
   122  	"P11",
   123  	"P12",
   124  	"P13",
   125  	"P14",
   126  	"P15",
   127  
   128  	// If you add registers, update asyncPreempt in runtime.
   129  
   130  	// pseudo-registers
   131  	"SB",
   132  }
   133  
   134  func init() {
   135  	// Make map from reg names to reg integers.
   136  	if len(regNamesARM64) > 128 {
   137  		panic("too many registers")
   138  	}
   139  	num := map[string]int{}
   140  	for i, name := range regNamesARM64 {
   141  		num[name] = i
   142  	}
   143  	buildReg := func(s string) regMask {
   144  		m := regMask{}
   145  		for _, r := range strings.Split(s, " ") {
   146  			if n, ok := num[r]; ok {
   147  				m = m.addReg(uint(n))
   148  				continue
   149  			}
   150  			panic("register " + r + " not found")
   151  		}
   152  		return m
   153  	}
   154  
   155  	// Common individual register masks
   156  	var (
   157  		gp         = buildReg("R0 R1 R2 R3 R4 R5 R6 R7 R8 R9 R10 R11 R12 R13 R14 R15 R16 R17 R19 R20 R21 R22 R23 R24 R25 R26 R30")
   158  		gpg        = gp.union(buildReg("g"))
   159  		gpsp       = gp.union(buildReg("SP"))
   160  		gpspg      = gpg.union(buildReg("SP"))
   161  		gpspsbg    = gpspg.union(buildReg("SB"))
   162  		fp         = buildReg("F0 F1 F2 F3 F4 F5 F6 F7 F8 F9 F10 F11 F12 F13 F14 F15 F16 F17 F18 F19 F20 F21 F22 F23 F24 F25 F26 F27 F28 F29 F30 F31")
   163  		pred       = buildReg("P0 P1 P2 P3 P4 P5 P6 P7 P8 P9 P10 P11 P12 P13 P14 P15")
   164  		callerSave = gp.union(fp).union(pred).union(buildReg("g")) // runtime.setg (and anything calling it) may clobber g
   165  		r25        = buildReg("R25")
   166  		r24to25    = buildReg("R24 R25")
   167  		f16to17    = buildReg("F16 F17")
   168  		rz         = buildReg("ZERO")
   169  		first16    = buildReg("R0 R1 R2 R3 R4 R5 R6 R7 R8 R9 R10 R11 R12 R13 R14 R15")
   170  	)
   171  	// Common regInfo
   172  	var (
   173  		gp01           = regInfo{inputs: nil, outputs: []regMask{gp}}
   174  		gp0flags1      = regInfo{inputs: []regMask{regMask{}}, outputs: []regMask{gp}}
   175  		gp11           = regInfo{inputs: []regMask{gpg}, outputs: []regMask{gp}}
   176  		gp11sp         = regInfo{inputs: []regMask{gpspg}, outputs: []regMask{gp}}
   177  		gp1flags       = regInfo{inputs: []regMask{gpg}}
   178  		gp1flagsflags  = regInfo{inputs: []regMask{gpg}}
   179  		gp1flags1      = regInfo{inputs: []regMask{gpg}, outputs: []regMask{gp}}
   180  		gp11flags      = regInfo{inputs: []regMask{gpg}, outputs: []regMask{gp, regMask{}}}
   181  		gp21           = regInfo{inputs: []regMask{gpg, gpg}, outputs: []regMask{gp}}
   182  		gp21nog        = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp}}
   183  		gp21flags      = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp, regMask{}}}
   184  		gp2flags       = regInfo{inputs: []regMask{gpg, gpg}}
   185  		gp2flagsflags  = regInfo{inputs: []regMask{gpg, gpg}}
   186  		gp2flags1      = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp}}
   187  		gp2flags1flags = regInfo{inputs: []regMask{gp, gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   188  		gp2load        = regInfo{inputs: []regMask{gpspsbg, gpg}, outputs: []regMask{gp}}
   189  		gp31           = regInfo{inputs: []regMask{gpg, gpg, gpg}, outputs: []regMask{gp}}
   190  		gpload         = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{gp}}
   191  		gpload2        = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{gpg, gpg}}
   192  		gpstore        = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz)}}
   193  		gpstore2       = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz), gpg.union(rz)}}
   194  		gpxchg         = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz)}, outputs: []regMask{gp}}
   195  		gpcas          = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz), gpg.union(rz)}, outputs: []regMask{gp}}
   196  		fp01           = regInfo{inputs: nil, outputs: []regMask{fp}}
   197  		pred01         = regInfo{inputs: nil, outputs: []regMask{pred}}
   198  		pred2pred      = regInfo{inputs: []regMask{pred, pred}, outputs: []regMask{pred}}
   199  		pred3pred      = regInfo{inputs: []regMask{pred, pred, pred}, outputs: []regMask{pred}}
   200  		pred2flags     = regInfo{inputs: []regMask{pred, pred}}
   201  		fp11           = regInfo{inputs: []regMask{fp}, outputs: []regMask{fp}}
   202  		fpgp           = regInfo{inputs: []regMask{fp}, outputs: []regMask{gp}}
   203  		fpgpfp         = regInfo{inputs: []regMask{fp, gp}, outputs: []regMask{fp}}
   204  		gpfp           = regInfo{inputs: []regMask{gp}, outputs: []regMask{fp}}
   205  		fp21           = regInfo{inputs: []regMask{fp, fp}, outputs: []regMask{fp}}
   206  		fp31           = regInfo{inputs: []regMask{fp, fp, fp}, outputs: []regMask{fp}}
   207  		fp2flags       = regInfo{inputs: []regMask{fp, fp}}
   208  		fp1flags       = regInfo{inputs: []regMask{fp}}
   209  		fp1predfp      = regInfo{inputs: []regMask{fp, pred}, outputs: []regMask{fp}}
   210  		fp2predpred    = regInfo{inputs: []regMask{fp, fp, pred}, outputs: []regMask{pred}}
   211  		fp2predfp      = regInfo{inputs: []regMask{fp, fp, pred}, outputs: []regMask{fp}}
   212  		fp3predfp      = regInfo{inputs: []regMask{fp, fp, fp, pred}, outputs: []regMask{fp}}
   213  		predload       = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{pred}}
   214  		predstore      = regInfo{inputs: []regMask{gpspsbg, pred}}
   215  		fpload         = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{fp}}
   216  		fpload2        = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{fp, fp}}
   217  		fp2load        = regInfo{inputs: []regMask{gpspsbg, gpg}, outputs: []regMask{fp}}
   218  		fpstore        = regInfo{inputs: []regMask{gpspsbg, fp}}
   219  		fpstoreidx     = regInfo{inputs: []regMask{gpspsbg, gpg, fp}}
   220  		gp2pred        = regInfo{inputs: []regMask{gpg, gpg}, outputs: []regMask{pred}}
   221  		fppredload     = regInfo{inputs: []regMask{gpspsbg, pred}, outputs: []regMask{fp}}
   222  		fppredstore    = regInfo{inputs: []regMask{gpspsbg, fp, pred}}
   223  		fpstore2       = regInfo{inputs: []regMask{gpspsbg, fp, fp}}
   224  		readflags      = regInfo{inputs: nil, outputs: []regMask{gp}}
   225  		prefreg        = regInfo{inputs: []regMask{gpspsbg}}
   226  	)
   227  	ops := []opData{
   228  		// binary ops
   229  		{name: "ADCSflags", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "ADCS", commutative: true},     // arg0+arg1+carry, set flags.
   230  		{name: "ADCzerocarry", argLength: 1, reg: gp0flags1, typ: "UInt64", asm: "ADC", earlyOk: true, zeroUpperBits: 56}, // ZR+ZR+carry
   231  		{name: "ADD", argLength: 2, reg: gp21, asm: "ADD", commutative: true, earlyOk: true},                              // arg0 + arg1
   232  		{name: "ADDconst", argLength: 1, reg: gp11sp, asm: "ADD", aux: "Int64", earlyOk: true},                            // arg0 + auxInt
   233  		{name: "ADDSconstflags", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "ADDS", aux: "Int64"},          // arg0+auxint, set flags.
   234  		{name: "ADDSflags", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "ADDS", commutative: true},          // arg0+arg1, set flags.
   235  		{name: "SUB", argLength: 2, reg: gp21, asm: "SUB", earlyOk: true},                                                 // arg0 - arg1
   236  		{name: "SUBconst", argLength: 1, reg: gp11, asm: "SUB", aux: "Int64", earlyOk: true},                              // arg0 - auxInt
   237  		{name: "SBCSflags", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "SBCS"},                        // arg0-(arg1+borrowing), set flags.
   238  		{name: "SUBSflags", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "SUBS"},                             // arg0 - arg1, set flags.
   239  		{name: "MUL", argLength: 2, reg: gp21, asm: "MUL", commutative: true, earlyOk: true},                              // arg0 * arg1
   240  		{name: "MULW", argLength: 2, reg: gp21, asm: "MULW", commutative: true, earlyOk: true, zeroUpperBits: 32},         // arg0 * arg1, 32-bit
   241  		{name: "MNEG", argLength: 2, reg: gp21, asm: "MNEG", commutative: true, earlyOk: true},                            // -arg0 * arg1
   242  		{name: "MNEGW", argLength: 2, reg: gp21, asm: "MNEGW", commutative: true, earlyOk: true, zeroUpperBits: 32},       // -arg0 * arg1, 32-bit
   243  		{name: "MULH", argLength: 2, reg: gp21, asm: "SMULH", commutative: true, earlyOk: true},                           // (arg0 * arg1) >> 64, signed
   244  		{name: "UMULH", argLength: 2, reg: gp21, asm: "UMULH", commutative: true, earlyOk: true},                          // (arg0 * arg1) >> 64, unsigned
   245  		{name: "MULL", argLength: 2, reg: gp21, asm: "SMULL", commutative: true, earlyOk: true},                           // arg0 * arg1, signed, 32-bit mult results in 64-bit
   246  		{name: "UMULL", argLength: 2, reg: gp21, asm: "UMULL", commutative: true, earlyOk: true},                          // arg0 * arg1, unsigned, 32-bit mult results in 64-bit
   247  		{name: "DIV", argLength: 2, reg: gp21, asm: "SDIV", earlyOk: true},                                                // arg0 / arg1, signed
   248  		{name: "UDIV", argLength: 2, reg: gp21, asm: "UDIV", earlyOk: true},                                               // arg0 / arg1, unsigned
   249  		{name: "DIVW", argLength: 2, reg: gp21, asm: "SDIVW", earlyOk: true, zeroUpperBits: 32},                           // arg0 / arg1, signed, 32 bit
   250  		{name: "UDIVW", argLength: 2, reg: gp21, asm: "UDIVW", earlyOk: true, zeroUpperBits: 32},                          // arg0 / arg1, unsigned, 32 bit
   251  		{name: "MOD", argLength: 2, reg: gp21, asm: "REM", earlyOk: true},                                                 // arg0 % arg1, signed
   252  		{name: "UMOD", argLength: 2, reg: gp21, asm: "UREM", earlyOk: true},                                               // arg0 % arg1, unsigned
   253  		{name: "MODW", argLength: 2, reg: gp21, asm: "REMW", earlyOk: true, zeroUpperBits: 32},                            // arg0 % arg1, signed, 32 bit
   254  		{name: "UMODW", argLength: 2, reg: gp21, asm: "UREMW", earlyOk: true, zeroUpperBits: 32},                          // arg0 % arg1, unsigned, 32 bit
   255  
   256  		{name: "FADDS", argLength: 2, reg: fp21, asm: "FADDS", commutative: true, earlyOk: true},   // arg0 + arg1
   257  		{name: "FADDD", argLength: 2, reg: fp21, asm: "FADDD", commutative: true, earlyOk: true},   // arg0 + arg1
   258  		{name: "FSUBS", argLength: 2, reg: fp21, asm: "FSUBS", earlyOk: true},                      // arg0 - arg1
   259  		{name: "FSUBD", argLength: 2, reg: fp21, asm: "FSUBD", earlyOk: true},                      // arg0 - arg1
   260  		{name: "FMULS", argLength: 2, reg: fp21, asm: "FMULS", commutative: true, earlyOk: true},   // arg0 * arg1
   261  		{name: "FMULD", argLength: 2, reg: fp21, asm: "FMULD", commutative: true, earlyOk: true},   // arg0 * arg1
   262  		{name: "FNMULS", argLength: 2, reg: fp21, asm: "FNMULS", commutative: true, earlyOk: true}, // -(arg0 * arg1)
   263  		{name: "FNMULD", argLength: 2, reg: fp21, asm: "FNMULD", commutative: true, earlyOk: true}, // -(arg0 * arg1)
   264  		{name: "FDIVS", argLength: 2, reg: fp21, asm: "FDIVS", earlyOk: true},                      // arg0 / arg1
   265  		{name: "FDIVD", argLength: 2, reg: fp21, asm: "FDIVD", earlyOk: true},                      // arg0 / arg1
   266  
   267  		{name: "AND", argLength: 2, reg: gp21, asm: "AND", commutative: true, earlyOk: true}, // arg0 & arg1
   268  		{name: "ANDconst", argLength: 1, reg: gp11, asm: "AND", aux: "Int64", earlyOk: true}, // arg0 & auxInt
   269  		{name: "OR", argLength: 2, reg: gp21, asm: "ORR", commutative: true, earlyOk: true},  // arg0 | arg1
   270  		{name: "ORconst", argLength: 1, reg: gp11, asm: "ORR", aux: "Int64", earlyOk: true},  // arg0 | auxInt
   271  		{name: "XOR", argLength: 2, reg: gp21, asm: "EOR", commutative: true, earlyOk: true}, // arg0 ^ arg1
   272  		{name: "XORconst", argLength: 1, reg: gp11, asm: "EOR", aux: "Int64", earlyOk: true}, // arg0 ^ auxInt
   273  		{name: "BIC", argLength: 2, reg: gp21, asm: "BIC", earlyOk: true},                    // arg0 &^ arg1
   274  		{name: "EON", argLength: 2, reg: gp21, asm: "EON", earlyOk: true},                    // arg0 ^ ^arg1
   275  		{name: "ORN", argLength: 2, reg: gp21, asm: "ORN", earlyOk: true},                    // arg0 | ^arg1
   276  
   277  		// unary ops
   278  		{name: "MVN", argLength: 1, reg: gp11, asm: "MVN", earlyOk: true},                              // ^arg0
   279  		{name: "NEG", argLength: 1, reg: gp11, asm: "NEG", earlyOk: true},                              // -arg0
   280  		{name: "NEGSflags", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "NEGS"},          // -arg0, set flags.
   281  		{name: "NGCzerocarry", argLength: 1, reg: gp0flags1, typ: "UInt64", asm: "NGC", earlyOk: true}, // -1 if borrowing, 0 otherwise.
   282  		{name: "FABSD", argLength: 1, reg: fp11, asm: "FABSD", earlyOk: true},                          // abs(arg0), float64
   283  		{name: "FABSS", argLength: 1, reg: fp11, asm: "FABSS", earlyOk: true},                          // abs(arg0), float32
   284  		{name: "FNEGS", argLength: 1, reg: fp11, asm: "FNEGS", earlyOk: true},                          // -arg0, float32
   285  		{name: "FNEGD", argLength: 1, reg: fp11, asm: "FNEGD", earlyOk: true},                          // -arg0, float64
   286  		{name: "FSQRTD", argLength: 1, reg: fp11, asm: "FSQRTD", earlyOk: true},                        // sqrt(arg0), float64
   287  		{name: "FSQRTS", argLength: 1, reg: fp11, asm: "FSQRTS", earlyOk: true},                        // sqrt(arg0), float32
   288  		{name: "FMIND", argLength: 2, reg: fp21, asm: "FMIND", earlyOk: true},                          // min(arg0, arg1)
   289  		{name: "FMINS", argLength: 2, reg: fp21, asm: "FMINS", earlyOk: true},                          // min(arg0, arg1)
   290  		{name: "FMAXD", argLength: 2, reg: fp21, asm: "FMAXD", earlyOk: true},                          // max(arg0, arg1)
   291  		{name: "FMAXS", argLength: 2, reg: fp21, asm: "FMAXS", earlyOk: true},                          // max(arg0, arg1)
   292  		{name: "REV", argLength: 1, reg: gp11, asm: "REV", earlyOk: true},                              // byte reverse, 64-bit
   293  		{name: "REVW", argLength: 1, reg: gp11, asm: "REVW", earlyOk: true, zeroUpperBits: 32},         // byte reverse, 32-bit
   294  		{name: "REV16", argLength: 1, reg: gp11, asm: "REV16", earlyOk: true},                          // byte reverse in each 16-bit halfword, 64-bit
   295  		{name: "REV16W", argLength: 1, reg: gp11, asm: "REV16W", earlyOk: true, zeroUpperBits: 32},     // byte reverse in each 16-bit halfword, 32-bit
   296  		{name: "RBIT", argLength: 1, reg: gp11, asm: "RBIT", earlyOk: true},                            // bit reverse, 64-bit
   297  		{name: "RBITW", argLength: 1, reg: gp11, asm: "RBITW", earlyOk: true, zeroUpperBits: 32},       // bit reverse, 32-bit
   298  		{name: "CLZ", argLength: 1, reg: gp11, asm: "CLZ", earlyOk: true, zeroUpperBits: 56},           // count leading zero, 64-bit
   299  		{name: "CLZW", argLength: 1, reg: gp11, asm: "CLZW", earlyOk: true, zeroUpperBits: 56},         // count leading zero, 32-bit
   300  		{name: "VCNT", argLength: 1, reg: fp11, asm: "VCNT", earlyOk: true},                            // count set bits for each 8-bit unit and store the result in each 8-bit unit
   301  		{name: "VUADDLV", argLength: 1, reg: fp11, asm: "VUADDLV", earlyOk: true},                      // unsigned sum of eight bytes in a 64-bit value, zero extended to 64-bit.
   302  		{name: "LoweredRound32F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   303  		{name: "LoweredRound64F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   304  
   305  		// 3-operand, the addend comes first
   306  		{name: "FMADDS", argLength: 3, reg: fp31, asm: "FMADDS", earlyOk: true},                  // +arg0 + (arg1 * arg2)
   307  		{name: "FMADDD", argLength: 3, reg: fp31, asm: "FMADDD", earlyOk: true},                  // +arg0 + (arg1 * arg2)
   308  		{name: "FNMADDS", argLength: 3, reg: fp31, asm: "FNMADDS", earlyOk: true},                // -arg0 - (arg1 * arg2)
   309  		{name: "FNMADDD", argLength: 3, reg: fp31, asm: "FNMADDD", earlyOk: true},                // -arg0 - (arg1 * arg2)
   310  		{name: "FMSUBS", argLength: 3, reg: fp31, asm: "FMSUBS", earlyOk: true},                  // +arg0 - (arg1 * arg2)
   311  		{name: "FMSUBD", argLength: 3, reg: fp31, asm: "FMSUBD", earlyOk: true},                  // +arg0 - (arg1 * arg2)
   312  		{name: "FNMSUBS", argLength: 3, reg: fp31, asm: "FNMSUBS", earlyOk: true},                // -arg0 + (arg1 * arg2)
   313  		{name: "FNMSUBD", argLength: 3, reg: fp31, asm: "FNMSUBD", earlyOk: true},                // -arg0 + (arg1 * arg2)
   314  		{name: "MADD", argLength: 3, reg: gp31, asm: "MADD", earlyOk: true},                      // +arg0 + (arg1 * arg2)
   315  		{name: "MADDW", argLength: 3, reg: gp31, asm: "MADDW", earlyOk: true, zeroUpperBits: 32}, // +arg0 + (arg1 * arg2), 32-bit
   316  		{name: "MSUB", argLength: 3, reg: gp31, asm: "MSUB", earlyOk: true},                      // +arg0 - (arg1 * arg2)
   317  		{name: "MSUBW", argLength: 3, reg: gp31, asm: "MSUBW", earlyOk: true, zeroUpperBits: 32}, // +arg0 - (arg1 * arg2), 32-bit
   318  
   319  		// shifts
   320  		{name: "SLL", argLength: 2, reg: gp21, asm: "LSL", earlyOk: true},                                           // arg0 << arg1, shift amount is mod 64
   321  		{name: "SLLconst", argLength: 1, reg: gp11, asm: "LSL", aux: "Int64", earlyOk: true},                        // arg0 << auxInt, auxInt should be in the range 0 to 63.
   322  		{name: "SRL", argLength: 2, reg: gp21, asm: "LSR", earlyOk: true},                                           // arg0 >> arg1, unsigned, shift amount is mod 64
   323  		{name: "SRLconst", argLength: 1, reg: gp11, asm: "LSR", aux: "Int64", earlyOk: true},                        // arg0 >> auxInt, unsigned, auxInt should be in the range 0 to 63.
   324  		{name: "SRA", argLength: 2, reg: gp21, asm: "ASR", earlyOk: true},                                           // arg0 >> arg1, signed, shift amount is mod 64
   325  		{name: "SRAconst", argLength: 1, reg: gp11, asm: "ASR", aux: "Int64", earlyOk: true},                        // arg0 >> auxInt, signed, auxInt should be in the range 0 to 63.
   326  		{name: "ROR", argLength: 2, reg: gp21, asm: "ROR", earlyOk: true},                                           // arg0 right rotate by (arg1 mod 64) bits
   327  		{name: "RORW", argLength: 2, reg: gp21, asm: "RORW", earlyOk: true, zeroUpperBits: 32},                      // arg0 right rotate by (arg1 mod 32) bits
   328  		{name: "RORconst", argLength: 1, reg: gp11, asm: "ROR", aux: "Int64", earlyOk: true},                        // arg0 right rotate by auxInt bits, auxInt should be in the range 0 to 63.
   329  		{name: "RORWconst", argLength: 1, reg: gp11, asm: "RORW", aux: "Int64", earlyOk: true, zeroUpperBits: 32},   // uint32(arg0) right rotate by auxInt bits, auxInt should be in the range 0 to 31.
   330  		{name: "EXTRconst", argLength: 2, reg: gp21, asm: "EXTR", aux: "Int64", earlyOk: true},                      // extract 64 bits from arg0:arg1 starting at lsb auxInt, auxInt should be in the range 0 to 63.
   331  		{name: "EXTRWconst", argLength: 2, reg: gp21, asm: "EXTRW", aux: "Int64", earlyOk: true, zeroUpperBits: 32}, // extract 32 bits from arg0[31:0]:arg1[31:0] starting at lsb auxInt and zero top 32 bits, auxInt should be in the range 0 to 31.
   332  
   333  		// comparisons
   334  		{name: "CMP", argLength: 2, reg: gp2flags, asm: "CMP", typ: "Flags"},                      // arg0 compare to arg1
   335  		{name: "CMPconst", argLength: 1, reg: gp1flags, asm: "CMP", aux: "Int64", typ: "Flags"},   // arg0 compare to auxInt
   336  		{name: "CMPW", argLength: 2, reg: gp2flags, asm: "CMPW", typ: "Flags"},                    // arg0 compare to arg1, 32 bit
   337  		{name: "CMPWconst", argLength: 1, reg: gp1flags, asm: "CMPW", aux: "Int32", typ: "Flags"}, // arg0 compare to auxInt, 32 bit
   338  		{name: "CMN", argLength: 2, reg: gp2flags, asm: "CMN", typ: "Flags", commutative: true},   // arg0 compare to -arg1, provided arg1 is not 1<<63
   339  		{name: "CMNconst", argLength: 1, reg: gp1flags, asm: "CMN", aux: "Int64", typ: "Flags"},   // arg0 compare to -auxInt
   340  		{name: "CMNW", argLength: 2, reg: gp2flags, asm: "CMNW", typ: "Flags", commutative: true}, // arg0 compare to -arg1, 32 bit, provided arg1 is not 1<<31
   341  		{name: "CMNWconst", argLength: 1, reg: gp1flags, asm: "CMNW", aux: "Int32", typ: "Flags"}, // arg0 compare to -auxInt, 32 bit
   342  		{name: "TST", argLength: 2, reg: gp2flags, asm: "TST", typ: "Flags", commutative: true},   // arg0 & arg1 compare to 0
   343  		{name: "TSTconst", argLength: 1, reg: gp1flags, asm: "TST", aux: "Int64", typ: "Flags"},   // arg0 & auxInt compare to 0
   344  		{name: "TSTW", argLength: 2, reg: gp2flags, asm: "TSTW", typ: "Flags", commutative: true}, // arg0 & arg1 compare to 0, 32 bit
   345  		{name: "TSTWconst", argLength: 1, reg: gp1flags, asm: "TSTW", aux: "Int32", typ: "Flags"}, // arg0 & auxInt compare to 0, 32 bit
   346  		{name: "FCMPS", argLength: 2, reg: fp2flags, asm: "FCMPS", typ: "Flags"},                  // arg0 compare to arg1, float32
   347  		{name: "FCMPD", argLength: 2, reg: fp2flags, asm: "FCMPD", typ: "Flags"},                  // arg0 compare to arg1, float64
   348  		{name: "FCMPS0", argLength: 1, reg: fp1flags, asm: "FCMPS", typ: "Flags"},                 // arg0 compare to 0, float32
   349  		{name: "FCMPD0", argLength: 1, reg: fp1flags, asm: "FCMPD", typ: "Flags"},                 // arg0 compare to 0, float64
   350  
   351  		// shifted ops
   352  		{name: "MVNshiftLL", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0<<auxInt), auxInt should be in the range 0 to 63.
   353  		{name: "MVNshiftRL", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   354  		{name: "MVNshiftRA", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   355  		{name: "MVNshiftRO", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   356  		{name: "NEGshiftLL", argLength: 1, reg: gp11, asm: "NEG", aux: "Int64", earlyOk: true},    // -(arg0<<auxInt), auxInt should be in the range 0 to 63.
   357  		{name: "NEGshiftRL", argLength: 1, reg: gp11, asm: "NEG", aux: "Int64", earlyOk: true},    // -(arg0>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   358  		{name: "NEGshiftRA", argLength: 1, reg: gp11, asm: "NEG", aux: "Int64", earlyOk: true},    // -(arg0>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   359  		{name: "ADDshiftLL", argLength: 2, reg: gp21, asm: "ADD", aux: "Int64", earlyOk: true},    // arg0 + arg1<<auxInt, auxInt should be in the range 0 to 63.
   360  		{name: "ADDshiftRL", argLength: 2, reg: gp21, asm: "ADD", aux: "Int64", earlyOk: true},    // arg0 + arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   361  		{name: "ADDshiftRA", argLength: 2, reg: gp21, asm: "ADD", aux: "Int64", earlyOk: true},    // arg0 + arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   362  		{name: "SUBshiftLL", argLength: 2, reg: gp21, asm: "SUB", aux: "Int64", earlyOk: true},    // arg0 - arg1<<auxInt, auxInt should be in the range 0 to 63.
   363  		{name: "SUBshiftRL", argLength: 2, reg: gp21, asm: "SUB", aux: "Int64", earlyOk: true},    // arg0 - arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   364  		{name: "SUBshiftRA", argLength: 2, reg: gp21, asm: "SUB", aux: "Int64", earlyOk: true},    // arg0 - arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   365  		{name: "ANDshiftLL", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1<<auxInt), auxInt should be in the range 0 to 63.
   366  		{name: "ANDshiftRL", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   367  		{name: "ANDshiftRA", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   368  		{name: "ANDshiftRO", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   369  		{name: "ORshiftLL", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1<<auxInt, auxInt should be in the range 0 to 63.
   370  		{name: "ORshiftRL", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   371  		{name: "ORshiftRA", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   372  		{name: "ORshiftRO", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1 ROR auxInt, signed shift, auxInt should be in the range 0 to 63.
   373  		{name: "XORshiftLL", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1<<auxInt, auxInt should be in the range 0 to 63.
   374  		{name: "XORshiftRL", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   375  		{name: "XORshiftRA", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   376  		{name: "XORshiftRO", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1 ROR auxInt, signed shift, auxInt should be in the range 0 to 63.
   377  		{name: "BICshiftLL", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1<<auxInt), auxInt should be in the range 0 to 63.
   378  		{name: "BICshiftRL", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   379  		{name: "BICshiftRA", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   380  		{name: "BICshiftRO", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   381  		{name: "EONshiftLL", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1<<auxInt), auxInt should be in the range 0 to 63.
   382  		{name: "EONshiftRL", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   383  		{name: "EONshiftRA", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   384  		{name: "EONshiftRO", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   385  		{name: "ORNshiftLL", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1<<auxInt), auxInt should be in the range 0 to 63.
   386  		{name: "ORNshiftRL", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   387  		{name: "ORNshiftRA", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   388  		{name: "ORNshiftRO", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   389  		{name: "CMPshiftLL", argLength: 2, reg: gp2flags, asm: "CMP", aux: "Int64", typ: "Flags"}, // arg0 compare to arg1<<auxInt, auxInt should be in the range 0 to 63.
   390  		{name: "CMPshiftRL", argLength: 2, reg: gp2flags, asm: "CMP", aux: "Int64", typ: "Flags"}, // arg0 compare to arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   391  		{name: "CMPshiftRA", argLength: 2, reg: gp2flags, asm: "CMP", aux: "Int64", typ: "Flags"}, // arg0 compare to arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   392  		{name: "CMNshiftLL", argLength: 2, reg: gp2flags, asm: "CMN", aux: "Int64", typ: "Flags"}, // (arg0 + arg1<<auxInt) compare to 0, auxInt should be in the range 0 to 63.
   393  		{name: "CMNshiftRL", argLength: 2, reg: gp2flags, asm: "CMN", aux: "Int64", typ: "Flags"}, // (arg0 + arg1>>auxInt) compare to 0, unsigned shift, auxInt should be in the range 0 to 63.
   394  		{name: "CMNshiftRA", argLength: 2, reg: gp2flags, asm: "CMN", aux: "Int64", typ: "Flags"}, // (arg0 + arg1>>auxInt) compare to 0, signed shift, auxInt should be in the range 0 to 63.
   395  		{name: "TSTshiftLL", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1<<auxInt) compare to 0, auxInt should be in the range 0 to 63.
   396  		{name: "TSTshiftRL", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1>>auxInt) compare to 0, unsigned shift, auxInt should be in the range 0 to 63.
   397  		{name: "TSTshiftRA", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1>>auxInt) compare to 0, signed shift, auxInt should be in the range 0 to 63.
   398  		{name: "TSTshiftRO", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1 ROR auxInt) compare to 0, signed shift, auxInt should be in the range 0 to 63.
   399  
   400  		// bitfield ops
   401  		// for all bitfield ops lsb is auxInt>>8, width is auxInt&0xff
   402  		// insert low width bits of arg1 into the result starting at bit lsb, copy other bits from arg0
   403  		{name: "BFI", argLength: 2, reg: gp21nog, asm: "BFI", aux: "ARM64BitField", resultInArg0: true, earlyOk: true},
   404  		// extract width bits of arg1 starting at bit lsb and insert at low end of result, copy other bits from arg0
   405  		{name: "BFXIL", argLength: 2, reg: gp21nog, asm: "BFXIL", aux: "ARM64BitField", resultInArg0: true, earlyOk: true},
   406  		// insert low width bits of arg0 into the result starting at bit lsb, bits to the left of the inserted bit field are set to the high/sign bit of the inserted bit field, bits to the right are zeroed
   407  		{name: "SBFIZ", argLength: 1, reg: gp11, asm: "SBFIZ", aux: "ARM64BitField", earlyOk: true},
   408  		// extract width bits of arg0 starting at bit lsb and insert at low end of result, remaining high bits are set to the high/sign bit of the extracted bitfield
   409  		{name: "SBFX", argLength: 1, reg: gp11, asm: "SBFX", aux: "ARM64BitField", earlyOk: true},
   410  		// insert low width bits of arg0 into the result starting at bit lsb, bits to the left and right of the inserted bit field are zeroed
   411  		{name: "UBFIZ", argLength: 1, reg: gp11, asm: "UBFIZ", aux: "ARM64BitField", earlyOk: true},
   412  		// extract width bits of arg0 starting at bit lsb and insert at low end of result, remaining high bits are zeroed
   413  		{name: "UBFX", argLength: 1, reg: gp11, asm: "UBFX", aux: "ARM64BitField", earlyOk: true},
   414  
   415  		// moves
   416  		{name: "MOVDconst", argLength: 0, reg: gp01, aux: "Int64", asm: "MOVD", typ: "UInt64", rematerializeable: true, earlyOk: true},      // 64 bits from auxint
   417  		{name: "FMOVSconst", argLength: 0, reg: fp01, aux: "Float64", asm: "FMOVS", typ: "Float32", rematerializeable: true, earlyOk: true}, // auxint as 64-bit float, convert to 32-bit float
   418  		{name: "FMOVDconst", argLength: 0, reg: fp01, aux: "Float64", asm: "FMOVD", typ: "Float64", rematerializeable: true, earlyOk: true}, // auxint as 64-bit float
   419  
   420  		{name: "MOVDaddr", argLength: 1, reg: regInfo{inputs: []regMask{buildReg("SP").union(buildReg("SB"))}, outputs: []regMask{gp}}, aux: "SymOff", asm: "MOVD", rematerializeable: true, symEffect: "Addr", earlyOk: true}, // arg0 + auxInt + aux.(*gc.Sym), arg0=SP/SB
   421  
   422  		{name: "MOVBload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVB", typ: "Int8", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                        // load from arg0 + auxInt + aux.  arg1=mem.
   423  		{name: "MOVBUload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVBU", typ: "UInt8", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 56},  // load from arg0 + auxInt + aux.  arg1=mem.
   424  		{name: "MOVHload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVH", typ: "Int16", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                       // load from arg0 + auxInt + aux.  arg1=mem.
   425  		{name: "MOVHUload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVHU", typ: "UInt16", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 48}, // load from arg0 + auxInt + aux.  arg1=mem.
   426  		{name: "MOVWload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVW", typ: "Int32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                       // load from arg0 + auxInt + aux.  arg1=mem.
   427  		{name: "MOVWUload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVWU", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // load from arg0 + auxInt + aux.  arg1=mem.
   428  		{name: "MOVDload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVD", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                      // load from arg0 + auxInt + aux.  arg1=mem.
   429  		{name: "FMOVSload", argLength: 2, reg: fpload, aux: "SymOff", asm: "FMOVS", typ: "Float32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                   // load from arg0 + auxInt + aux.  arg1=mem.
   430  		{name: "FMOVDload", argLength: 2, reg: fpload, aux: "SymOff", asm: "FMOVD", typ: "Float64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                   // load from arg0 + auxInt + aux.  arg1=mem.
   431  		{name: "FMOVQload", argLength: 2, reg: fpload, aux: "SymOff", asm: "FMOVQ", typ: "Vec128", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // load from arg0 + auxInt + aux.  arg1=mem.
   432  
   433  		// LDP instructions load the contents of two adjacent locations in memory into registers.
   434  		// Address to start loading is addr = arg0 + auxInt + aux.
   435  		// x := *(*T)(addr)
   436  		// y := *(*T)(addr+sizeof(T))
   437  		// arg1=mem
   438  		// Returns the tuple <x,y>.
   439  		{name: "LDP", argLength: 2, reg: gpload2, aux: "SymOff", asm: "LDP", typ: "(UInt64,UInt64)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                      // T=int64 (gp reg destination)
   440  		{name: "LDPW", argLength: 2, reg: gpload2, aux: "SymOff", asm: "LDPW", typ: "(UInt32,UInt32)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // T=int32 (gp reg destination) unsigned extension
   441  		{name: "LDPSW", argLength: 2, reg: gpload2, aux: "SymOff", asm: "LDPSW", typ: "(Int32,Int32)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // T=int32 (gp reg destination) signed extension
   442  		{name: "FLDPD", argLength: 2, reg: fpload2, aux: "SymOff", asm: "FLDPD", typ: "(Float64,Float64)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                // T=float64 (fp reg destination)
   443  		{name: "FLDPS", argLength: 2, reg: fpload2, aux: "SymOff", asm: "FLDPS", typ: "(Float32,Float32)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                // T=float32 (fp reg destination)
   444  		{name: "FLDPQ", argLength: 2, reg: fpload2, aux: "SymOff", asm: "FLDPQ", typ: "(Vec128,Vec128)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                  // T=vec128 (fp reg destination)
   445  
   446  		// register indexed load
   447  		{name: "MOVDloadidx", argLength: 3, reg: gp2load, asm: "MOVD", typ: "UInt64", addrSinkArg0: true, addrSinkArg1: true},                      // load 64-bit dword from arg0 + arg1, arg2 = mem.
   448  		{name: "MOVWloadidx", argLength: 3, reg: gp2load, asm: "MOVW", typ: "Int32", addrSinkArg0: true, addrSinkArg1: true},                       // load 32-bit word from arg0 + arg1, sign-extended to 64-bit, arg2=mem.
   449  		{name: "MOVWUloadidx", argLength: 3, reg: gp2load, asm: "MOVWU", typ: "UInt32", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 32}, // load 32-bit word from arg0 + arg1, zero-extended to 64-bit, arg2=mem.
   450  		{name: "MOVHloadidx", argLength: 3, reg: gp2load, asm: "MOVH", typ: "Int16", addrSinkArg0: true, addrSinkArg1: true},                       // load 16-bit word from arg0 + arg1, sign-extended to 64-bit, arg2=mem.
   451  		{name: "MOVHUloadidx", argLength: 3, reg: gp2load, asm: "MOVHU", typ: "UInt16", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 48}, // load 16-bit word from arg0 + arg1, zero-extended to 64-bit, arg2=mem.
   452  		{name: "MOVBloadidx", argLength: 3, reg: gp2load, asm: "MOVB", typ: "Int8", addrSinkArg0: true, addrSinkArg1: true},                        // load 8-bit word from arg0 + arg1, sign-extended to 64-bit, arg2=mem.
   453  		{name: "MOVBUloadidx", argLength: 3, reg: gp2load, asm: "MOVBU", typ: "UInt8", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 56},  // load 8-bit word from arg0 + arg1, zero-extended to 64-bit, arg2=mem.
   454  		{name: "FMOVSloadidx", argLength: 3, reg: fp2load, asm: "FMOVS", typ: "Float32", addrSinkArg0: true, addrSinkArg1: true},                   // load 32-bit float from arg0 + arg1, arg2=mem.
   455  		{name: "FMOVDloadidx", argLength: 3, reg: fp2load, asm: "FMOVD", typ: "Float64", addrSinkArg0: true, addrSinkArg1: true},                   // load 64-bit float from arg0 + arg1, arg2=mem.
   456  
   457  		// shifted register indexed load
   458  		{name: "MOVHloadidx2", argLength: 3, reg: gp2load, asm: "MOVH", typ: "Int16", addrSinkArg0: true},                       // load 16-bit half-word from arg0 + arg1*2, sign-extended to 64-bit, arg2=mem.
   459  		{name: "MOVHUloadidx2", argLength: 3, reg: gp2load, asm: "MOVHU", typ: "UInt16", addrSinkArg0: true, zeroUpperBits: 48}, // load 16-bit half-word from arg0 + arg1*2, zero-extended to 64-bit, arg2=mem.
   460  		{name: "MOVWloadidx4", argLength: 3, reg: gp2load, asm: "MOVW", typ: "Int32", addrSinkArg0: true},                       // load 32-bit word from arg0 + arg1*4, sign-extended to 64-bit, arg2=mem.
   461  		{name: "MOVWUloadidx4", argLength: 3, reg: gp2load, asm: "MOVWU", typ: "UInt32", addrSinkArg0: true, zeroUpperBits: 32}, // load 32-bit word from arg0 + arg1*4, zero-extended to 64-bit, arg2=mem.
   462  		{name: "MOVDloadidx8", argLength: 3, reg: gp2load, asm: "MOVD", typ: "UInt64", addrSinkArg0: true},                      // load 64-bit double-word from arg0 + arg1*8, arg2 = mem.
   463  		{name: "FMOVSloadidx4", argLength: 3, reg: fp2load, asm: "FMOVS", typ: "Float32", addrSinkArg0: true},                   // load 32-bit float from arg0 + arg1*4, arg2 = mem.
   464  		{name: "FMOVDloadidx8", argLength: 3, reg: fp2load, asm: "FMOVD", typ: "Float64", addrSinkArg0: true},                   // load 64-bit float from arg0 + arg1*8, arg2 = mem.
   465  
   466  		{name: "MOVBstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVB", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 1 byte of arg1 to arg0 + auxInt + aux.  arg2=mem.
   467  		{name: "MOVHstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVH", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 2 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   468  		{name: "MOVWstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVW", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 4 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   469  		{name: "MOVDstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVD", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 8 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   470  		{name: "FMOVSstore", argLength: 3, reg: fpstore, aux: "SymOff", asm: "FMOVS", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 4 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   471  		{name: "FMOVDstore", argLength: 3, reg: fpstore, aux: "SymOff", asm: "FMOVD", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 8 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   472  		{name: "FMOVQstore", argLength: 3, reg: fpstore, aux: "SymOff", asm: "FMOVQ", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 16 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   473  
   474  		// STP instructions store the contents of two registers to adjacent locations in memory.
   475  		// Address to start storing is addr = arg0 + auxInt + aux.
   476  		// *(*T)(addr) = arg1
   477  		// *(*T)(addr+sizeof(T)) = arg2
   478  		// arg3=mem. Returns mem.
   479  		{name: "STP", argLength: 4, reg: gpstore2, aux: "SymOff", asm: "STP", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},     // T=int64 (gp reg source)
   480  		{name: "STPW", argLength: 4, reg: gpstore2, aux: "SymOff", asm: "STPW", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // T=int32 (gp reg source)
   481  		{name: "FSTPD", argLength: 4, reg: fpstore2, aux: "SymOff", asm: "FSTPD", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // T=float64 (fp reg source)
   482  		{name: "FSTPS", argLength: 4, reg: fpstore2, aux: "SymOff", asm: "FSTPS", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // T=float32 (fp reg source)
   483  		{name: "FSTPQ", argLength: 4, reg: fpstore2, aux: "SymOff", asm: "FSTPQ", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // T=vec128 (fp reg source)
   484  
   485  		// register indexed store
   486  		{name: "MOVBstoreidx", argLength: 4, reg: gpstore2, asm: "MOVB", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 1 byte of arg2 to arg0 + arg1, arg3 = mem.
   487  		{name: "MOVHstoreidx", argLength: 4, reg: gpstore2, asm: "MOVH", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 2 bytes of arg2 to arg0 + arg1, arg3 = mem.
   488  		{name: "MOVWstoreidx", argLength: 4, reg: gpstore2, asm: "MOVW", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 4 bytes of arg2 to arg0 + arg1, arg3 = mem.
   489  		{name: "MOVDstoreidx", argLength: 4, reg: gpstore2, asm: "MOVD", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 8 bytes of arg2 to arg0 + arg1, arg3 = mem.
   490  		{name: "FMOVSstoreidx", argLength: 4, reg: fpstoreidx, asm: "FMOVS", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true}, // store 32-bit float of arg2 to arg0 + arg1, arg3=mem.
   491  		{name: "FMOVDstoreidx", argLength: 4, reg: fpstoreidx, asm: "FMOVD", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true}, // store 64-bit float of arg2 to arg0 + arg1, arg3=mem.
   492  
   493  		// shifted register indexed store
   494  		{name: "MOVHstoreidx2", argLength: 4, reg: gpstore2, asm: "MOVH", typ: "Mem", addrSinkArg0: true},     // store 2 bytes of arg2 to arg0 + arg1*2, arg3 = mem.
   495  		{name: "MOVWstoreidx4", argLength: 4, reg: gpstore2, asm: "MOVW", typ: "Mem", addrSinkArg0: true},     // store 4 bytes of arg2 to arg0 + arg1*4, arg3 = mem.
   496  		{name: "MOVDstoreidx8", argLength: 4, reg: gpstore2, asm: "MOVD", typ: "Mem", addrSinkArg0: true},     // store 8 bytes of arg2 to arg0 + arg1*8, arg3 = mem.
   497  		{name: "FMOVSstoreidx4", argLength: 4, reg: fpstoreidx, asm: "FMOVS", typ: "Mem", addrSinkArg0: true}, // store 32-bit float of arg2 to arg0 + arg1*4, arg3=mem.
   498  		{name: "FMOVDstoreidx8", argLength: 4, reg: fpstoreidx, asm: "FMOVD", typ: "Mem", addrSinkArg0: true}, // store 64-bit float of arg2 to arg0 + arg1*8, arg3=mem.
   499  
   500  		{name: "FMOVDgpfp", argLength: 1, reg: gpfp, asm: "FMOVD", earlyOk: true},                    // move int64 to float64 (no conversion)
   501  		{name: "FMOVDfpgp", argLength: 1, reg: fpgp, asm: "FMOVD", earlyOk: true},                    // move float64 to int64 (no conversion)
   502  		{name: "FMOVSgpfp", argLength: 1, reg: gpfp, asm: "FMOVS", earlyOk: true},                    // move 32bits from int to float reg (no conversion)
   503  		{name: "FMOVSfpgp", argLength: 1, reg: fpgp, asm: "FMOVS", earlyOk: true, zeroUpperBits: 32}, // move 32bits from float to int reg, zero extend (no conversion)
   504  
   505  		// conversions
   506  		{name: "MOVBreg", argLength: 1, reg: gp11, asm: "MOVB", earlyOk: true},                      // move from arg0, sign-extended from byte
   507  		{name: "MOVBUreg", argLength: 1, reg: gp11, asm: "MOVBU", earlyOk: true, zeroUpperBits: 56}, // move from arg0, unsign-extended from byte
   508  		{name: "MOVHreg", argLength: 1, reg: gp11, asm: "MOVH", earlyOk: true},                      // move from arg0, sign-extended from half
   509  		{name: "MOVHUreg", argLength: 1, reg: gp11, asm: "MOVHU", earlyOk: true, zeroUpperBits: 48}, // move from arg0, unsign-extended from half
   510  		{name: "MOVWreg", argLength: 1, reg: gp11, asm: "MOVW", earlyOk: true},                      // move from arg0, sign-extended from word
   511  		{name: "MOVWUreg", argLength: 1, reg: gp11, asm: "MOVWU", earlyOk: true, zeroUpperBits: 32}, // move from arg0, unsign-extended from word
   512  		{name: "MOVDreg", argLength: 1, reg: gp11, asm: "MOVD", earlyOk: true},                      // move from arg0
   513  
   514  		{name: "MOVDnop", argLength: 1, reg: regInfo{inputs: []regMask{gp}, outputs: []regMask{gp}}, resultInArg0: true, earlyOk: true}, // nop, return arg0 in same register
   515  
   516  		{name: "SCVTFWS", argLength: 1, reg: gpfp, asm: "SCVTFWS", earlyOk: true},                      // int32 -> float32
   517  		{name: "SCVTFWD", argLength: 1, reg: gpfp, asm: "SCVTFWD", earlyOk: true},                      // int32 -> float64
   518  		{name: "UCVTFWS", argLength: 1, reg: gpfp, asm: "UCVTFWS", earlyOk: true},                      // uint32 -> float32
   519  		{name: "UCVTFWD", argLength: 1, reg: gpfp, asm: "UCVTFWD", earlyOk: true},                      // uint32 -> float64
   520  		{name: "SCVTFS", argLength: 1, reg: gpfp, asm: "SCVTFS", earlyOk: true},                        // int64 -> float32
   521  		{name: "SCVTFD", argLength: 1, reg: gpfp, asm: "SCVTFD", earlyOk: true},                        // int64 -> float64
   522  		{name: "UCVTFS", argLength: 1, reg: gpfp, asm: "UCVTFS", earlyOk: true},                        // uint64 -> float32
   523  		{name: "UCVTFD", argLength: 1, reg: gpfp, asm: "UCVTFD", earlyOk: true},                        // uint64 -> float64
   524  		{name: "FCVTZSSW", argLength: 1, reg: fpgp, asm: "FCVTZSSW", earlyOk: true, zeroUpperBits: 32}, // float32 -> int32
   525  		{name: "FCVTZSDW", argLength: 1, reg: fpgp, asm: "FCVTZSDW", earlyOk: true, zeroUpperBits: 32}, // float64 -> int32
   526  		{name: "FCVTZUSW", argLength: 1, reg: fpgp, asm: "FCVTZUSW", earlyOk: true, zeroUpperBits: 32}, // float32 -> uint32
   527  		{name: "FCVTZUDW", argLength: 1, reg: fpgp, asm: "FCVTZUDW", earlyOk: true, zeroUpperBits: 32}, // float64 -> uint32
   528  		{name: "FCVTZSS", argLength: 1, reg: fpgp, asm: "FCVTZSS", earlyOk: true},                      // float32 -> int64
   529  		{name: "FCVTZSD", argLength: 1, reg: fpgp, asm: "FCVTZSD", earlyOk: true},                      // float64 -> int64
   530  		{name: "FCVTZUS", argLength: 1, reg: fpgp, asm: "FCVTZUS", earlyOk: true},                      // float32 -> uint64
   531  		{name: "FCVTZUD", argLength: 1, reg: fpgp, asm: "FCVTZUD", earlyOk: true},                      // float64 -> uint64
   532  		{name: "FCVTSD", argLength: 1, reg: fp11, asm: "FCVTSD", earlyOk: true},                        // float32 -> float64
   533  		{name: "FCVTDS", argLength: 1, reg: fp11, asm: "FCVTDS", earlyOk: true},                        // float64 -> float32
   534  
   535  		// 64-bit floating-point round to integers in 64-bit FP format
   536  		{name: "FRINTAD", argLength: 1, reg: fp11, asm: "FRINTAD", earlyOk: true}, // Round (ties Away from zero; 0.5 -> 1, -0.5 -> -1)
   537  		{name: "FRINTMD", argLength: 1, reg: fp11, asm: "FRINTMD", earlyOk: true}, // Floor (towards Minus; 0.5 -> 0, -0.5 -> -1)
   538  		{name: "FRINTND", argLength: 1, reg: fp11, asm: "FRINTND", earlyOk: true}, // Round (ties to even; ; 0.5 -> 0, 1.5 -> 2)
   539  		{name: "FRINTPD", argLength: 1, reg: fp11, asm: "FRINTPD", earlyOk: true}, // Ceil (towards Positive; 0.5 -> 1, -0.5 -> 0)
   540  		{name: "FRINTZD", argLength: 1, reg: fp11, asm: "FRINTZD", earlyOk: true}, // Trunc (towards Zero; 0.5 -> 0, -0.5 -> 0))
   541  		// 32-bit floating-point round to integers in 32-bit FP format
   542  		{name: "FRINTAS", argLength: 1, reg: fp11, asm: "FRINTAS", earlyOk: true}, // Round (ties Away from zero; 0.5 -> 1, -0.5 -> -1)
   543  		{name: "FRINTMS", argLength: 1, reg: fp11, asm: "FRINTMS", earlyOk: true}, // Floor (towards Minus; 0.5 -> 0, -0.5 -> -1)
   544  		{name: "FRINTNS", argLength: 1, reg: fp11, asm: "FRINTNS", earlyOk: true}, // Round (ties to even; ; 0.5 -> 0, 1.5 -> 2)
   545  		{name: "FRINTPS", argLength: 1, reg: fp11, asm: "FRINTPS", earlyOk: true}, // Ceil (towards Positive; 0.5 -> 1, -0.5 -> 0)
   546  		{name: "FRINTZS", argLength: 1, reg: fp11, asm: "FRINTZS", earlyOk: true}, // Trunc (towards Zero; 0.5 -> 0, -0.5 -> 0))
   547  
   548  		// conditional instructions; auxint is
   549  		// one of the arm64 comparison pseudo-ops (LessThan, LessThanU, etc.)
   550  		{name: "CSEL", argLength: 3, reg: gp2flags1, asm: "CSEL", aux: "CCop", earlyOk: true},   // auxint(flags) ? arg0 : arg1
   551  		{name: "CSEL0", argLength: 2, reg: gp1flags1, asm: "CSEL", aux: "CCop", earlyOk: true},  // auxint(flags) ? arg0 : 0
   552  		{name: "CSINC", argLength: 3, reg: gp2flags1, asm: "CSINC", aux: "CCop", earlyOk: true}, // auxint(flags) ? arg0 : arg1 + 1
   553  		{name: "CSINV", argLength: 3, reg: gp2flags1, asm: "CSINV", aux: "CCop", earlyOk: true}, // auxint(flags) ? arg0 : ^arg1
   554  		{name: "CSNEG", argLength: 3, reg: gp2flags1, asm: "CSNEG", aux: "CCop", earlyOk: true}, // auxint(flags) ? arg0 : -arg1
   555  		{name: "FCSELD", argLength: 3, reg: fp21, asm: "FCSELD", aux: "CCop", earlyOk: true},    // auxint(flags) ? arg0 : arg1, 64-bit float
   556  		{name: "FCSELS", argLength: 3, reg: fp21, asm: "FCSELS", aux: "CCop", earlyOk: true},    // auxint(flags) ? arg0 : arg1, 32-bit float
   557  		{name: "CSETM", argLength: 1, reg: readflags, asm: "CSETM", aux: "CCop", earlyOk: true}, // auxint(flags) ? -1 : 0
   558  
   559  		// conditional comparison instructions; auxint is
   560  		// combination of Cond, Nzcv and optional ConstValue
   561  		// Behavior:
   562  		//   If the condition 'Cond' evaluates to true against current flags,
   563  		//   flags are set to the result of the comparison operation.
   564  		//   Otherwise, flags are set to the fallback value 'Nzcv'.
   565  		{name: "CCMP", argLength: 3, reg: gp2flagsflags, asm: "CCMP", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMP arg0 arg1 else flags = Nzcv
   566  		{name: "CCMN", argLength: 3, reg: gp2flagsflags, asm: "CCMN", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMN arg0 arg1 else flags = Nzcv
   567  		{name: "CCMPconst", argLength: 2, reg: gp1flagsflags, asm: "CCMP", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CMPconst [ConstValue] arg0 else flags = Nzcv
   568  		{name: "CCMNconst", argLength: 2, reg: gp1flagsflags, asm: "CCMN", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CMNconst [ConstValue] arg0 else flags = Nzcv
   569  
   570  		{name: "CCMPW", argLength: 3, reg: gp2flagsflags, asm: "CCMPW", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMPW arg0 arg1 else flags = Nzcv
   571  		{name: "CCMNW", argLength: 3, reg: gp2flagsflags, asm: "CCMNW", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMNW arg0 arg1 else flags = Nzcv
   572  		{name: "CCMPWconst", argLength: 2, reg: gp1flagsflags, asm: "CCMPW", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CCMPWconst [ConstValue] arg0 else flags = Nzcv
   573  		{name: "CCMNWconst", argLength: 2, reg: gp1flagsflags, asm: "CCMNW", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CCMNWconst [ConstValue] arg0 else flags = Nzcv
   574  
   575  		// function calls
   576  		{name: "CALLstatic", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                                       // call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
   577  		{name: "CALLtail", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},                                         // tail call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
   578  		{name: "CALLtailinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},             // tail call fn by pointer. arg0=codeptr, last arg=mem, auxint=argsize, returns mem
   579  		{name: "CALLclosure", argLength: -1, reg: regInfo{inputs: []regMask{gpsp, buildReg("R26"), regMask{}}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true}, // call function via closure.  arg0=codeptr, arg1=closure, last arg=mem, auxint=argsize, returns mem
   580  		{name: "CALLinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                 // call fn by pointer.  arg0=codeptr, last arg=mem, auxint=argsize, returns mem
   581  
   582  		// pseudo-ops
   583  		{name: "LoweredNilCheck", argLength: 2, reg: regInfo{inputs: []regMask{gpg}}, nilCheck: true, faultOnNilArg0: true},                                                                                                                                                      // panic if arg0 is nil.  arg1=mem.
   584  		{name: "LoweredMemEq", argLength: 4, reg: regInfo{inputs: []regMask{buildReg("R0"), buildReg("R1"), buildReg("R2")}, outputs: []regMask{buildReg("R0")}, clobbers: callerSave}, typ: "Bool", faultOnNilArg0: true, faultOnNilArg1: true, clobberFlags: true, call: true}, // arg0, arg1 - pointers to memory, arg2=size, arg3=mem.
   585  
   586  		{name: "Equal", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},            // bool, true flags encode x==y false otherwise.
   587  		{name: "NotEqual", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},         // bool, true flags encode x!=y false otherwise.
   588  		{name: "LessThan", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},         // bool, true flags encode signed x<y false otherwise.
   589  		{name: "LessEqual", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},        // bool, true flags encode signed x<=y false otherwise.
   590  		{name: "GreaterThan", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},      // bool, true flags encode signed x>y false otherwise.
   591  		{name: "GreaterEqual", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},     // bool, true flags encode signed x>=y false otherwise.
   592  		{name: "LessThanU", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},        // bool, true flags encode unsigned x<y false otherwise.
   593  		{name: "LessEqualU", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},       // bool, true flags encode unsigned x<=y false otherwise.
   594  		{name: "GreaterThanU", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},     // bool, true flags encode unsigned x>y false otherwise.
   595  		{name: "GreaterEqualU", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},    // bool, true flags encode unsigned x>=y false otherwise.
   596  		{name: "LessThanF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},        // bool, true flags encode floating-point x<y false otherwise.
   597  		{name: "LessEqualF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},       // bool, true flags encode floating-point x<=y false otherwise.
   598  		{name: "GreaterThanF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},     // bool, true flags encode floating-point x>y false otherwise.
   599  		{name: "GreaterEqualF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},    // bool, true flags encode floating-point x>=y false otherwise.
   600  		{name: "NotLessThanF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},     // bool, true flags encode floating-point x>=y || x is unordered with y, false otherwise.
   601  		{name: "NotLessEqualF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},    // bool, true flags encode floating-point x>y || x is unordered with y, false otherwise.
   602  		{name: "NotGreaterThanF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},  // bool, true flags encode floating-point x<=y || x is unordered with y, false otherwise.
   603  		{name: "NotGreaterEqualF", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56}, // bool, true flags encode floating-point x<y || x is unordered with y, false otherwise.
   604  		{name: "LessThanNoov", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56},     // bool, true flags encode signed x<y but without honoring overflow, false otherwise.
   605  		{name: "GreaterEqualNoov", argLength: 1, reg: readflags, earlyOk: true, zeroUpperBits: 56}, // bool, true flags encode signed x>=y but without honoring overflow, false otherwise.
   606  
   607  		// medium zeroing
   608  		// arg0 = address of memory to zero
   609  		// arg1 = mem
   610  		// auxint = # of bytes to zero
   611  		// returns mem
   612  		{
   613  			name:      "LoweredZero",
   614  			aux:       "Int64",
   615  			argLength: 2,
   616  			reg: regInfo{
   617  				inputs: []regMask{gp},
   618  			},
   619  			faultOnNilArg0: true,
   620  			addrSinkArg0:   true,
   621  		},
   622  
   623  		// large zeroing
   624  		// arg0 = address of memory to zero
   625  		// arg1 = mem
   626  		// auxint = # of bytes to zero
   627  		// returns mem
   628  		{
   629  			name:      "LoweredZeroLoop",
   630  			aux:       "Int64",
   631  			argLength: 2,
   632  			reg: regInfo{
   633  				inputs:       []regMask{gp},
   634  				clobbersArg0: true,
   635  			},
   636  			faultOnNilArg0: true,
   637  			addrSinkArg0:   true,
   638  			needIntTemp:    true,
   639  		},
   640  
   641  		// medium copying
   642  		// arg0 = address of dst memory
   643  		// arg1 = address of src memory
   644  		// arg2 = mem
   645  		// auxint = # of bytes to copy
   646  		// returns mem
   647  		{
   648  			name:      "LoweredMove",
   649  			aux:       "Int64",
   650  			argLength: 3,
   651  			reg: regInfo{
   652  				inputs:   []regMask{gp.minus(r25), gp.minus(r25)},
   653  				clobbers: r25.union(f16to17), // TODO: figure out needIntTemp + x2 for floats
   654  			},
   655  			faultOnNilArg0: true,
   656  			faultOnNilArg1: true,
   657  			addrSinkArg0:   true,
   658  			addrSinkArg1:   true,
   659  		},
   660  
   661  		// large copying
   662  		// arg0 = address of dst memory
   663  		// arg1 = address of src memory
   664  		// arg2 = mem
   665  		// auxint = # of bytes to copy
   666  		// returns mem
   667  		{
   668  			name:      "LoweredMoveLoop",
   669  			aux:       "Int64",
   670  			argLength: 3,
   671  			reg: regInfo{
   672  				inputs:       []regMask{gp.minus(r24to25), gp.minus(r24to25)},
   673  				clobbers:     r24to25.union(f16to17), // TODO: figure out needIntTemp x2 + x2 for floats
   674  				clobbersArg0: true,
   675  				clobbersArg1: true,
   676  			},
   677  			faultOnNilArg0: true,
   678  			faultOnNilArg1: true,
   679  			addrSinkArg0:   true,
   680  			addrSinkArg1:   true,
   681  		},
   682  
   683  		// Scheduler ensures LoweredGetClosurePtr occurs only in entry block,
   684  		// and sorts it to the very beginning of the block to prevent other
   685  		// use of R26 (arm64.REGCTXT, the closure pointer)
   686  		{name: "LoweredGetClosurePtr", reg: regInfo{outputs: []regMask{buildReg("R26")}}, zeroWidth: true},
   687  
   688  		// LoweredGetCallerSP returns the SP of the caller of the current function. arg0=mem
   689  		{name: "LoweredGetCallerSP", argLength: 1, reg: gp01, rematerializeable: true},
   690  
   691  		// LoweredGetCallerPC evaluates to the PC to which its "caller" will return.
   692  		// I.e., if f calls g "calls" sys.GetCallerPC,
   693  		// the result should be the PC within f that g will return to.
   694  		// See runtime/stubs.go for a more detailed discussion.
   695  		{name: "LoweredGetCallerPC", reg: gp01, rematerializeable: true},
   696  
   697  		// Constant flag value.
   698  		// Note: there's an "unordered" outcome for floating-point
   699  		// comparisons, but we don't use such a beast yet.
   700  		// This op is for temporary use by rewrite rules. It
   701  		// cannot appear in the generated assembly.
   702  		{name: "FlagConstant", aux: "FlagConstant"},
   703  
   704  		// (InvertFlags (CMP a b)) == (CMP b a)
   705  		// InvertFlags is a pseudo-op which can't appear in assembly output.
   706  		{name: "InvertFlags", argLength: 1}, // reverse direction of arg0
   707  
   708  		// atomic loads.
   709  		// load from arg0. arg1=mem. auxint must be zero.
   710  		// returns <value,memory> so they can be properly ordered with other loads.
   711  		{name: "LDAR", argLength: 2, reg: gpload, asm: "LDAR", faultOnNilArg0: true},
   712  		{name: "LDARB", argLength: 2, reg: gpload, asm: "LDARB", faultOnNilArg0: true, zeroUpperBits: 56},
   713  		{name: "LDARW", argLength: 2, reg: gpload, asm: "LDARW", faultOnNilArg0: true, zeroUpperBits: 32},
   714  
   715  		// atomic stores.
   716  		// store arg1 to arg0. arg2=mem. returns memory. auxint must be zero.
   717  		{name: "STLRB", argLength: 3, reg: gpstore, asm: "STLRB", faultOnNilArg0: true, hasSideEffects: true},
   718  		{name: "STLR", argLength: 3, reg: gpstore, asm: "STLR", faultOnNilArg0: true, hasSideEffects: true},
   719  		{name: "STLRW", argLength: 3, reg: gpstore, asm: "STLRW", faultOnNilArg0: true, hasSideEffects: true},
   720  
   721  		// atomic exchange.
   722  		// store arg1 to arg0. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   723  		// LDAXR	(Rarg0), Rout
   724  		// STLXR	Rarg1, (Rarg0), Rtmp
   725  		// CBNZ		Rtmp, -2(PC)
   726  		{name: "LoweredAtomicExchange64", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   727  		{name: "LoweredAtomicExchange32", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 32},
   728  		{name: "LoweredAtomicExchange8", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 56},
   729  
   730  		// atomic exchange variant.
   731  		// store arg1 to arg0. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   732  		// SWPALD	Rarg1, (Rarg0), Rout
   733  		{name: "LoweredAtomicExchange64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   734  		{name: "LoweredAtomicExchange32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, zeroUpperBits: 32},
   735  		{name: "LoweredAtomicExchange8Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 56},
   736  
   737  		// atomic add.
   738  		// *arg0 += arg1. arg2=mem. returns <new content of *arg0, memory>. auxint must be zero.
   739  		// LDAXR	(Rarg0), Rout
   740  		// ADD		Rarg1, Rout
   741  		// STLXR	Rout, (Rarg0), Rtmp
   742  		// CBNZ		Rtmp, -3(PC)
   743  		// Unlike the other 32-bit atomics, no zeroUpperBits: the final write
   744  		// to Rout is the 64-bit ADD.
   745  		{name: "LoweredAtomicAdd64", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   746  		{name: "LoweredAtomicAdd32", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   747  
   748  		// atomic add variant.
   749  		// *arg0 += arg1. arg2=mem. returns <new content of *arg0, memory>. auxint must be zero.
   750  		// LDADDAL	(Rarg0), Rarg1, Rout
   751  		// ADD		Rarg1, Rout
   752  		{name: "LoweredAtomicAdd64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   753  		{name: "LoweredAtomicAdd32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   754  
   755  		// atomic compare and swap.
   756  		// arg0 = pointer, arg1 = old value, arg2 = new value, arg3 = memory. auxint must be zero.
   757  		// if *arg0 == arg1 {
   758  		//   *arg0 = arg2
   759  		//   return (true, memory)
   760  		// } else {
   761  		//   return (false, memory)
   762  		// }
   763  		// LDAXR	(Rarg0), Rtmp
   764  		// CMP		Rarg1, Rtmp
   765  		// BNE		3(PC)
   766  		// STLXR	Rarg2, (Rarg0), Rtmp
   767  		// CBNZ		Rtmp, -4(PC)
   768  		// CSET		EQ, Rout
   769  		{name: "LoweredAtomicCas64", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 56},
   770  		{name: "LoweredAtomicCas32", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 56},
   771  
   772  		// atomic compare and swap variant.
   773  		// arg0 = pointer, arg1 = old value, arg2 = new value, arg3 = memory. auxint must be zero.
   774  		// if *arg0 == arg1 {
   775  		//   *arg0 = arg2
   776  		//   return (true, memory)
   777  		// } else {
   778  		//   return (false, memory)
   779  		// }
   780  		// MOV  	Rarg1, Rtmp
   781  		// CASAL	Rtmp, (Rarg0), Rarg2
   782  		// CMP  	Rarg1, Rtmp
   783  		// CSET 	EQ, Rout
   784  		{name: "LoweredAtomicCas64Variant", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 56},
   785  		{name: "LoweredAtomicCas32Variant", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 56},
   786  
   787  		// atomic and/or.
   788  		// *arg0 &= (|=) arg1. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   789  		// LDAXR	(Rarg0), Rout
   790  		// AND/OR	Rarg1, Rout, tempReg
   791  		// STLXR	tempReg, (Rarg0), Rtmp
   792  		// CBNZ		Rtmp, -3(PC)
   793  		{name: "LoweredAtomicAnd8", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "AND", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true, zeroUpperBits: 56},
   794  		{name: "LoweredAtomicOr8", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "ORR", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true, zeroUpperBits: 56},
   795  		{name: "LoweredAtomicAnd64", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "AND", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   796  		{name: "LoweredAtomicOr64", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "ORR", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   797  		{name: "LoweredAtomicAnd32", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "AND", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true, zeroUpperBits: 32},
   798  		{name: "LoweredAtomicOr32", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "ORR", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true, zeroUpperBits: 32},
   799  
   800  		// atomic and/or variant.
   801  		// *arg0 &= (|=) arg1. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   802  		//   AND:
   803  		// MNV       Rarg1, Rtemp
   804  		// LDANDALB  Rtemp, (Rarg0), Rout
   805  		//   OR:
   806  		// LDORALB  Rarg1, (Rarg0), Rout
   807  		{name: "LoweredAtomicAnd8Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 56},
   808  		{name: "LoweredAtomicOr8Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, zeroUpperBits: 56},
   809  		{name: "LoweredAtomicAnd64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   810  		{name: "LoweredAtomicOr64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   811  		{name: "LoweredAtomicAnd32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, zeroUpperBits: 32},
   812  		{name: "LoweredAtomicOr32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, zeroUpperBits: 32},
   813  
   814  		// LoweredWB invokes runtime.gcWriteBarrier. arg0=mem, auxint=# of buffer entries needed
   815  		// It saves all GP registers if necessary,
   816  		// but clobbers R30 (LR) because it's a call.
   817  		// R16 and R17 may be clobbered by linker trampoline.
   818  		// Returns a pointer to a write barrier buffer in R25.
   819  		{name: "LoweredWB", argLength: 1, reg: regInfo{clobbers: callerSave.minus(gpg).union(buildReg("R16 R17 R30")), outputs: []regMask{buildReg("R25")}}, clobberFlags: true, aux: "Int64"},
   820  
   821  		// LoweredPanicBoundsRR takes x and y, two values that caused a bounds check to fail.
   822  		// the RC and CR versions are used when one of the arguments is a constant. CC is used
   823  		// when both are constant (normally both 0, as prove derives the fact that a [0] bounds
   824  		// failure means the length must have also been 0).
   825  		// AuxInt contains a report code (see PanicBounds in genericOps.go).
   826  		{name: "LoweredPanicBoundsRR", argLength: 3, aux: "Int64", reg: regInfo{inputs: []regMask{first16, first16}}, typ: "Mem", call: true}, // arg0=x, arg1=y, arg2=mem, returns memory.
   827  		{name: "LoweredPanicBoundsRC", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{first16}}, typ: "Mem", call: true},   // arg0=x, arg1=mem, returns memory.
   828  		{name: "LoweredPanicBoundsCR", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{first16}}, typ: "Mem", call: true},   // arg0=y, arg1=mem, returns memory.
   829  		{name: "LoweredPanicBoundsCC", argLength: 1, aux: "PanicBoundsCC", reg: regInfo{}, typ: "Mem", call: true},                            // arg0=mem, returns memory.
   830  
   831  		// Prefetch instruction
   832  		// Do prefetch arg0 address with option aux. arg0=addr, arg1=memory, aux=option.
   833  		{name: "PRFM", argLength: 2, aux: "Int64", reg: prefreg, asm: "PRFM", hasSideEffects: true},
   834  
   835  		// Publication barrier
   836  		{name: "DMB", argLength: 1, aux: "Int64", asm: "DMB", hasSideEffects: true},       // Do data barrier. arg0=memory, aux=option.
   837  		{name: "ZERO", zeroWidth: true, fixedReg: true, earlyOk: true, zeroUpperBits: 56}, // reads-as-zero register
   838  
   839  		// Broadcast constant to each lane of a SIMD register. aux=constant.
   840  		// TODO: add the other arrangements after assembler supports them, to be used in simdgen-generated opt rules.
   841  		{name: "VMOVI16B", argLength: 0, reg: fp01, asm: "VMOVI", aux: "UInt8", commutative: false, typ: "Vec128", resultInArg0: false},
   842  
   843  		// SVE whole-register (unpredicated, VL-scaled) load/store of a scalable
   844  		// vector. These lower generic Load/Store of a 256-bit SIMD value; the
   845  		// scalable Z bank reuses the fp register masks.
   846  		{name: "ZLDRload", argLength: 2, reg: fpload, aux: "SymOff", asm: "ZLDR", typ: "Vec256", faultOnNilArg0: true, symEffect: "Read"}, // load from arg0 + auxInt + aux.  arg1=mem.
   847  		{name: "ZSTRstore", argLength: 3, reg: fpstore, aux: "SymOff", asm: "ZSTR", faultOnNilArg0: true, symEffect: "Write"},             // store arg1 to arg0 + auxInt + aux.  arg2=mem.
   848  		{name: "ZSELB", argLength: 3, reg: fp2predfp, asm: "ZSEL", typ: "Vec256"},                                                         // arg0=x, arg1=y, arg2=predicate; per-element select, constructive.
   849  		{name: "ZSELH", argLength: 3, reg: fp2predfp, asm: "ZSEL", typ: "Vec256"},                                                         // arg0=x, arg1=y, arg2=predicate; per-element select, constructive.
   850  		{name: "ZSELS", argLength: 3, reg: fp2predfp, asm: "ZSEL", typ: "Vec256"},                                                         // arg0=x, arg1=y, arg2=predicate; per-element select, constructive.
   851  		{name: "ZSELD", argLength: 3, reg: fp2predfp, asm: "ZSEL", typ: "Vec256"},                                                         // arg0=x, arg1=y, arg2=predicate; per-element select, constructive.
   852  		{name: "PLDRload", argLength: 2, reg: predload, aux: "SymOff", asm: "PLDR", typ: "Mask", faultOnNilArg0: true, symEffect: "Read"}, // load a predicate from arg0 + auxInt + aux.  arg1=mem.
   853  		{name: "PSTRstore", argLength: 3, reg: predstore, aux: "SymOff", asm: "PSTR", faultOnNilArg0: true, symEffect: "Write"},           // store predicate arg1 to arg0 + auxInt + aux.  arg2=mem.
   854  		// PPFALSEB sets every bit of a predicate false, it's the zero value of a predicate.
   855  		{name: "PPFALSEB", argLength: 0, reg: pred01, asm: "PPFALSE", typ: "Mask"},
   856  		// PPNEXT{B,H,S,D} set only the first active element of arg1 that follows
   857  		// the last active element of arg0 (or the first active element of arg1
   858  		// when arg0 has none), at the given element size.
   859  		{name: "PPNEXTB", argLength: 2, reg: pred2pred, asm: "PPNEXT", typ: "Mask", resultInArg0: true, clobberFlags: true}, // arg0=predicate, arg1=candidates
   860  		{name: "PPNEXTH", argLength: 2, reg: pred2pred, asm: "PPNEXT", typ: "Mask", resultInArg0: true, clobberFlags: true}, // arg0=predicate, arg1=candidates
   861  		{name: "PPNEXTS", argLength: 2, reg: pred2pred, asm: "PPNEXT", typ: "Mask", resultInArg0: true, clobberFlags: true}, // arg0=predicate, arg1=candidates
   862  		{name: "PPNEXTD", argLength: 2, reg: pred2pred, asm: "PPNEXT", typ: "Mask", resultInArg0: true, clobberFlags: true}, // arg0=predicate, arg1=candidates
   863  		// PBICSB computes arg1 AND NOT arg2 on the lanes of arg0 and sets the
   864  		// flags from the result: Z if no lane is set.
   865  		{name: "PBICSB", argLength: 3, reg: pred3pred, asm: "PBICS", typ: "(Mask,Flags)"}, // arg0=governing predicate, arg1, arg2
   866  		// PPTEST sets the flags from the lanes of arg1 that arg0 governs: Z if
   867  		// none is set.
   868  		{name: "PPTEST", argLength: 2, reg: pred2flags, asm: "PPTEST", typ: "Flags"}, // arg0=governing predicate, arg1
   869  		// ZDUPBconst broadcasts an 8-bit immediate to every byte lane; with [0] it
   870  		// zeroes a whole scalable vector, lowering ZeroSIMD for a 256-bit value.
   871  		{name: "ZDUPBconst", argLength: 0, aux: "Int8", reg: fp01, asm: "ZDUP", typ: "Vec256"},
   872  		// ZDUP{H,S,D}const broadcast a signed 8-bit immediate to every lane of
   873  		// the wider widths.
   874  		{name: "ZDUPHconst", argLength: 0, aux: "Int8", reg: fp01, asm: "ZDUP", typ: "Vec256"},
   875  		{name: "ZDUPSconst", argLength: 0, aux: "Int8", reg: fp01, asm: "ZDUP", typ: "Vec256"},
   876  		{name: "ZDUPDconst", argLength: 0, aux: "Int8", reg: fp01, asm: "ZDUP", typ: "Vec256"},
   877  		// ZDUPB..ZDUPD broadcast a general register to every lane of a scalable
   878  		// vector.
   879  		{name: "ZDUPB", argLength: 1, reg: gpfp, asm: "ZDUPW", typ: "Vec256"}, // arg0=scalar
   880  		{name: "ZDUPH", argLength: 1, reg: gpfp, asm: "ZDUPW", typ: "Vec256"}, // arg0=scalar
   881  		{name: "ZDUPS", argLength: 1, reg: gpfp, asm: "ZDUPW", typ: "Vec256"}, // arg0=scalar
   882  		{name: "ZDUPD", argLength: 1, reg: gpfp, asm: "ZDUP", typ: "Vec256"},  // arg0=scalar
   883  		// ZDUPIB..ZDUPID broadcast element auxint of a vector register to every
   884  		// lane.
   885  		{name: "ZDUPIB", argLength: 1, aux: "UInt8", reg: fp11, asm: "ZDUP", typ: "Vec256"}, // arg0=vector, auxint=element index
   886  		{name: "ZDUPIH", argLength: 1, aux: "UInt8", reg: fp11, asm: "ZDUP", typ: "Vec256"}, // arg0=vector, auxint=element index
   887  		{name: "ZDUPIS", argLength: 1, aux: "UInt8", reg: fp11, asm: "ZDUP", typ: "Vec256"}, // arg0=vector, auxint=element index
   888  		{name: "ZDUPID", argLength: 1, aux: "UInt8", reg: fp11, asm: "ZDUP", typ: "Vec256"}, // arg0=vector, auxint=element index
   889  		// RDVL reads the architecture vector length in bytes (aux = scale). Used
   890  		// at package init to verify the hardware VL fits the fixed 256-bit model.
   891  		{name: "RDVL", argLength: 0, aux: "Int64", reg: gp01, asm: "RDVL", typ: "Int64"},
   892  
   893  		{name: "PWHILELTB", argLength: 2, reg: gp2pred, asm: "PWHILELT", typ: "(Mask,Flags)"},                                                       // arg0=lo, arg1=hi; predicate enabling byte (.B) lanes [lo,hi).
   894  		{name: "PWHILELTH", argLength: 2, reg: gp2pred, asm: "PWHILELT", typ: "(Mask,Flags)"},                                                       // as PWHILELTB but for .H (16-bit) lanes.
   895  		{name: "PWHILELTS", argLength: 2, reg: gp2pred, asm: "PWHILELT", typ: "(Mask,Flags)"},                                                       // as PWHILELTB but for .S (32-bit) lanes.
   896  		{name: "PWHILELTD", argLength: 2, reg: gp2pred, asm: "PWHILELT", typ: "(Mask,Flags)"},                                                       // as PWHILELTB but for .D (64-bit) lanes.
   897  		{name: "ZLD1BPredload", argLength: 3, reg: fppredload, aux: "SymOff", asm: "ZLD1B", typ: "Vec256", faultOnNilArg0: true, symEffect: "Read"}, // predicated load from arg0+auxInt+aux governed by arg1; arg2=mem.
   898  		{name: "ZST1BPredstore", argLength: 4, reg: fppredstore, aux: "SymOff", asm: "ZST1B", typ: "Mem", faultOnNilArg0: true, symEffect: "Write"}, // predicated store of arg1 to arg0+auxInt+aux governed by arg2; arg3=mem.
   899  	}
   900  
   901  	blocks := []blockData{
   902  		{name: "EQ", controls: 1},
   903  		{name: "NE", controls: 1},
   904  		{name: "LT", controls: 1},
   905  		{name: "LE", controls: 1},
   906  		{name: "GT", controls: 1},
   907  		{name: "GE", controls: 1},
   908  		{name: "ULT", controls: 1},
   909  		{name: "ULE", controls: 1},
   910  		{name: "UGT", controls: 1},
   911  		{name: "UGE", controls: 1},
   912  		{name: "Z", controls: 1},                  // Control == 0 (take a register instead of flags)
   913  		{name: "NZ", controls: 1},                 // Control != 0
   914  		{name: "ZW", controls: 1},                 // Control == 0, 32-bit
   915  		{name: "NZW", controls: 1},                // Control != 0, 32-bit
   916  		{name: "TBZ", controls: 1, aux: "Int64"},  // Control & (1 << AuxInt) == 0
   917  		{name: "TBNZ", controls: 1, aux: "Int64"}, // Control & (1 << AuxInt) != 0
   918  		{name: "FLT", controls: 1},
   919  		{name: "FLE", controls: 1},
   920  		{name: "FGT", controls: 1},
   921  		{name: "FGE", controls: 1},
   922  		{name: "LTnoov", controls: 1}, // 'LT' but without honoring overflow
   923  		{name: "LEnoov", controls: 1}, // 'LE' but without honoring overflow
   924  		{name: "GTnoov", controls: 1}, // 'GT' but without honoring overflow
   925  		{name: "GEnoov", controls: 1}, // 'GE' but without honoring overflow
   926  
   927  		// JUMPTABLE implements jump tables.
   928  		// Aux is the symbol (an *obj.LSym) for the jump table.
   929  		// control[0] is the index into the jump table.
   930  		// control[1] is the address of the jump table (the address of the symbol stored in Aux).
   931  		{name: "JUMPTABLE", controls: 2, aux: "Sym"},
   932  	}
   933  
   934  	archs = append(archs, arch{
   935  		name:               "ARM64",
   936  		pkg:                "cmd/internal/obj/arm64",
   937  		genfile:            "../../arm64/ssa.go",
   938  		genSIMDfile:        "../../arm64/simdssa.go ../../arm64/simdssa_sve.go",
   939  		ops:                append(append(ops, simdARM64Ops(fp11, fp21, fp31, fpgp, fpgpfp, fp21)...), simdARM64SVEOps(fp11, fp21, fp1predfp, fp2predpred, fp2predfp, fp2predfp, fp3predfp, fp3predfp)...),
   940  		blocks:             blocks,
   941  		regnames:           regNamesARM64,
   942  		ParamIntRegNames:   "R0 R1 R2 R3 R4 R5 R6 R7 R8 R9 R10 R11 R12 R13 R14 R15",
   943  		ParamFloatRegNames: "F0 F1 F2 F3 F4 F5 F6 F7 F8 F9 F10 F11 F12 F13 F14 F15",
   944  		gpregmask:          gp,
   945  		fpregmask:          fp,
   946  		simdregmask:        fp,
   947  		specialregmask:     pred,
   948  		framepointerreg:    -1, // not used
   949  		linkreg:            int8(num["R30"]),
   950  	})
   951  }
   952  

View as plain text