Source file src/cmd/compile/internal/ssa/_gen/AMD64Ops.go

     1  // Copyright 2015 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package main
     6  
     7  import "strings"
     8  
     9  // Notes:
    10  //  - Integer types live in the low portion of registers. Upper portions are junk.
    11  //  - Boolean types use the low-order byte of a register. 0=false, 1=true.
    12  //    Upper bytes are junk.
    13  //  - Floating-point types live in the low natural slot of an sse2 register.
    14  //    Unused portions are junk.
    15  //  - We do not use AH,BH,CH,DH registers.
    16  //  - When doing sub-register operations, we try to write the whole
    17  //    destination register to avoid a partial-register write.
    18  //  - Unused portions of AuxInt (or the Val portion of ValAndOff) are
    19  //    filled by sign-extending the used portion.  Users of AuxInt which interpret
    20  //    AuxInt as unsigned (e.g. shifts) must be careful.
    21  //  - All SymOff opcodes require their offset to fit in an int32.
    22  
    23  // Suffixes encode the bit width of various instructions.
    24  // Q (quad word) = 64 bit
    25  // L (long word) = 32 bit
    26  // W (word)      = 16 bit
    27  // B (byte)      = 8 bit
    28  // D (double)    = 64 bit float
    29  // S (single)    = 32 bit float
    30  
    31  // copied from ../../amd64/reg.go
    32  var regNamesAMD64 = []string{
    33  	"AX",
    34  	"CX",
    35  	"DX",
    36  	"BX",
    37  	"SP",
    38  	"BP",
    39  	"SI",
    40  	"DI",
    41  	"R8",
    42  	"R9",
    43  	"R10",
    44  	"R11",
    45  	"R12",
    46  	"R13",
    47  	"g", // a.k.a. R14
    48  	"R15",
    49  	"X0",
    50  	"X1",
    51  	"X2",
    52  	"X3",
    53  	"X4",
    54  	"X5",
    55  	"X6",
    56  	"X7",
    57  	"X8",
    58  	"X9",
    59  	"X10",
    60  	"X11",
    61  	"X12",
    62  	"X13",
    63  	"X14",
    64  	"X15", // constant 0 in ABIInternal
    65  	"X16",
    66  	"X17",
    67  	"X18",
    68  	"X19",
    69  	"X20",
    70  	"X21",
    71  	"X22",
    72  	"X23",
    73  	"X24",
    74  	"X25",
    75  	"X26",
    76  	"X27",
    77  	"X28",
    78  	"X29",
    79  	"X30",
    80  	"X31",
    81  
    82  	// TODO: update asyncPreempt for K registers.
    83  	// asyncPreempt also needs to store Z0-Z15 properly.
    84  	"K0",
    85  	"K1",
    86  	"K2",
    87  	"K3",
    88  	"K4",
    89  	"K5",
    90  	"K6",
    91  	"K7",
    92  	// If you add registers, update asyncPreempt in runtime
    93  
    94  	// pseudo-registers
    95  	"SB",
    96  }
    97  
    98  func init() {
    99  	// Make map from reg names to reg integers.
   100  	if len(regNamesAMD64) > 64 {
   101  		panic("too many registers")
   102  	}
   103  	num := map[string]int{}
   104  	for i, name := range regNamesAMD64 {
   105  		num[name] = i
   106  	}
   107  	buildReg := func(s string) regMask {
   108  		m := regMask{}
   109  		for _, r := range strings.Split(s, " ") {
   110  			if n, ok := num[r]; ok {
   111  				m = m.addReg(uint(n))
   112  				continue
   113  			}
   114  			panic("register " + r + " not found")
   115  		}
   116  		return m
   117  	}
   118  
   119  	// Common individual register masks
   120  	var (
   121  		ax         = buildReg("AX")
   122  		cx         = buildReg("CX")
   123  		dx         = buildReg("DX")
   124  		gp         = buildReg("AX CX DX BX BP SI DI R8 R9 R10 R11 R12 R13 R15")
   125  		g          = buildReg("g")
   126  		fp         = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14")
   127  		v          = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14")
   128  		w          = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14 X16 X17 X18 X19 X20 X21 X22 X23 X24 X25 X26 X27 X28 X29 X30 X31")
   129  		x15        = buildReg("X15")
   130  		mask       = buildReg("K1 K2 K3 K4 K5 K6 K7")
   131  		gpsp       = gp.union(buildReg("SP"))
   132  		gpspsb     = gpsp.union(buildReg("SB"))
   133  		gpspsbg    = gpspsb.union(g)
   134  		callerSave = gp.union(w).union(mask).union(g) // runtime.setg (and anything calling it) may clobber g
   135  
   136  		vz = v.union(x15)
   137  		wz = w.union(x15)
   138  		x0 = buildReg("X0")
   139  	)
   140  	// Common slices of register masks
   141  	var (
   142  		gponly   = []regMask{gp}
   143  		fponly   = []regMask{fp}
   144  		vonly    = []regMask{v}
   145  		wonly    = []regMask{w}
   146  		maskonly = []regMask{mask}
   147  		vzonly   = []regMask{vz}
   148  		wzonly   = []regMask{wz}
   149  	)
   150  
   151  	// Common regInfo
   152  	var (
   153  		gp01           = regInfo{inputs: nil, outputs: gponly}
   154  		gp11           = regInfo{inputs: []regMask{gp}, outputs: gponly}
   155  		gp11sp         = regInfo{inputs: []regMask{gpsp}, outputs: gponly}
   156  		gp11sb         = regInfo{inputs: []regMask{gpspsbg}, outputs: gponly}
   157  		gp21           = regInfo{inputs: []regMask{gp, gp}, outputs: gponly}
   158  		gp21sp         = regInfo{inputs: []regMask{gpsp, gp}, outputs: gponly}
   159  		gp21sp2        = regInfo{inputs: []regMask{gp, gpsp}, outputs: gponly}
   160  		gp21sb         = regInfo{inputs: []regMask{gpspsbg, gpsp}, outputs: gponly}
   161  		gp21shift      = regInfo{inputs: []regMask{gp, cx}, outputs: []regMask{gp}}
   162  		gp11div        = regInfo{inputs: []regMask{ax, gpsp.minus(dx)}, outputs: []regMask{ax, dx}}
   163  		gp21hmul       = regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{dx}, clobbers: ax}
   164  		gp21flags      = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp, regMask{}}}
   165  		gp2flags1flags = regInfo{inputs: []regMask{gp, gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   166  
   167  		gp2flags     = regInfo{inputs: []regMask{gpsp, gpsp}}
   168  		gp1flags     = regInfo{inputs: []regMask{gpsp}}
   169  		gp0flagsLoad = regInfo{inputs: []regMask{gpspsbg, regMask{}}}
   170  		gp1flagsLoad = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   171  		gp2flagsLoad = regInfo{inputs: []regMask{gpspsbg, gpsp, gpsp, regMask{}}}
   172  		flagsgp      = regInfo{inputs: nil, outputs: gponly}
   173  
   174  		gp11flags      = regInfo{inputs: []regMask{gp}, outputs: []regMask{gp, regMask{}}}
   175  		gp1flags1flags = regInfo{inputs: []regMask{gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   176  
   177  		readflags = regInfo{inputs: nil, outputs: gponly}
   178  
   179  		gpload         = regInfo{inputs: []regMask{gpspsbg, regMask{}}, outputs: gponly}
   180  		gp21load       = regInfo{inputs: []regMask{gp, gpspsbg, regMask{}}, outputs: gponly}
   181  		gploadidx      = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}, outputs: gponly}
   182  		gp21loadidx    = regInfo{inputs: []regMask{gp, gpspsbg, gpsp, regMask{}}, outputs: gponly}
   183  		gp21shxload    = regInfo{inputs: []regMask{gpspsbg, gp, regMask{}}, outputs: gponly}
   184  		gp21shxloadidx = regInfo{inputs: []regMask{gpspsbg, gpsp, gp, regMask{}}, outputs: gponly}
   185  
   186  		gpstore         = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   187  		gpstoreconst    = regInfo{inputs: []regMask{gpspsbg, regMask{}}}
   188  		gpstoreidx      = regInfo{inputs: []regMask{gpspsbg, gpsp, gpsp, regMask{}}}
   189  		gpstoreconstidx = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   190  		gpstorexchg     = regInfo{inputs: []regMask{gp, gpspsbg, regMask{}}, outputs: []regMask{gp}}
   191  		cmpxchg         = regInfo{inputs: []regMask{gp, ax, gp, regMask{}}, outputs: []regMask{gp, regMask{}}, clobbers: ax}
   192  		atomicLogic     = regInfo{inputs: []regMask{gp.minus(ax), gp.minus(ax), regMask{}}, outputs: []regMask{ax, regMask{}}}
   193  
   194  		fp01        = regInfo{inputs: nil, outputs: fponly}
   195  		fp21        = regInfo{inputs: []regMask{fp, fp}, outputs: fponly}
   196  		fp31        = regInfo{inputs: []regMask{fp, fp, fp}, outputs: fponly}
   197  		fp21load    = regInfo{inputs: []regMask{fp, gpspsbg, regMask{}}, outputs: fponly}
   198  		fp21loadidx = regInfo{inputs: []regMask{fp, gpspsbg, gpspsb, regMask{}}, outputs: fponly}
   199  		fpgp        = regInfo{inputs: fponly, outputs: gponly}
   200  		gpfp        = regInfo{inputs: gponly, outputs: fponly}
   201  		fp11        = regInfo{inputs: fponly, outputs: fponly}
   202  		fp2flags    = regInfo{inputs: []regMask{fp, fp}}
   203  
   204  		fpload    = regInfo{inputs: []regMask{gpspsb, {}}, outputs: fponly}
   205  		fploadidx = regInfo{inputs: []regMask{gpspsb, gpsp, {}}, outputs: fponly}
   206  		vload     = regInfo{inputs: []regMask{gpspsb, {}}, outputs: vonly}
   207  		wload     = regInfo{inputs: []regMask{gpspsb, {}}, outputs: wonly}
   208  
   209  		fpstore    = regInfo{inputs: []regMask{gpspsb, fp, {}}}
   210  		fpstoreidx = regInfo{inputs: []regMask{gpspsb, gpsp, fp, {}}}
   211  		vstore     = regInfo{inputs: []regMask{gpspsb, vz, {}}}
   212  		wstore     = regInfo{inputs: []regMask{gpspsb, wz, {}}}
   213  
   214  		// masked loads/stores, vector register or mask register
   215  		vloadv  = regInfo{inputs: []regMask{gpspsb, v, {}}, outputs: vonly}
   216  		vstorev = regInfo{inputs: []regMask{gpspsb, v, vz, {}}}
   217  		wloadk  = regInfo{inputs: []regMask{gpspsb, mask, {}}, outputs: wonly}
   218  		wstorek = regInfo{inputs: []regMask{gpspsb, mask, wz, {}}}
   219  
   220  		v11     = regInfo{inputs: vonly, outputs: vonly}            // used in resultInArg0 ops, arg0 must not be x15
   221  		v21     = regInfo{inputs: []regMask{v, vz}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   222  		vk      = regInfo{inputs: vzonly, outputs: maskonly}
   223  		kv      = regInfo{inputs: maskonly, outputs: vonly}
   224  		v2k     = regInfo{inputs: []regMask{vz, vz}, outputs: maskonly}
   225  		vkv     = regInfo{inputs: []regMask{vz, mask}, outputs: vonly}
   226  		v2kv    = regInfo{inputs: []regMask{vz, vz, mask}, outputs: vonly}
   227  		v2kk    = regInfo{inputs: []regMask{vz, vz, mask}, outputs: maskonly}
   228  		v31     = regInfo{inputs: []regMask{v, vz, vz}, outputs: vonly}       // used in resultInArg0 ops, arg0 must not be x15
   229  		v3kv    = regInfo{inputs: []regMask{v, vz, vz, mask}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   230  		vgpv    = regInfo{inputs: []regMask{vz, gp}, outputs: vonly}
   231  		vgp     = regInfo{inputs: vonly, outputs: gponly}
   232  		vfpv    = regInfo{inputs: []regMask{vz, fp}, outputs: vonly}
   233  		vfpkv   = regInfo{inputs: []regMask{vz, fp, mask}, outputs: vonly}
   234  		fpv     = regInfo{inputs: []regMask{fp}, outputs: vonly}
   235  		gpv     = regInfo{inputs: []regMask{gp}, outputs: vonly}
   236  		v2flags = regInfo{inputs: []regMask{vz, vz}}
   237  
   238  		w11   = regInfo{inputs: wonly, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   239  		w21   = regInfo{inputs: []regMask{wz, wz}, outputs: wonly}
   240  		wk    = regInfo{inputs: wzonly, outputs: maskonly}
   241  		kw    = regInfo{inputs: maskonly, outputs: wonly}
   242  		w2k   = regInfo{inputs: []regMask{wz, wz}, outputs: maskonly}
   243  		wkw   = regInfo{inputs: []regMask{wz, mask}, outputs: wonly}
   244  		w2kw  = regInfo{inputs: []regMask{w, wz, mask}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   245  		w2kk  = regInfo{inputs: []regMask{wz, wz, mask}, outputs: maskonly}
   246  		w31   = regInfo{inputs: []regMask{w, wz, wz}, outputs: wonly}       // used in resultInArg0 ops, arg0 must not be x15
   247  		w3kw  = regInfo{inputs: []regMask{w, wz, wz, mask}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   248  		wgpw  = regInfo{inputs: []regMask{wz, gp}, outputs: wonly}
   249  		wgp   = regInfo{inputs: wzonly, outputs: gponly}
   250  		wfpw  = regInfo{inputs: []regMask{wz, fp}, outputs: wonly}
   251  		wfpkw = regInfo{inputs: []regMask{wz, fp, mask}, outputs: wonly}
   252  
   253  		// These register masks are used by SIMD only, they follow the pattern:
   254  		// Mem last, k mask second to last (if any), address right before mem and k mask.
   255  		wkwload    = regInfo{inputs: []regMask{gpspsb, mask, regMask{}}, outputs: wonly}
   256  		v21load    = regInfo{inputs: []regMask{v, gpspsb, regMask{}}, outputs: vonly}     // used in resultInArg0 ops, arg0 must not be x15
   257  		v31load    = regInfo{inputs: []regMask{v, vz, gpspsb, regMask{}}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   258  		v11load    = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: vonly}
   259  		w21load    = regInfo{inputs: []regMask{wz, gpspsb, regMask{}}, outputs: wonly}
   260  		w31load    = regInfo{inputs: []regMask{w, wz, gpspsb, regMask{}}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   261  		w2kload    = regInfo{inputs: []regMask{wz, gpspsb, regMask{}}, outputs: maskonly}
   262  		w2kwload   = regInfo{inputs: []regMask{wz, gpspsb, mask, regMask{}}, outputs: wonly}
   263  		w11load    = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: wonly}
   264  		w3kwload   = regInfo{inputs: []regMask{w, wz, gpspsb, mask, regMask{}}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   265  		w2kkload   = regInfo{inputs: []regMask{wz, gpspsb, mask, regMask{}}, outputs: maskonly}
   266  		v31x0AtIn2 = regInfo{inputs: []regMask{v, vz, x0}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   267  
   268  		kload  = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: maskonly}
   269  		kstore = regInfo{inputs: []regMask{gpspsb, mask, regMask{}}}
   270  		gpk    = regInfo{inputs: gponly, outputs: maskonly}
   271  		kgp    = regInfo{inputs: maskonly, outputs: gponly}
   272  		k2k    = regInfo{inputs: []regMask{mask, mask}, outputs: maskonly}
   273  
   274  		x15only = regInfo{inputs: nil, outputs: []regMask{x15}}
   275  
   276  		prefreg = regInfo{inputs: []regMask{gpspsbg}}
   277  	)
   278  
   279  	var AMD64ops = []opData{
   280  		// {ADD,SUB,MUL,DIV}Sx: floating-point arithmetic
   281  		// x==S for float32, x==D for float64
   282  		// computes arg0 OP arg1
   283  		{name: "ADDSS", argLength: 2, reg: fp21, asm: "ADDSS", commutative: true, resultInArg0: true, earlyOk: true},
   284  		{name: "ADDSD", argLength: 2, reg: fp21, asm: "ADDSD", commutative: true, resultInArg0: true, earlyOk: true},
   285  		{name: "SUBSS", argLength: 2, reg: fp21, asm: "SUBSS", resultInArg0: true, earlyOk: true},
   286  		{name: "SUBSD", argLength: 2, reg: fp21, asm: "SUBSD", resultInArg0: true, earlyOk: true},
   287  		{name: "MULSS", argLength: 2, reg: fp21, asm: "MULSS", commutative: true, resultInArg0: true, earlyOk: true},
   288  		{name: "MULSD", argLength: 2, reg: fp21, asm: "MULSD", commutative: true, resultInArg0: true, earlyOk: true},
   289  		{name: "DIVSS", argLength: 2, reg: fp21, asm: "DIVSS", resultInArg0: true, earlyOk: true},
   290  		{name: "DIVSD", argLength: 2, reg: fp21, asm: "DIVSD", resultInArg0: true, earlyOk: true},
   291  
   292  		// MOVSxload: floating-point loads
   293  		// x==S for float32, x==D for float64
   294  		// load from arg0+auxint+aux, arg1 = mem
   295  		{name: "MOVSSload", argLength: 2, reg: fpload, asm: "MOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   296  		{name: "MOVSDload", argLength: 2, reg: fpload, asm: "MOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   297  
   298  		// MOVSxconst: floatint-point constants
   299  		// x==S for float32, x==D for float64
   300  		{name: "MOVSSconst", reg: fp01, asm: "MOVSS", aux: "Float32", rematerializeable: true, earlyOk: true},
   301  		{name: "MOVSDconst", reg: fp01, asm: "MOVSD", aux: "Float64", rematerializeable: true, earlyOk: true},
   302  
   303  		// MOVSxloadidx: floating-point indexed loads
   304  		// x==S for float32, x==D for float64
   305  		// load from arg0 + scale*arg1+auxint+aux, arg2 = mem
   306  		{name: "MOVSSloadidx1", argLength: 3, reg: fploadidx, asm: "MOVSS", scale: 1, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   307  		{name: "MOVSSloadidx4", argLength: 3, reg: fploadidx, asm: "MOVSS", scale: 4, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   308  		{name: "MOVSDloadidx1", argLength: 3, reg: fploadidx, asm: "MOVSD", scale: 1, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   309  		{name: "MOVSDloadidx8", argLength: 3, reg: fploadidx, asm: "MOVSD", scale: 8, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   310  
   311  		// MOVSxstore: floating-point stores
   312  		// x==S for float32, x==D for float64
   313  		// does *(arg0+auxint+aux) = arg1, arg2 = mem
   314  		{name: "MOVSSstore", argLength: 3, reg: fpstore, asm: "MOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   315  		{name: "MOVSDstore", argLength: 3, reg: fpstore, asm: "MOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   316  
   317  		// MOVSxstoreidx: floating-point indexed stores
   318  		// x==S for float32, x==D for float64
   319  		// does *(arg0+scale*arg1+auxint+aux) = arg2, arg3 = mem
   320  		{name: "MOVSSstoreidx1", argLength: 4, reg: fpstoreidx, asm: "MOVSS", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   321  		{name: "MOVSSstoreidx4", argLength: 4, reg: fpstoreidx, asm: "MOVSS", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   322  		{name: "MOVSDstoreidx1", argLength: 4, reg: fpstoreidx, asm: "MOVSD", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   323  		{name: "MOVSDstoreidx8", argLength: 4, reg: fpstoreidx, asm: "MOVSD", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   324  
   325  		// {ADD,SUB,MUL,DIV}Sxload: floating-point load / op combo
   326  		// x==S for float32, x==D for float64
   327  		// computes arg0 OP *(arg1+auxint+aux), arg2=mem
   328  		{name: "ADDSSload", argLength: 3, reg: fp21load, asm: "ADDSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   329  		{name: "ADDSDload", argLength: 3, reg: fp21load, asm: "ADDSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   330  		{name: "SUBSSload", argLength: 3, reg: fp21load, asm: "SUBSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   331  		{name: "SUBSDload", argLength: 3, reg: fp21load, asm: "SUBSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   332  		{name: "MULSSload", argLength: 3, reg: fp21load, asm: "MULSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   333  		{name: "MULSDload", argLength: 3, reg: fp21load, asm: "MULSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   334  		{name: "DIVSSload", argLength: 3, reg: fp21load, asm: "DIVSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   335  		{name: "DIVSDload", argLength: 3, reg: fp21load, asm: "DIVSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   336  
   337  		// {ADD,SUB,MUL,DIV}Sxloadidx: floating-point indexed load / op combo
   338  		// x==S for float32, x==D for float64
   339  		// computes arg0 OP *(arg1+scale*arg2+auxint+aux), arg3=mem
   340  		{name: "ADDSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "ADDSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   341  		{name: "ADDSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "ADDSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   342  		{name: "ADDSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "ADDSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   343  		{name: "ADDSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "ADDSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   344  		{name: "SUBSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "SUBSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   345  		{name: "SUBSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "SUBSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   346  		{name: "SUBSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "SUBSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   347  		{name: "SUBSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "SUBSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   348  		{name: "MULSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "MULSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   349  		{name: "MULSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "MULSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   350  		{name: "MULSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "MULSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   351  		{name: "MULSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "MULSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   352  		{name: "DIVSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "DIVSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   353  		{name: "DIVSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "DIVSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   354  		{name: "DIVSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "DIVSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   355  		{name: "DIVSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "DIVSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   356  
   357  		// {ADD,SUB,MUL,DIV,AND,OR,XOR}x: binary integer ops
   358  		//   unadorned versions compute arg0 OP arg1
   359  		//       const versions compute arg0 OP auxint (auxint is a sign-extended 32-bit value)
   360  		// constmodify versions compute *(arg0+ValAndOff(AuxInt).Off().aux) OP= ValAndOff(AuxInt).Val(), arg1 = mem
   361  		// x==L operations zero the upper 4 bytes of the destination register (not meaningful for constmodify versions).
   362  		{name: "ADDQ", argLength: 2, reg: gp21sp, asm: "ADDQ", commutative: true, clobberFlags: true, earlyOk: true},
   363  		{name: "ADDL", argLength: 2, reg: gp21sp, asm: "ADDL", commutative: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   364  		{name: "ADDQconst", argLength: 1, reg: gp11sp, asm: "ADDQ", aux: "Int32", typ: "UInt64", clobberFlags: true, earlyOk: true},
   365  		{name: "ADDLconst", argLength: 1, reg: gp11sp, asm: "ADDL", aux: "Int32", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   366  		{name: "ADDQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   367  		{name: "ADDLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   368  		{name: "ADDWconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   369  		{name: "ADDBconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   370  
   371  		{name: "SUBQ", argLength: 2, reg: gp21sp2, asm: "SUBQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   372  		{name: "SUBL", argLength: 2, reg: gp21, asm: "SUBL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   373  		{name: "SUBQconst", argLength: 1, reg: gp11, asm: "SUBQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},
   374  		{name: "SUBLconst", argLength: 1, reg: gp11, asm: "SUBL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   375  
   376  		{name: "MULQ", argLength: 2, reg: gp21, asm: "IMULQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},
   377  		{name: "MULL", argLength: 2, reg: gp21, asm: "IMULL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   378  		{name: "MULQconst", argLength: 1, reg: gp11, asm: "IMUL3Q", aux: "Int32", clobberFlags: true, earlyOk: true},
   379  		{name: "MULLconst", argLength: 1, reg: gp11, asm: "IMUL3L", aux: "Int32", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   380  
   381  		// Let x = arg0*arg1 (full 32x32->64  unsigned multiply). Returns uint32(x), and flags set to overflow if uint32(x) != x.
   382  		{name: "MULLU", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{ax, regMask{}}, clobbers: dx}, typ: "(UInt32,Flags)", asm: "MULL", commutative: true, clobberFlags: true, zeroUpperBits: 32},
   383  		// Let x = arg0*arg1 (full 64x64->128 unsigned multiply). Returns uint64(x), and flags set to overflow if uint64(x) != x.
   384  		{name: "MULQU", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{ax, regMask{}}, clobbers: dx}, typ: "(UInt64,Flags)", asm: "MULQ", commutative: true, clobberFlags: true},
   385  
   386  		// HMULx[U]: computes the high bits of an integer multiply.
   387  		// computes arg0 * arg1 >> (x==L?32:64)
   388  		// The multiply is unsigned for the U versions, signed for the non-U versions.
   389  		// HMULx[U] are intentionally not marked as commutative, even though they are.
   390  		// This is because they have asymmetric register requirements.
   391  		// There are rewrite rules to try to place arguments in preferable slots.
   392  		{name: "HMULQ", argLength: 2, reg: gp21hmul, asm: "IMULQ", clobberFlags: true, earlyOk: true},
   393  		{name: "HMULL", argLength: 2, reg: gp21hmul, asm: "IMULL", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   394  		{name: "HMULQU", argLength: 2, reg: gp21hmul, asm: "MULQ", clobberFlags: true, earlyOk: true},
   395  		{name: "HMULLU", argLength: 2, reg: gp21hmul, asm: "MULL", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   396  
   397  		// (arg0 + arg1) / 2 as unsigned, all 64 result bits
   398  		{name: "AVGQU", argLength: 2, reg: gp21, commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},
   399  
   400  		// DIVx[U] computes [arg0 / arg1, arg0 % arg1]
   401  		// For signed versions, AuxInt non-zero means that the divisor has been proved to be not -1.
   402  		{name: "DIVQ", argLength: 2, reg: gp11div, typ: "(Int64,Int64)", asm: "IDIVQ", aux: "Bool", clobberFlags: true},
   403  		{name: "DIVL", argLength: 2, reg: gp11div, typ: "(Int32,Int32)", asm: "IDIVL", aux: "Bool", clobberFlags: true, zeroUpperBits: 32},
   404  		{name: "DIVW", argLength: 2, reg: gp11div, typ: "(Int16,Int16)", asm: "IDIVW", aux: "Bool", clobberFlags: true},
   405  		{name: "DIVQU", argLength: 2, reg: gp11div, typ: "(UInt64,UInt64)", asm: "DIVQ", clobberFlags: true},
   406  		{name: "DIVLU", argLength: 2, reg: gp11div, typ: "(UInt32,UInt32)", asm: "DIVL", clobberFlags: true, zeroUpperBits: 32},
   407  		{name: "DIVWU", argLength: 2, reg: gp11div, typ: "(UInt16,UInt16)", asm: "DIVW", clobberFlags: true},
   408  
   409  		// computes -arg0, flags set for 0-arg0.
   410  		{name: "NEGLflags", argLength: 1, reg: gp11flags, typ: "(UInt32,Flags)", asm: "NEGL", resultInArg0: true, zeroUpperBits: 32},
   411  		// compute arg0+auxint. flags set for arg0+auxint.
   412  		// NOTE: we pretend the CF/OF flags are undefined for these instructions,
   413  		// so we can use INC/DEC instead of ADDQconst if auxint is +/-1. (INC/DEC don't modify CF.)
   414  		{name: "ADDQconstflags", argLength: 1, reg: gp11flags, aux: "Int32", asm: "ADDQ", resultInArg0: true},
   415  		{name: "ADDLconstflags", argLength: 1, reg: gp11flags, aux: "Int32", asm: "ADDL", resultInArg0: true, zeroUpperBits: 32},
   416  
   417  		// The following 4 add opcodes return the low 64 bits of the sum in the first result and
   418  		// the carry (the 65th bit) in the carry flag.
   419  		{name: "ADDQcarry", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "ADDQ", commutative: true, resultInArg0: true}, // r = arg0+arg1
   420  		{name: "ADCQ", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "ADCQ", commutative: true, resultInArg0: true}, // r = arg0+arg1+carry(arg2)
   421  		{name: "ADDQconstcarry", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "ADDQ", aux: "Int32", resultInArg0: true}, // r = arg0+auxint
   422  		{name: "ADCQconst", argLength: 2, reg: gp1flags1flags, typ: "(UInt64,Flags)", asm: "ADCQ", aux: "Int32", resultInArg0: true}, // r = arg0+auxint+carry(arg1)
   423  
   424  		// The following 4 add opcodes return the low 64 bits of the difference in the first result and
   425  		// the borrow (if the result is negative) in the carry flag.
   426  		{name: "SUBQborrow", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "SUBQ", resultInArg0: true},                    // r = arg0-arg1
   427  		{name: "SBBQ", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "SBBQ", resultInArg0: true},                     // r = arg0-(arg1+carry(arg2))
   428  		{name: "SUBQconstborrow", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "SUBQ", aux: "Int32", resultInArg0: true}, // r = arg0-auxint
   429  		{name: "SBBQconst", argLength: 2, reg: gp1flags1flags, typ: "(UInt64,Flags)", asm: "SBBQ", aux: "Int32", resultInArg0: true},  // r = arg0-(auxint+carry(arg1))
   430  
   431  		{name: "MULQU2", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{dx, ax}}, commutative: true, asm: "MULQ", clobberFlags: true, earlyOk: true}, // arg0 * arg1, returns (hi, lo)
   432  		// MULXQ is the BMI2 unsigned 64x64->128 multiply. arg0 must be in DX
   433  		// (the implicit operand); arg1 is any register or memory. Outputs are
   434  		// (hi, lo) and may be placed in any general-purpose registers (they
   435  		// must be different from each other; the assembler/register allocator
   436  		// arranges this). Unlike MULQ, MULXQ does not affect the flags, which
   437  		// makes it interleavable with ADCX/ADOX carry chains.
   438  		{name: "MULXQ", argLength: 2, reg: regInfo{inputs: []regMask{dx, gpsp}, outputs: []regMask{gp, gp}}, commutative: true, asm: "MULXQ"},      // arg0 * arg1, returns (hi, lo); does not affect flags
   439  		{name: "DIVQU2", argLength: 3, reg: regInfo{inputs: []regMask{dx, ax, gpsp}, outputs: []regMask{ax, dx}}, asm: "DIVQ", clobberFlags: true}, // arg0:arg1 / arg2 (128-bit divided by 64-bit), returns (q, r)
   440  
   441  		{name: "ANDQ", argLength: 2, reg: gp21, asm: "ANDQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & arg1
   442  		{name: "ANDL", argLength: 2, reg: gp21, asm: "ANDL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 & arg1
   443  		{name: "ANDQconst", argLength: 1, reg: gp11, asm: "ANDQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & auxint
   444  		{name: "ANDLconst", argLength: 1, reg: gp11, asm: "ANDL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 & auxint
   445  		{name: "ANDQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   446  		{name: "ANDLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   447  		{name: "ANDWconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   448  		{name: "ANDBconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   449  
   450  		{name: "ORQ", argLength: 2, reg: gp21, asm: "ORQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | arg1
   451  		{name: "ORL", argLength: 2, reg: gp21, asm: "ORL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 | arg1
   452  		{name: "ORQconst", argLength: 1, reg: gp11, asm: "ORQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | auxint
   453  		{name: "ORLconst", argLength: 1, reg: gp11, asm: "ORL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 | auxint
   454  		{name: "ORQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   455  		{name: "ORLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   456  		{name: "ORWconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   457  		{name: "ORBconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   458  
   459  		{name: "XORQ", argLength: 2, reg: gp21, asm: "XORQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ arg1
   460  		{name: "XORL", argLength: 2, reg: gp21, asm: "XORL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 ^ arg1
   461  		{name: "XORQconst", argLength: 1, reg: gp11, asm: "XORQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ auxint
   462  		{name: "XORLconst", argLength: 1, reg: gp11, asm: "XORL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 ^ auxint
   463  		{name: "XORQconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   464  		{name: "XORLconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   465  		{name: "XORWconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   466  		{name: "XORBconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   467  
   468  		// CMPx: compare arg0 to arg1.
   469  		{name: "CMPQ", argLength: 2, reg: gp2flags, asm: "CMPQ", typ: "Flags"},
   470  		{name: "CMPL", argLength: 2, reg: gp2flags, asm: "CMPL", typ: "Flags"},
   471  		{name: "CMPW", argLength: 2, reg: gp2flags, asm: "CMPW", typ: "Flags"},
   472  		{name: "CMPB", argLength: 2, reg: gp2flags, asm: "CMPB", typ: "Flags"},
   473  
   474  		// CMPxconst: compare arg0 to auxint.
   475  		{name: "CMPQconst", argLength: 1, reg: gp1flags, asm: "CMPQ", typ: "Flags", aux: "Int32"},
   476  		{name: "CMPLconst", argLength: 1, reg: gp1flags, asm: "CMPL", typ: "Flags", aux: "Int32"},
   477  		{name: "CMPWconst", argLength: 1, reg: gp1flags, asm: "CMPW", typ: "Flags", aux: "Int16"},
   478  		{name: "CMPBconst", argLength: 1, reg: gp1flags, asm: "CMPB", typ: "Flags", aux: "Int8"},
   479  
   480  		// CMPxload: compare *(arg0+auxint+aux) to arg1 (in that order). arg2=mem.
   481  		{name: "CMPQload", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   482  		{name: "CMPLload", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   483  		{name: "CMPWload", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   484  		{name: "CMPBload", argLength: 3, reg: gp1flagsLoad, asm: "CMPB", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   485  
   486  		// CMPxconstload: compare *(arg0+ValAndOff(AuxInt).Off()+aux) to ValAndOff(AuxInt).Val() (in that order). arg1=mem.
   487  		{name: "CMPQconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPQ", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   488  		{name: "CMPLconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPL", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   489  		{name: "CMPWconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPW", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   490  		{name: "CMPBconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPB", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   491  
   492  		// CMPxloadidx: compare *(arg0+N*arg1+auxint+aux) to arg2 (in that order). arg3=mem.
   493  		{name: "CMPQloadidx8", argLength: 4, reg: gp2flagsLoad, asm: "CMPQ", scale: 8, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   494  		{name: "CMPQloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPQ", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   495  		{name: "CMPLloadidx4", argLength: 4, reg: gp2flagsLoad, asm: "CMPL", scale: 4, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   496  		{name: "CMPLloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPL", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   497  		{name: "CMPWloadidx2", argLength: 4, reg: gp2flagsLoad, asm: "CMPW", scale: 2, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   498  		{name: "CMPWloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPW", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   499  		{name: "CMPBloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPB", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   500  
   501  		// CMPxconstloadidx: compare *(arg0+N*arg1+ValAndOff(AuxInt).Off()+aux) to ValAndOff(AuxInt).Val() (in that order). arg2=mem.
   502  		{name: "CMPQconstloadidx8", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", scale: 8, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   503  		{name: "CMPQconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   504  		{name: "CMPLconstloadidx4", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", scale: 4, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   505  		{name: "CMPLconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   506  		{name: "CMPWconstloadidx2", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", scale: 2, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   507  		{name: "CMPWconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   508  		{name: "CMPBconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPB", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   509  
   510  		// UCOMISx: floating-point compare arg0 to arg1
   511  		// x==S for float32, x==D for float64
   512  		{name: "UCOMISS", argLength: 2, reg: fp2flags, asm: "UCOMISS", typ: "Flags"},
   513  		{name: "UCOMISD", argLength: 2, reg: fp2flags, asm: "UCOMISD", typ: "Flags"},
   514  
   515  		// bit test/set/clear operations
   516  		{name: "BTL", argLength: 2, reg: gp2flags, asm: "BTL", typ: "Flags"},                                                           // test whether bit arg0%32 in arg1 is set
   517  		{name: "BTQ", argLength: 2, reg: gp2flags, asm: "BTQ", typ: "Flags"},                                                           // test whether bit arg0%64 in arg1 is set
   518  		{name: "BTCL", argLength: 2, reg: gp21, asm: "BTCL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32}, // complement bit arg1%32 in arg0
   519  		{name: "BTCQ", argLength: 2, reg: gp21, asm: "BTCQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                    // complement bit arg1%64 in arg0
   520  		{name: "BTRL", argLength: 2, reg: gp21, asm: "BTRL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32}, // reset bit arg1%32 in arg0
   521  		{name: "BTRQ", argLength: 2, reg: gp21, asm: "BTRQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                    // reset bit arg1%64 in arg0
   522  		{name: "BTSL", argLength: 2, reg: gp21, asm: "BTSL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32}, // set bit arg1%32 in arg0
   523  		{name: "BTSQ", argLength: 2, reg: gp21, asm: "BTSQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                    // set bit arg1%64 in arg0
   524  		{name: "BTLconst", argLength: 1, reg: gp1flags, asm: "BTL", typ: "Flags", aux: "Int8", earlyOk: true},                          // test whether bit auxint in arg0 is set, 0 <= auxint < 32
   525  		{name: "BTQconst", argLength: 1, reg: gp1flags, asm: "BTQ", typ: "Flags", aux: "Int8", earlyOk: true},                          // test whether bit auxint in arg0 is set, 0 <= auxint < 64
   526  		{name: "BTCQconst", argLength: 1, reg: gp11, asm: "BTCQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true},  // complement bit auxint in arg0, 31 <= auxint < 64
   527  		{name: "BTRQconst", argLength: 1, reg: gp11, asm: "BTRQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true},  // reset bit auxint in arg0, 31 <= auxint < 64
   528  		{name: "BTSQconst", argLength: 1, reg: gp11, asm: "BTSQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true},  // set bit auxint in arg0, 31 <= auxint < 64
   529  
   530  		// BT[SRC]Qconstmodify
   531  		//
   532  		//  S: set bit
   533  		//  R: reset (clear) bit
   534  		//  C: complement bit
   535  		//
   536  		// Apply operation to bit ValAndOff(AuxInt).Val() in the 64 bits at
   537  		// memory address arg0+ValAndOff(AuxInt).Off()+aux
   538  		// Bit index must be in range (31-63).
   539  		// (We use OR/AND/XOR for thinner targets and lower bit indexes.)
   540  		// arg1=mem, returns mem
   541  		//
   542  		// Note that there aren't non-const versions of these instructions.
   543  		// Well, there are such instructions, but they are slow and weird so we don't use them.
   544  		{name: "BTSQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTSQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   545  		{name: "BTRQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTRQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   546  		{name: "BTCQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTCQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   547  
   548  		// TESTx: compare (arg0 & arg1) to 0
   549  		{name: "TESTQ", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTQ", typ: "Flags"},
   550  		{name: "TESTL", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTL", typ: "Flags"},
   551  		{name: "TESTW", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTW", typ: "Flags"},
   552  		{name: "TESTB", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTB", typ: "Flags"},
   553  
   554  		// TESTxconst: compare (arg0 & auxint) to 0
   555  		{name: "TESTQconst", argLength: 1, reg: gp1flags, asm: "TESTQ", typ: "Flags", aux: "Int32"},
   556  		{name: "TESTLconst", argLength: 1, reg: gp1flags, asm: "TESTL", typ: "Flags", aux: "Int32"},
   557  		{name: "TESTWconst", argLength: 1, reg: gp1flags, asm: "TESTW", typ: "Flags", aux: "Int16"},
   558  		{name: "TESTBconst", argLength: 1, reg: gp1flags, asm: "TESTB", typ: "Flags", aux: "Int8"},
   559  
   560  		// S{HL, HR, AR}x: shift operations
   561  		// SHL: shift left
   562  		// SHR: shift right logical (0s are shifted in from beyond the word size)
   563  		// SAR: shift right arithmetic (sign bit is shifted in from beyond the word size)
   564  		// arg0 is the value being shifted
   565  		// arg1 is the amount to shift, interpreted mod (Q=64,L=32,W=32,B=32)
   566  		// (Note: x86 is weird, the 16 and 8 byte shifts still use all 5 bits of shift amount!)
   567  		// For *const versions, use auxint instead of arg1 as the shift amount. auxint must be in the range 0 to (Q=63,L=31,W=15,B=7) inclusive.
   568  		{name: "SHLQ", argLength: 2, reg: gp21shift, asm: "SHLQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   569  		{name: "SHLL", argLength: 2, reg: gp21shift, asm: "SHLL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   570  		{name: "SHLQconst", argLength: 1, reg: gp11, asm: "SHLQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   571  		{name: "SHLLconst", argLength: 1, reg: gp11, asm: "SHLL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   572  
   573  		{name: "SHRQ", argLength: 2, reg: gp21shift, asm: "SHRQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   574  		{name: "SHRL", argLength: 2, reg: gp21shift, asm: "SHRL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   575  		{name: "SHRW", argLength: 2, reg: gp21shift, asm: "SHRW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   576  		{name: "SHRB", argLength: 2, reg: gp21shift, asm: "SHRB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   577  		{name: "SHRQconst", argLength: 1, reg: gp11, asm: "SHRQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   578  		{name: "SHRLconst", argLength: 1, reg: gp11, asm: "SHRL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   579  		{name: "SHRWconst", argLength: 1, reg: gp11, asm: "SHRW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   580  		{name: "SHRBconst", argLength: 1, reg: gp11, asm: "SHRB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   581  
   582  		{name: "SARQ", argLength: 2, reg: gp21shift, asm: "SARQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   583  		{name: "SARL", argLength: 2, reg: gp21shift, asm: "SARL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   584  		{name: "SARW", argLength: 2, reg: gp21shift, asm: "SARW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   585  		{name: "SARB", argLength: 2, reg: gp21shift, asm: "SARB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   586  		{name: "SARQconst", argLength: 1, reg: gp11, asm: "SARQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   587  		{name: "SARLconst", argLength: 1, reg: gp11, asm: "SARL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   588  		{name: "SARWconst", argLength: 1, reg: gp11, asm: "SARW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   589  		{name: "SARBconst", argLength: 1, reg: gp11, asm: "SARB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   590  
   591  		// RO{L,R}x: rotate instructions
   592  		// computes arg0 rotate (L=left,R=right) arg1 bits.
   593  		// Bits are rotated within the low (Q=64,L=32,W=16,B=8) bits of the register.
   594  		// For *const versions use auxint instead of arg1 as the rotate amount. auxint must be in the range 0 to (Q=63,L=31,W=15,B=7) inclusive.
   595  		// x==L versions zero the upper 32 bits of the destination register.
   596  		// x==W and x==B versions leave the upper bits unspecified.
   597  		{name: "ROLQ", argLength: 2, reg: gp21shift, asm: "ROLQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   598  		{name: "ROLL", argLength: 2, reg: gp21shift, asm: "ROLL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   599  		{name: "ROLW", argLength: 2, reg: gp21shift, asm: "ROLW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   600  		{name: "ROLB", argLength: 2, reg: gp21shift, asm: "ROLB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   601  		{name: "RORQ", argLength: 2, reg: gp21shift, asm: "RORQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   602  		{name: "RORL", argLength: 2, reg: gp21shift, asm: "RORL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   603  		{name: "RORW", argLength: 2, reg: gp21shift, asm: "RORW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   604  		{name: "RORB", argLength: 2, reg: gp21shift, asm: "RORB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   605  		{name: "ROLQconst", argLength: 1, reg: gp11, asm: "ROLQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   606  		{name: "ROLLconst", argLength: 1, reg: gp11, asm: "ROLL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   607  		{name: "ROLWconst", argLength: 1, reg: gp11, asm: "ROLW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   608  		{name: "ROLBconst", argLength: 1, reg: gp11, asm: "ROLB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   609  
   610  		// [ADD,SUB,AND,OR]xload: integer load/op combo
   611  		// L = int32, Q = int64
   612  		// x==L operations zero the upper 4 bytes of the destination register.
   613  		// computes arg0 op *(arg1+auxint+aux), arg2=mem
   614  		{name: "ADDLload", argLength: 3, reg: gp21load, asm: "ADDL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   615  		{name: "ADDQload", argLength: 3, reg: gp21load, asm: "ADDQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   616  		{name: "SUBQload", argLength: 3, reg: gp21load, asm: "SUBQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   617  		{name: "SUBLload", argLength: 3, reg: gp21load, asm: "SUBL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   618  		{name: "ANDLload", argLength: 3, reg: gp21load, asm: "ANDL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   619  		{name: "ANDQload", argLength: 3, reg: gp21load, asm: "ANDQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   620  		{name: "ORQload", argLength: 3, reg: gp21load, asm: "ORQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   621  		{name: "ORLload", argLength: 3, reg: gp21load, asm: "ORL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   622  		{name: "XORQload", argLength: 3, reg: gp21load, asm: "XORQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   623  		{name: "XORLload", argLength: 3, reg: gp21load, asm: "XORL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   624  
   625  		// integer indexed load/op combo
   626  		// L = int32, Q = int64
   627  		// L operations zero the upper 4 bytes of the destination register.
   628  		// computes arg0 op *(arg1+scale*arg2+auxint+aux), arg3=mem
   629  		{name: "ADDLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   630  		{name: "ADDLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   631  		{name: "ADDLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   632  		{name: "ADDQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ADDQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   633  		{name: "ADDQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ADDQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   634  		{name: "SUBLloadidx1", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   635  		{name: "SUBLloadidx4", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   636  		{name: "SUBLloadidx8", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   637  		{name: "SUBQloadidx1", argLength: 4, reg: gp21loadidx, asm: "SUBQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   638  		{name: "SUBQloadidx8", argLength: 4, reg: gp21loadidx, asm: "SUBQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   639  		{name: "ANDLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   640  		{name: "ANDLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   641  		{name: "ANDLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   642  		{name: "ANDQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ANDQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   643  		{name: "ANDQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ANDQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   644  		{name: "ORLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   645  		{name: "ORLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   646  		{name: "ORLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   647  		{name: "ORQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ORQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   648  		{name: "ORQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ORQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   649  		{name: "XORLloadidx1", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   650  		{name: "XORLloadidx4", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   651  		{name: "XORLloadidx8", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   652  		{name: "XORQloadidx1", argLength: 4, reg: gp21loadidx, asm: "XORQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   653  		{name: "XORQloadidx8", argLength: 4, reg: gp21loadidx, asm: "XORQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   654  
   655  		// direct binary op on memory (read-modify-write)
   656  		// L = int32, Q = int64
   657  		// does *(arg0+auxint+aux) op= arg1, arg2=mem
   658  		{name: "ADDQmodify", argLength: 3, reg: gpstore, asm: "ADDQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   659  		{name: "SUBQmodify", argLength: 3, reg: gpstore, asm: "SUBQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   660  		{name: "ANDQmodify", argLength: 3, reg: gpstore, asm: "ANDQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   661  		{name: "ORQmodify", argLength: 3, reg: gpstore, asm: "ORQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   662  		{name: "XORQmodify", argLength: 3, reg: gpstore, asm: "XORQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   663  		{name: "ADDLmodify", argLength: 3, reg: gpstore, asm: "ADDL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   664  		{name: "SUBLmodify", argLength: 3, reg: gpstore, asm: "SUBL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   665  		{name: "ANDLmodify", argLength: 3, reg: gpstore, asm: "ANDL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   666  		{name: "ORLmodify", argLength: 3, reg: gpstore, asm: "ORL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   667  		{name: "XORLmodify", argLength: 3, reg: gpstore, asm: "XORL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   668  
   669  		// indexed direct binary op on memory.
   670  		// does *(arg0+scale*arg1+auxint+aux) op= arg2, arg3=mem
   671  		{name: "ADDQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ADDQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   672  		{name: "ADDQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ADDQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   673  		{name: "SUBQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "SUBQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   674  		{name: "SUBQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "SUBQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   675  		{name: "ANDQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ANDQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   676  		{name: "ANDQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ANDQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   677  		{name: "ORQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ORQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   678  		{name: "ORQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ORQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   679  		{name: "XORQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "XORQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   680  		{name: "XORQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "XORQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   681  		{name: "ADDLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   682  		{name: "ADDLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   683  		{name: "ADDLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   684  		{name: "SUBLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   685  		{name: "SUBLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   686  		{name: "SUBLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   687  		{name: "ANDLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   688  		{name: "ANDLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   689  		{name: "ANDLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   690  		{name: "ORLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   691  		{name: "ORLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   692  		{name: "ORLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   693  		{name: "XORLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   694  		{name: "XORLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   695  		{name: "XORLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   696  
   697  		// indexed direct binary op on memory with constant argument.
   698  		// does *(arg0+scale*arg1+ValAndOff(AuxInt).Off()+aux) op= ValAndOff(AuxInt).Val(), arg2=mem
   699  		{name: "ADDQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   700  		{name: "ADDQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ADDQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   701  		{name: "ANDQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   702  		{name: "ANDQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ANDQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   703  		{name: "ORQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   704  		{name: "ORQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ORQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   705  		{name: "XORQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   706  		{name: "XORQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "XORQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   707  		{name: "ADDLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   708  		{name: "ADDLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   709  		{name: "ADDLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   710  		{name: "ADDWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   711  		{name: "ADDWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "ADDW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   712  		{name: "ADDBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   713  		{name: "ANDLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   714  		{name: "ANDLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   715  		{name: "ANDLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   716  		{name: "ANDWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   717  		{name: "ANDWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "ANDW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   718  		{name: "ANDBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   719  		{name: "ORLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   720  		{name: "ORLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   721  		{name: "ORLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   722  		{name: "ORWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   723  		{name: "ORWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "ORW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   724  		{name: "ORBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   725  		{name: "XORLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   726  		{name: "XORLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   727  		{name: "XORLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   728  		{name: "XORWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   729  		{name: "XORWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "XORW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   730  		{name: "XORBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   731  
   732  		// {NEG,NOT}x: unary ops
   733  		// computes [NEG:-,NOT:^]arg0
   734  		// L = int32, Q = int64
   735  		// L operations zero the upper 4 bytes of the destination register.
   736  		{name: "NEGQ", argLength: 1, reg: gp11, asm: "NEGQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   737  		{name: "NEGL", argLength: 1, reg: gp11, asm: "NEGL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   738  		{name: "NOTQ", argLength: 1, reg: gp11, asm: "NOTQ", resultInArg0: true, earlyOk: true},
   739  		{name: "NOTL", argLength: 1, reg: gp11, asm: "NOTL", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   740  
   741  		// BS{F,R}Q returns a tuple [result, flags]
   742  		// result is undefined if the input is zero.
   743  		// flags are set to "equal" if the input is zero, "not equal" otherwise.
   744  		// BS{F,R}L returns only the result.
   745  		{name: "BSFQ", argLength: 1, reg: gp11flags, asm: "BSFQ", typ: "(UInt64,Flags)", earlyOk: true},        // # of low-order zeroes in 64-bit arg
   746  		{name: "BSFL", argLength: 1, reg: gp11, asm: "BSFL", typ: "UInt32", clobberFlags: true, earlyOk: true}, // # of low-order zeroes in 32-bit arg
   747  		{name: "BSRQ", argLength: 1, reg: gp11flags, asm: "BSRQ", typ: "(UInt64,Flags)", earlyOk: true},        // # of high-order zeroes in 64-bit arg
   748  		{name: "BSRL", argLength: 1, reg: gp11, asm: "BSRL", typ: "UInt32", clobberFlags: true, earlyOk: true}, // # of high-order zeroes in 32-bit arg
   749  
   750  		// CMOV instructions: 64, 32 and 16-bit sizes.
   751  		// if arg2 encodes a true result, return arg1, else arg0
   752  		{name: "CMOVQEQ", argLength: 3, reg: gp21, asm: "CMOVQEQ", resultInArg0: true, earlyOk: true},
   753  		{name: "CMOVQNE", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, earlyOk: true},
   754  		{name: "CMOVQLT", argLength: 3, reg: gp21, asm: "CMOVQLT", resultInArg0: true, earlyOk: true},
   755  		{name: "CMOVQGT", argLength: 3, reg: gp21, asm: "CMOVQGT", resultInArg0: true, earlyOk: true},
   756  		{name: "CMOVQLE", argLength: 3, reg: gp21, asm: "CMOVQLE", resultInArg0: true, earlyOk: true},
   757  		{name: "CMOVQGE", argLength: 3, reg: gp21, asm: "CMOVQGE", resultInArg0: true, earlyOk: true},
   758  		{name: "CMOVQLS", argLength: 3, reg: gp21, asm: "CMOVQLS", resultInArg0: true, earlyOk: true},
   759  		{name: "CMOVQHI", argLength: 3, reg: gp21, asm: "CMOVQHI", resultInArg0: true, earlyOk: true},
   760  		{name: "CMOVQCC", argLength: 3, reg: gp21, asm: "CMOVQCC", resultInArg0: true, earlyOk: true},
   761  		{name: "CMOVQCS", argLength: 3, reg: gp21, asm: "CMOVQCS", resultInArg0: true, earlyOk: true},
   762  
   763  		{name: "CMOVLEQ", argLength: 3, reg: gp21, asm: "CMOVLEQ", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   764  		{name: "CMOVLNE", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   765  		{name: "CMOVLLT", argLength: 3, reg: gp21, asm: "CMOVLLT", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   766  		{name: "CMOVLGT", argLength: 3, reg: gp21, asm: "CMOVLGT", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   767  		{name: "CMOVLLE", argLength: 3, reg: gp21, asm: "CMOVLLE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   768  		{name: "CMOVLGE", argLength: 3, reg: gp21, asm: "CMOVLGE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   769  		{name: "CMOVLLS", argLength: 3, reg: gp21, asm: "CMOVLLS", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   770  		{name: "CMOVLHI", argLength: 3, reg: gp21, asm: "CMOVLHI", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   771  		{name: "CMOVLCC", argLength: 3, reg: gp21, asm: "CMOVLCC", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   772  		{name: "CMOVLCS", argLength: 3, reg: gp21, asm: "CMOVLCS", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   773  
   774  		{name: "CMOVWEQ", argLength: 3, reg: gp21, asm: "CMOVWEQ", resultInArg0: true, earlyOk: true},
   775  		{name: "CMOVWNE", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, earlyOk: true},
   776  		{name: "CMOVWLT", argLength: 3, reg: gp21, asm: "CMOVWLT", resultInArg0: true, earlyOk: true},
   777  		{name: "CMOVWGT", argLength: 3, reg: gp21, asm: "CMOVWGT", resultInArg0: true, earlyOk: true},
   778  		{name: "CMOVWLE", argLength: 3, reg: gp21, asm: "CMOVWLE", resultInArg0: true, earlyOk: true},
   779  		{name: "CMOVWGE", argLength: 3, reg: gp21, asm: "CMOVWGE", resultInArg0: true, earlyOk: true},
   780  		{name: "CMOVWLS", argLength: 3, reg: gp21, asm: "CMOVWLS", resultInArg0: true, earlyOk: true},
   781  		{name: "CMOVWHI", argLength: 3, reg: gp21, asm: "CMOVWHI", resultInArg0: true, earlyOk: true},
   782  		{name: "CMOVWCC", argLength: 3, reg: gp21, asm: "CMOVWCC", resultInArg0: true, earlyOk: true},
   783  		{name: "CMOVWCS", argLength: 3, reg: gp21, asm: "CMOVWCS", resultInArg0: true, earlyOk: true},
   784  
   785  		// CMOV with floating point instructions. We need separate pseudo-op to handle
   786  		// InvertFlags correctly, and to generate special code that handles NaN (unordered flag).
   787  		// NOTE: the fact that CMOV*EQF here is marked to generate CMOV*NE is not a bug. See
   788  		// code generation in amd64/ssa.go.
   789  		{name: "CMOVQEQF", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, needIntTemp: true, earlyOk: true},
   790  		{name: "CMOVQNEF", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, earlyOk: true},
   791  		{name: "CMOVQGTF", argLength: 3, reg: gp21, asm: "CMOVQHI", resultInArg0: true, earlyOk: true},
   792  		{name: "CMOVQGEF", argLength: 3, reg: gp21, asm: "CMOVQCC", resultInArg0: true, earlyOk: true},
   793  		{name: "CMOVLEQF", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, needIntTemp: true, earlyOk: true, zeroUpperBits: 32},
   794  		{name: "CMOVLNEF", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   795  		{name: "CMOVLGTF", argLength: 3, reg: gp21, asm: "CMOVLHI", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   796  		{name: "CMOVLGEF", argLength: 3, reg: gp21, asm: "CMOVLCC", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   797  		{name: "CMOVWEQF", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, needIntTemp: true, earlyOk: true},
   798  		{name: "CMOVWNEF", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, earlyOk: true},
   799  		{name: "CMOVWGTF", argLength: 3, reg: gp21, asm: "CMOVWHI", resultInArg0: true, earlyOk: true},
   800  		{name: "CMOVWGEF", argLength: 3, reg: gp21, asm: "CMOVWCC", resultInArg0: true, earlyOk: true},
   801  
   802  		// BSWAPx swaps the low-order (L=4,Q=8) bytes of arg0.
   803  		// Q: abcdefgh -> hgfedcba
   804  		// L: abcdefgh -> 0000hgfe (L zeros the upper 4 bytes)
   805  		{name: "BSWAPQ", argLength: 1, reg: gp11, asm: "BSWAPQ", resultInArg0: true, earlyOk: true},
   806  		{name: "BSWAPL", argLength: 1, reg: gp11, asm: "BSWAPL", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   807  
   808  		// POPCNTx counts the number of set bits in the low-order (L=32,Q=64) bits of arg0.
   809  		// POPCNTx instructions are only guaranteed to be available if GOAMD64>=v2.
   810  		// For GOAMD64<v2, any use must be preceded by a successful runtime check of runtime.x86HasPOPCNT.
   811  		{name: "POPCNTQ", argLength: 1, reg: gp11, asm: "POPCNTQ", clobberFlags: true, zeroUpperBits: 56},
   812  		{name: "POPCNTL", argLength: 1, reg: gp11, asm: "POPCNTL", clobberFlags: true, zeroUpperBits: 56},
   813  
   814  		// SQRTSx computes sqrt(arg0)
   815  		// S = float32, D = float64
   816  		{name: "SQRTSD", argLength: 1, reg: fp11, asm: "SQRTSD", earlyOk: true},
   817  		{name: "SQRTSS", argLength: 1, reg: fp11, asm: "SQRTSS", earlyOk: true},
   818  
   819  		// ROUNDSD rounds arg0 to an integer depending on auxint
   820  		// 0 means math.RoundToEven, 1 means math.Floor, 2 math.Ceil, 3 math.Trunc
   821  		// (The result is still a float64.)
   822  		// ROUNDSD instruction is only guaraneteed to be available if GOAMD64>=v2.
   823  		// For GOAMD64<v2, any use must be preceded by a successful check of runtime.x86HasSSE41.
   824  		{name: "ROUNDSD", argLength: 1, reg: fp11, aux: "Int8", asm: "ROUNDSD"},
   825  		{name: "ROUNDSS", argLength: 1, reg: fp11, aux: "Int8", asm: "ROUNDSS"},
   826  		// See why we need those in issue #71204
   827  		{name: "LoweredRound32F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   828  		{name: "LoweredRound64F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   829  
   830  		// VFMADD231Sx only exist on platforms with the FMA3 instruction set.
   831  		// Any use must be preceded by a successful check of runtime.x86HasFMA or a check of GOAMD64>=v3.
   832  		// x==S for float32, x==D for float64
   833  		// arg0 + arg1*arg2, with no intermediate rounding.
   834  		{name: "VFMADD231SS", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMADD231SS"},
   835  		{name: "VFMADD231SD", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMADD231SD"},
   836  		// arg1*arg2 - arg0, with no intermediate rounding.
   837  		{name: "VFMSUB231SS", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMSUB231SS"},
   838  		{name: "VFMSUB231SD", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMSUB231SD"},
   839  		// arg0 - arg1*arg2, with no intermediate rounding.
   840  		{name: "VFNMADD231SS", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFNMADD231SS"},
   841  		{name: "VFNMADD231SD", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFNMADD231SD"},
   842  
   843  		// Note that these operations don't exactly match the semantics of Go's
   844  		// builtin min/max. In particular, these aren't commutative, because on
   845  		// various special cases the 2nd argument is preferred.
   846  		{name: "MINSD", argLength: 2, reg: fp21, resultInArg0: true, asm: "MINSD", earlyOk: true}, // min(arg0,arg1)
   847  		{name: "MINSS", argLength: 2, reg: fp21, resultInArg0: true, asm: "MINSS", earlyOk: true}, // min(arg0,arg1)
   848  		{name: "MAXSD", argLength: 2, reg: fp21, resultInArg0: true, asm: "MAXSD", earlyOk: true}, // max(arg0,arg1)
   849  		{name: "MAXSS", argLength: 2, reg: fp21, resultInArg0: true, asm: "MAXSS", earlyOk: true}, // max(arg0,arg1)
   850  
   851  		{name: "SBBQcarrymask", argLength: 1, reg: flagsgp, asm: "SBBQ", earlyOk: true},                    // (int64)(-1) if carry is set, 0 if carry is clear.
   852  		{name: "SBBLcarrymask", argLength: 1, reg: flagsgp, asm: "SBBL", earlyOk: true, zeroUpperBits: 32}, // (int32)(-1) if carry is set, 0 if carry is clear.
   853  		// Note: SBBW and SBBB are subsumed by SBBL
   854  
   855  		{name: "SETEQ", argLength: 1, reg: readflags, asm: "SETEQ", earlyOk: true}, // extract == condition from arg0
   856  		{name: "SETNE", argLength: 1, reg: readflags, asm: "SETNE", earlyOk: true}, // extract != condition from arg0
   857  		{name: "SETL", argLength: 1, reg: readflags, asm: "SETLT", earlyOk: true},  // extract signed < condition from arg0
   858  		{name: "SETLE", argLength: 1, reg: readflags, asm: "SETLE", earlyOk: true}, // extract signed <= condition from arg0
   859  		{name: "SETG", argLength: 1, reg: readflags, asm: "SETGT", earlyOk: true},  // extract signed > condition from arg0
   860  		{name: "SETGE", argLength: 1, reg: readflags, asm: "SETGE", earlyOk: true}, // extract signed >= condition from arg0
   861  		{name: "SETB", argLength: 1, reg: readflags, asm: "SETCS", earlyOk: true},  // extract unsigned < condition from arg0
   862  		{name: "SETBE", argLength: 1, reg: readflags, asm: "SETLS", earlyOk: true}, // extract unsigned <= condition from arg0
   863  		{name: "SETA", argLength: 1, reg: readflags, asm: "SETHI", earlyOk: true},  // extract unsigned > condition from arg0
   864  		{name: "SETAE", argLength: 1, reg: readflags, asm: "SETCC", earlyOk: true}, // extract unsigned >= condition from arg0
   865  		{name: "SETO", argLength: 1, reg: readflags, asm: "SETOS", earlyOk: true},  // extract if overflow flag is set from arg0
   866  		// Variants that store result to memory
   867  		{name: "SETEQstore", argLength: 3, reg: gpstoreconst, asm: "SETEQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract == condition from arg1 to arg0+auxint+aux, arg2=mem
   868  		{name: "SETNEstore", argLength: 3, reg: gpstoreconst, asm: "SETNE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract != condition from arg1 to arg0+auxint+aux, arg2=mem
   869  		{name: "SETLstore", argLength: 3, reg: gpstoreconst, asm: "SETLT", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract signed < condition from arg1 to arg0+auxint+aux, arg2=mem
   870  		{name: "SETLEstore", argLength: 3, reg: gpstoreconst, asm: "SETLE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract signed <= condition from arg1 to arg0+auxint+aux, arg2=mem
   871  		{name: "SETGstore", argLength: 3, reg: gpstoreconst, asm: "SETGT", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract signed > condition from arg1 to arg0+auxint+aux, arg2=mem
   872  		{name: "SETGEstore", argLength: 3, reg: gpstoreconst, asm: "SETGE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract signed >= condition from arg1 to arg0+auxint+aux, arg2=mem
   873  		{name: "SETBstore", argLength: 3, reg: gpstoreconst, asm: "SETCS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract unsigned < condition from arg1 to arg0+auxint+aux, arg2=mem
   874  		{name: "SETBEstore", argLength: 3, reg: gpstoreconst, asm: "SETLS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract unsigned <= condition from arg1 to arg0+auxint+aux, arg2=mem
   875  		{name: "SETAstore", argLength: 3, reg: gpstoreconst, asm: "SETHI", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract unsigned > condition from arg1 to arg0+auxint+aux, arg2=mem
   876  		{name: "SETAEstore", argLength: 3, reg: gpstoreconst, asm: "SETCC", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract unsigned >= condition from arg1 to arg0+auxint+aux, arg2=mem
   877  		{name: "SETEQstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETEQ", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract == condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   878  		{name: "SETNEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETNE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract != condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   879  		{name: "SETLstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLT", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract signed < condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   880  		{name: "SETLEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract signed <= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   881  		{name: "SETGstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETGT", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract signed > condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   882  		{name: "SETGEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETGE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract signed >= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   883  		{name: "SETBstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETCS", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract unsigned < condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   884  		{name: "SETBEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLS", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract unsigned <= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   885  		{name: "SETAstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETHI", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract unsigned > condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   886  		{name: "SETAEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETCC", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract unsigned >= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   887  
   888  		// Need different opcodes for floating point conditions because
   889  		// any comparison involving a NaN is always FALSE and thus
   890  		// the patterns for inverting conditions cannot be used.
   891  		{name: "SETEQF", argLength: 1, reg: flagsgp, asm: "SETEQ", clobberFlags: true, needIntTemp: true, earlyOk: true}, // extract == condition from arg0
   892  		{name: "SETNEF", argLength: 1, reg: flagsgp, asm: "SETNE", clobberFlags: true, needIntTemp: true, earlyOk: true}, // extract != condition from arg0
   893  		{name: "SETORD", argLength: 1, reg: flagsgp, asm: "SETPC", earlyOk: true},                                        // extract "ordered" (No Nan present) condition from arg0
   894  		{name: "SETNAN", argLength: 1, reg: flagsgp, asm: "SETPS", earlyOk: true},                                        // extract "unordered" (Nan present) condition from arg0
   895  
   896  		{name: "SETGF", argLength: 1, reg: flagsgp, asm: "SETHI", earlyOk: true},  // extract floating > condition from arg0
   897  		{name: "SETGEF", argLength: 1, reg: flagsgp, asm: "SETCC", earlyOk: true}, // extract floating >= condition from arg0
   898  
   899  		{name: "MOVBQSX", argLength: 1, reg: gp11, asm: "MOVBQSX", earlyOk: true},                    // sign extend arg0 from int8 to int64
   900  		{name: "MOVBQZX", argLength: 1, reg: gp11, asm: "MOVBLZX", earlyOk: true, zeroUpperBits: 56}, // zero extend arg0 from int8 to int64
   901  		{name: "MOVWQSX", argLength: 1, reg: gp11, asm: "MOVWQSX", earlyOk: true},                    // sign extend arg0 from int16 to int64
   902  		{name: "MOVWQZX", argLength: 1, reg: gp11, asm: "MOVWLZX", earlyOk: true, zeroUpperBits: 48}, // zero extend arg0 from int16 to int64
   903  		{name: "MOVLQSX", argLength: 1, reg: gp11, asm: "MOVLQSX", earlyOk: true},                    // sign extend arg0 from int32 to int64
   904  		{name: "MOVLQZX", argLength: 1, reg: gp11, asm: "MOVL", earlyOk: true, zeroUpperBits: 32},    // zero extend arg0 from int32 to int64
   905  
   906  		{name: "MOVLconst", reg: gp01, asm: "MOVL", typ: "UInt32", aux: "Int32", rematerializeable: true, earlyOk: true, zeroUpperBits: 32}, // 32 low bits of auxint (upper 32 are zeroed)
   907  		{name: "MOVQconst", reg: gp01, asm: "MOVQ", typ: "UInt64", aux: "Int64", rematerializeable: true, earlyOk: true},                    // auxint
   908  
   909  		{name: "CVTTSD2SL", argLength: 1, reg: fpgp, asm: "CVTTSD2SL", earlyOk: true, zeroUpperBits: 32}, // convert float64 to int32
   910  		{name: "CVTTSD2SQ", argLength: 1, reg: fpgp, asm: "CVTTSD2SQ", earlyOk: true},                    // convert float64 to int64
   911  		{name: "CVTTSS2SL", argLength: 1, reg: fpgp, asm: "CVTTSS2SL", earlyOk: true, zeroUpperBits: 32}, // convert float32 to int32
   912  		{name: "CVTTSS2SQ", argLength: 1, reg: fpgp, asm: "CVTTSS2SQ", earlyOk: true},                    // convert float32 to int64
   913  		{name: "CVTSL2SS", argLength: 1, reg: gpfp, asm: "CVTSL2SS", earlyOk: true},                      // convert int32 to float32
   914  		{name: "CVTSL2SD", argLength: 1, reg: gpfp, asm: "CVTSL2SD", earlyOk: true},                      // convert int32 to float64
   915  		{name: "CVTSQ2SS", argLength: 1, reg: gpfp, asm: "CVTSQ2SS", earlyOk: true},                      // convert int64 to float32
   916  		{name: "CVTSQ2SD", argLength: 1, reg: gpfp, asm: "CVTSQ2SD", earlyOk: true},                      // convert int64 to float64
   917  		{name: "CVTSD2SS", argLength: 1, reg: fp11, asm: "CVTSD2SS", earlyOk: true},                      // convert float64 to float32
   918  		{name: "CVTSS2SD", argLength: 1, reg: fp11, asm: "CVTSS2SD", earlyOk: true},                      // convert float32 to float64
   919  
   920  		// Move values between int and float registers, with no conversion.
   921  		// TODO: should we have generic versions of these?
   922  		{name: "MOVQi2f", argLength: 1, reg: gpfp, typ: "Float64", earlyOk: true},                   // move 64 bits from int to float reg
   923  		{name: "MOVQf2i", argLength: 1, reg: fpgp, typ: "UInt64", earlyOk: true},                    // move 64 bits from float to int reg
   924  		{name: "MOVLi2f", argLength: 1, reg: gpfp, typ: "Float32", earlyOk: true},                   // move 32 bits from int to float reg
   925  		{name: "MOVLf2i", argLength: 1, reg: fpgp, typ: "UInt32", earlyOk: true, zeroUpperBits: 32}, // move 32 bits from float to int reg, zero extend
   926  
   927  		{name: "PXOR", argLength: 2, reg: fp21, asm: "PXOR", commutative: true, resultInArg0: true, earlyOk: true}, // exclusive or, applied to X regs (for float negation).
   928  		{name: "POR", argLength: 2, reg: fp21, asm: "POR", commutative: true, resultInArg0: true, earlyOk: true},   // inclusive or, applied to X regs (for float min/max).
   929  
   930  		{name: "LEAQ", argLength: 1, reg: gp11sb, asm: "LEAQ", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true},                    // arg0 + auxint + offset encoded in aux
   931  		{name: "LEAL", argLength: 1, reg: gp11sb, asm: "LEAL", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true, zeroUpperBits: 32}, // arg0 + auxint + offset encoded in aux
   932  		{name: "LEAW", argLength: 1, reg: gp11sb, asm: "LEAW", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true},                    // arg0 + auxint + offset encoded in aux
   933  
   934  		// LEAxn computes arg0 + n*arg1 + auxint + aux
   935  		// x==L zeroes the upper 4 bytes.
   936  		{name: "LEAQ1", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + arg1 + auxint + aux
   937  		{name: "LEAL1", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32}, // arg0 + arg1 + auxint + aux
   938  		{name: "LEAW1", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + arg1 + auxint + aux
   939  		{name: "LEAQ2", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 2*arg1 + auxint + aux
   940  		{name: "LEAL2", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32},                    // arg0 + 2*arg1 + auxint + aux
   941  		{name: "LEAW2", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 2*arg1 + auxint + aux
   942  		{name: "LEAQ4", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 4*arg1 + auxint + aux
   943  		{name: "LEAL4", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32},                    // arg0 + 4*arg1 + auxint + aux
   944  		{name: "LEAW4", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 4*arg1 + auxint + aux
   945  		{name: "LEAQ8", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 8*arg1 + auxint + aux
   946  		{name: "LEAL8", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32},                    // arg0 + 8*arg1 + auxint + aux
   947  		{name: "LEAW8", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 8*arg1 + auxint + aux
   948  		// Note: LEAx{1,2,4,8} must not have OpSB as either argument.
   949  
   950  		// MOVxload: loads
   951  		// Load (Q=8,L=4,W=2,B=1) bytes from (arg0+auxint+aux), arg1=mem.
   952  		// "+auxint+aux" == add auxint and the offset of the symbol in aux (if any) to the effective address
   953  		// Standard versions zero extend the result. SX versions sign extend the result.
   954  		{name: "MOVBload", argLength: 2, reg: gpload, asm: "MOVBLZX", aux: "SymOff", typ: "UInt8", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 56},
   955  		{name: "MOVBQSXload", argLength: 2, reg: gpload, asm: "MOVBQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   956  		{name: "MOVWload", argLength: 2, reg: gpload, asm: "MOVWLZX", aux: "SymOff", typ: "UInt16", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 48},
   957  		{name: "MOVWQSXload", argLength: 2, reg: gpload, asm: "MOVWQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   958  		{name: "MOVLload", argLength: 2, reg: gpload, asm: "MOVL", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},
   959  		{name: "MOVLQSXload", argLength: 2, reg: gpload, asm: "MOVLQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   960  		{name: "MOVQload", argLength: 2, reg: gpload, asm: "MOVQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   961  
   962  		// MOVxstore: stores
   963  		// Store (Q=8,L=4,W=2,B=1) low bytes of arg1.
   964  		// Does *(arg0+auxint+aux) = arg1, arg2=mem.
   965  		{name: "MOVBstore", argLength: 3, reg: gpstore, asm: "MOVB", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   966  		{name: "MOVWstore", argLength: 3, reg: gpstore, asm: "MOVW", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   967  		{name: "MOVLstore", argLength: 3, reg: gpstore, asm: "MOVL", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   968  		{name: "MOVQstore", argLength: 3, reg: gpstore, asm: "MOVQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   969  
   970  		// MOVOload/store: 16 byte load/store
   971  		// These operations are only used to move data around: there is no *O arithmetic, for example.
   972  		{name: "MOVOload", argLength: 2, reg: fpload, asm: "MOVUPS", aux: "SymOff", typ: "Int128", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // load 16 bytes from arg0+auxint+aux. arg1=mem
   973  		{name: "MOVOstore", argLength: 3, reg: fpstore, asm: "MOVUPS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 16 bytes in arg1 to arg0+auxint+aux. arg2=mem
   974  
   975  		// MOVxloadidx: indexed loads
   976  		// load (Q=8,L=4,W=2,B=1) bytes from (arg0+scale*arg1+auxint+aux), arg2=mem.
   977  		// Results are zero-extended. (TODO: sign-extending indexed loads)
   978  		{name: "MOVBloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBLZX", scale: 1, aux: "SymOff", typ: "UInt8", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 56},
   979  		{name: "MOVWloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVWLZX", scale: 1, aux: "SymOff", typ: "UInt16", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 48},
   980  		{name: "MOVWloadidx2", argLength: 3, reg: gploadidx, asm: "MOVWLZX", scale: 2, aux: "SymOff", typ: "UInt16", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 48},
   981  		{name: "MOVLloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 32},
   982  		{name: "MOVLloadidx4", argLength: 3, reg: gploadidx, asm: "MOVL", scale: 4, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},
   983  		{name: "MOVLloadidx8", argLength: 3, reg: gploadidx, asm: "MOVL", scale: 8, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},
   984  		{name: "MOVQloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   985  		{name: "MOVQloadidx8", argLength: 3, reg: gploadidx, asm: "MOVQ", scale: 8, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true},
   986  
   987  		// MOVxstoreidx: indexed stores
   988  		// Store (Q=8,L=4,W=2,B=1) low bytes of arg2.
   989  		// Does *(arg0+scale*arg1+auxint+aux) = arg2, arg3=mem.
   990  		{name: "MOVBstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVB", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   991  		{name: "MOVWstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVW", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   992  		{name: "MOVWstoreidx2", argLength: 4, reg: gpstoreidx, asm: "MOVW", scale: 2, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   993  		{name: "MOVLstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   994  		{name: "MOVLstoreidx4", argLength: 4, reg: gpstoreidx, asm: "MOVL", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   995  		{name: "MOVLstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVL", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   996  		{name: "MOVQstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   997  		{name: "MOVQstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVQ", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   998  
   999  		// TODO: add size-mismatched indexed loads/stores, like MOVBstoreidx4?
  1000  
  1001  		// MOVxstoreconst: constant stores
  1002  		// Store (O=16,Q=8,L=4,W=2,B=1) constant bytes.
  1003  		// Does *(arg0+ValAndOff(AuxInt).Off()+aux) = ValAndOff(AuxInt).Val(), arg1=mem.
  1004  		// O version can only store the constant 0.
  1005  		{name: "MOVBstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVB", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1006  		{name: "MOVWstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVW", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1007  		{name: "MOVLstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVL", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1008  		{name: "MOVQstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVQ", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1009  		{name: "MOVOstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVUPS", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1010  
  1011  		// MOVxstoreconstidx: constant indexed stores
  1012  		// Store (Q=8,L=4,W=2,B=1) constant bytes.
  1013  		// Does *(arg0+scale*arg1+ValAndOff(AuxInt).Off()+aux) = ValAndOff(AuxInt).Val(), arg2=mem.
  1014  		{name: "MOVBstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVB", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1015  		{name: "MOVWstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVW", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1016  		{name: "MOVWstoreconstidx2", argLength: 3, reg: gpstoreconstidx, asm: "MOVW", scale: 2, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
  1017  		{name: "MOVLstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1018  		{name: "MOVLstoreconstidx4", argLength: 3, reg: gpstoreconstidx, asm: "MOVL", scale: 4, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
  1019  		{name: "MOVQstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1020  		{name: "MOVQstoreconstidx8", argLength: 3, reg: gpstoreconstidx, asm: "MOVQ", scale: 8, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
  1021  
  1022  		// arg0 = pointer to start of memory to zero
  1023  		// arg1 = mem
  1024  		// auxint = # of bytes to zero
  1025  		// returns mem
  1026  		{
  1027  			name:      "LoweredZero",
  1028  			aux:       "Int64",
  1029  			argLength: 2,
  1030  			reg: regInfo{
  1031  				inputs: []regMask{gp},
  1032  			},
  1033  			faultOnNilArg0: true,
  1034  			addrSinkArg0:   true,
  1035  		},
  1036  
  1037  		// arg0 = pointer to start of memory to zero
  1038  		// arg1 = mem
  1039  		// auxint = # of bytes to zero
  1040  		// returns mem
  1041  		{
  1042  			name:      "LoweredZeroLoop",
  1043  			aux:       "Int64",
  1044  			argLength: 2,
  1045  			reg: regInfo{
  1046  				inputs:       []regMask{gp},
  1047  				clobbersArg0: true,
  1048  			},
  1049  			clobberFlags:   true,
  1050  			faultOnNilArg0: true,
  1051  			addrSinkArg0:   true,
  1052  			needIntTemp:    true,
  1053  		},
  1054  
  1055  		// arg0 = address of memory to zero
  1056  		// arg1 = # of 8-byte words to zero
  1057  		// arg2 = value to store (will always be zero)
  1058  		// arg3 = mem
  1059  		// returns mem
  1060  		{
  1061  			name:      "REPSTOSQ",
  1062  			argLength: 4,
  1063  			reg: regInfo{
  1064  				inputs:   []regMask{buildReg("DI"), buildReg("CX"), buildReg("AX")},
  1065  				clobbers: buildReg("DI CX"),
  1066  			},
  1067  			faultOnNilArg0: true,
  1068  			addrSinkArg0:   true,
  1069  		},
  1070  
  1071  		// With a register ABI, the actual register info for these instructions (i.e., what is used in regalloc) is augmented with per-call-site bindings of additional arguments to specific in and out registers.
  1072  		{name: "CALLstatic", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                                      // call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
  1073  		{name: "CALLtail", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},                                        // tail call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
  1074  		{name: "CALLtailinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},            // tail call fn by pointer. arg0=codeptr, last arg=mem, auxint=argsize, returns mem
  1075  		{name: "CALLclosure", argLength: -1, reg: regInfo{inputs: []regMask{gpsp, buildReg("DX"), regMask{}}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true}, // call function via closure.  arg0=codeptr, arg1=closure, last arg=mem, auxint=argsize, returns mem
  1076  		{name: "CALLinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                // call fn by pointer.  arg0=codeptr, last arg=mem, auxint=argsize, returns mem
  1077  
  1078  		// arg0 = destination pointer
  1079  		// arg1 = source pointer
  1080  		// arg2 = mem
  1081  		// auxint = # of bytes to copy
  1082  		// returns memory
  1083  		{
  1084  			name:      "LoweredMove",
  1085  			aux:       "Int64",
  1086  			argLength: 3,
  1087  			reg: regInfo{
  1088  				inputs:   []regMask{gp, gp},
  1089  				clobbers: buildReg("X14"), // uses X14 as a temporary
  1090  			},
  1091  			faultOnNilArg0: true,
  1092  			faultOnNilArg1: true,
  1093  			addrSinkArg0:   true,
  1094  			addrSinkArg1:   true,
  1095  		},
  1096  		// arg0 = destination pointer
  1097  		// arg1 = source pointer
  1098  		// arg2 = mem
  1099  		// auxint = # of bytes to copy
  1100  		// returns memory
  1101  		{
  1102  			name:      "LoweredMoveLoop",
  1103  			aux:       "Int64",
  1104  			argLength: 3,
  1105  			reg: regInfo{
  1106  				inputs:       []regMask{gp, gp},
  1107  				clobbers:     buildReg("X14"), // uses X14 as a temporary
  1108  				clobbersArg0: true,
  1109  				clobbersArg1: true,
  1110  			},
  1111  			clobberFlags:   true,
  1112  			faultOnNilArg0: true,
  1113  			faultOnNilArg1: true,
  1114  			addrSinkArg0:   true,
  1115  			addrSinkArg1:   true,
  1116  			needIntTemp:    true,
  1117  		},
  1118  
  1119  		// arg0 = destination pointer
  1120  		// arg1 = source pointer
  1121  		// arg2 = # of 8-byte words to copy
  1122  		// arg3 = mem
  1123  		// returns memory
  1124  		{
  1125  			name:      "REPMOVSQ",
  1126  			argLength: 4,
  1127  			reg: regInfo{
  1128  				inputs:   []regMask{buildReg("DI"), buildReg("SI"), buildReg("CX")},
  1129  				clobbers: buildReg("DI SI CX"),
  1130  			},
  1131  			faultOnNilArg0: true,
  1132  			faultOnNilArg1: true,
  1133  			addrSinkArg0:   true,
  1134  			addrSinkArg1:   true,
  1135  		},
  1136  
  1137  		// (InvertFlags (CMPQ a b)) == (CMPQ b a)
  1138  		// So if we want (SETL (CMPQ a b)) but we can't do that because a is a constant,
  1139  		// then we do (SETL (InvertFlags (CMPQ b a))) instead.
  1140  		// Rewrites will convert this to (SETG (CMPQ b a)).
  1141  		// InvertFlags is a pseudo-op which can't appear in assembly output.
  1142  		{name: "InvertFlags", argLength: 1}, // reverse direction of arg0
  1143  
  1144  		// Pseudo-ops
  1145  		{name: "LoweredGetG", argLength: 1, reg: gp01}, // arg0=mem
  1146  		// Scheduler ensures LoweredGetClosurePtr occurs only in entry block,
  1147  		// and sorts it to the very beginning of the block to prevent other
  1148  		// use of DX (the closure pointer)
  1149  		{name: "LoweredGetClosurePtr", reg: regInfo{outputs: []regMask{buildReg("DX")}}, zeroWidth: true},
  1150  		// LoweredGetCallerPC evaluates to the PC to which its "caller" will return.
  1151  		// I.e., if f calls g "calls" sys.GetCallerPC,
  1152  		// the result should be the PC within f that g will return to.
  1153  		// See runtime/stubs.go for a more detailed discussion.
  1154  		{name: "LoweredGetCallerPC", reg: gp01, rematerializeable: true},
  1155  		// LoweredGetCallerSP returns the SP of the caller of the current function. arg0=mem
  1156  		{name: "LoweredGetCallerSP", argLength: 1, reg: gp01, rematerializeable: true},
  1157  		//arg0=ptr,arg1=mem, returns void.  Faults if ptr is nil.
  1158  		{name: "LoweredNilCheck", argLength: 2, reg: regInfo{inputs: []regMask{gpsp}}, clobberFlags: true, nilCheck: true, faultOnNilArg0: true},
  1159  		// LoweredWB invokes runtime.gcWriteBarrier{auxint}. arg0=mem, auxint=# of buffer entries needed.
  1160  		// It saves all GP registers if necessary, but may clobber others.
  1161  		// Returns a pointer to a write barrier buffer in R11.
  1162  		{name: "LoweredWB", argLength: 1, reg: regInfo{clobbers: callerSave.minus(gp.union(g)), outputs: []regMask{buildReg("R11")}}, clobberFlags: true, aux: "Int64"},
  1163  
  1164  		{name: "LoweredHasCPUFeature", argLength: 0, reg: gp01, rematerializeable: true, typ: "UInt64", aux: "Sym", symEffect: "None", zeroUpperBits: 56},
  1165  
  1166  		// LoweredPanicBoundsRR takes x and y, two values that caused a bounds check to fail.
  1167  		// the RC and CR versions are used when one of the arguments is a constant. CC is used
  1168  		// when both are constant (normally both 0, as prove derives the fact that a [0] bounds
  1169  		// failure means the length must have also been 0).
  1170  		// AuxInt contains a report code (see PanicBounds in genericOps.go).
  1171  		{name: "LoweredPanicBoundsRR", argLength: 3, aux: "Int64", reg: regInfo{inputs: []regMask{gp, gp}}, typ: "Mem", call: true},    // arg0=x, arg1=y, arg2=mem, returns memory.
  1172  		{name: "LoweredPanicBoundsRC", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{gp}}, typ: "Mem", call: true}, // arg0=x, arg1=mem, returns memory.
  1173  		{name: "LoweredPanicBoundsCR", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{gp}}, typ: "Mem", call: true}, // arg0=y, arg1=mem, returns memory.
  1174  		{name: "LoweredPanicBoundsCC", argLength: 1, aux: "PanicBoundsCC", reg: regInfo{}, typ: "Mem", call: true},                     // arg0=mem, returns memory.
  1175  
  1176  		// Constant flag values. For any comparison, there are 5 possible
  1177  		// outcomes: the three from the signed total order (<,==,>) and the
  1178  		// three from the unsigned total order. The == cases overlap.
  1179  		// Note: there's a sixth "unordered" outcome for floating-point
  1180  		// comparisons, but we don't use such a beast yet.
  1181  		// These ops are for temporary use by rewrite rules. They
  1182  		// cannot appear in the generated assembly.
  1183  		{name: "FlagEQ"},     // equal
  1184  		{name: "FlagLT_ULT"}, // signed < and unsigned <
  1185  		{name: "FlagLT_UGT"}, // signed < and unsigned >
  1186  		{name: "FlagGT_UGT"}, // signed > and unsigned >
  1187  		{name: "FlagGT_ULT"}, // signed > and unsigned <
  1188  
  1189  		// Atomic loads.  These are just normal loads but return <value,memory> tuples
  1190  		// so they can be properly ordered with other loads.
  1191  		// load from arg0+auxint+aux.  arg1=mem.
  1192  		{name: "MOVBatomicload", argLength: 2, reg: gpload, asm: "MOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1193  		{name: "MOVLatomicload", argLength: 2, reg: gpload, asm: "MOVL", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", zeroUpperBits: 32},
  1194  		{name: "MOVQatomicload", argLength: 2, reg: gpload, asm: "MOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1195  
  1196  		// Atomic stores and exchanges.  Stores use XCHG to get the right memory ordering semantics.
  1197  		// store arg0 to arg1+auxint+aux, arg2=mem.
  1198  		// These ops return a tuple of <old contents of *(arg1+auxint+aux), memory>.
  1199  		// Note: arg0 and arg1 are backwards compared to MOVLstore (to facilitate resultInArg0)!
  1200  		{name: "XCHGB", argLength: 3, reg: gpstorexchg, asm: "XCHGB", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1201  		{name: "XCHGL", argLength: 3, reg: gpstorexchg, asm: "XCHGL", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr", zeroUpperBits: 32},
  1202  		{name: "XCHGQ", argLength: 3, reg: gpstorexchg, asm: "XCHGQ", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1203  
  1204  		// Atomic adds and exchange.
  1205  		// *(arg1+auxint+aux) += arg0.  arg2=mem.
  1206  		// Returns a tuple of <old contents of *(arg1+auxint+aux), memory>.
  1207  		// Note: arg0 and arg1 are backwards compared to MOVLstore (to facilitate resultInArg0)!
  1208  		{name: "XADDLlock", argLength: 3, reg: gpstorexchg, asm: "XADDL", typ: "(UInt32,Mem)", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr", zeroUpperBits: 32},
  1209  		{name: "XADDQlock", argLength: 3, reg: gpstorexchg, asm: "XADDQ", typ: "(UInt64,Mem)", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1210  		{name: "AddTupleFirst32", argLength: 2}, // arg1=tuple <x,y>.  Returns <x+arg0,y>.
  1211  		{name: "AddTupleFirst64", argLength: 2}, // arg1=tuple <x,y>.  Returns <x+arg0,y>.
  1212  
  1213  		// Atomic adds, used when we do atomic.Add* and do not use the returned value.
  1214  		// (*arg0+auxint+aux) += arg1.  arg2=mem.
  1215  		// returns memory
  1216  		// Note: arg0 and arg1 are backwards compared to XADD*lock.
  1217  		{name: "ADDLlock", argLength: 3, reg: gpstore, asm: "ADDL", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1218  		{name: "ADDQlock", argLength: 3, reg: gpstore, asm: "ADDQ", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1219  
  1220  		// Atomic subtracts, used when we do atomic.Add* and do not use the returned value.
  1221  		// (*arg0+auxint+aux) -= arg1.  arg2=mem.
  1222  		// returns memory
  1223  		// Note: arg0 and arg1 are backwards compared to XADD*lock.
  1224  		{name: "SUBLlock", argLength: 3, reg: gpstore, asm: "SUBL", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1225  		{name: "SUBQlock", argLength: 3, reg: gpstore, asm: "SUBQ", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1226  
  1227  		// TODO: ADDlockconst & SUBlockconst
  1228  
  1229  		// Atomic inc & dec.
  1230  		// (*arg0+auxint+aux) += 1.  arg1=mem.
  1231  		// (*arg0+auxint+aux) -= 1.  arg1=mem.
  1232  		// returns memory
  1233  		// Note: arg0 is backwards compared to XADD*lock.
  1234  		{name: "INCLlock", argLength: 2, reg: gpstoreconst, asm: "INCL", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1235  		{name: "INCQlock", argLength: 2, reg: gpstoreconst, asm: "INCQ", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1236  		{name: "DECLlock", argLength: 2, reg: gpstoreconst, asm: "DECL", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1237  		{name: "DECQlock", argLength: 2, reg: gpstoreconst, asm: "DECQ", typ: "Mem", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1238  
  1239  		// Compare and swap.
  1240  		// arg0 = pointer, arg1 = old value, arg2 = new value, arg3 = memory.
  1241  		// if *(arg0+auxint+aux) == arg1 {
  1242  		//   *(arg0+auxint+aux) = arg2
  1243  		//   return (true, memory)
  1244  		// } else {
  1245  		//   return (false, memory)
  1246  		// }
  1247  		// Note that these instructions also return the old value in AX, but we ignore it.
  1248  		// TODO: have these return flags instead of bool.  The current system generates:
  1249  		//    CMPXCHGQ ...
  1250  		//    SETEQ AX
  1251  		//    CMPB  AX, $0
  1252  		//    JNE ...
  1253  		// instead of just
  1254  		//    CMPXCHGQ ...
  1255  		//    JEQ ...
  1256  		// but we can't do that because memory-using ops can't generate flags yet
  1257  		// (flagalloc wants to move flag-generating instructions around).
  1258  		{name: "CMPXCHGLlock", argLength: 4, reg: cmpxchg, asm: "CMPXCHGL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1259  		{name: "CMPXCHGQlock", argLength: 4, reg: cmpxchg, asm: "CMPXCHGQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1260  
  1261  		// Atomic memory updates using logical operations.
  1262  		// Old style that just returns the memory state.
  1263  		{name: "ANDBlock", argLength: 3, reg: gpstore, asm: "ANDB", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1264  		{name: "ANDLlock", argLength: 3, reg: gpstore, asm: "ANDL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1265  		{name: "ANDQlock", argLength: 3, reg: gpstore, asm: "ANDQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1266  		{name: "ORBlock", argLength: 3, reg: gpstore, asm: "ORB", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1267  		{name: "ORLlock", argLength: 3, reg: gpstore, asm: "ORL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1268  		{name: "ORQlock", argLength: 3, reg: gpstore, asm: "ORQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1269  
  1270  		// Atomic memory updates using logical operations.
  1271  		// *(arg0+auxint+aux) op= arg1. arg2=mem.
  1272  		// New style that returns a tuple of <old contents of *(arg0+auxint+aux), memory>.
  1273  		{name: "LoweredAtomicAnd64", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ANDQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1274  		{name: "LoweredAtomicAnd32", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ANDL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true, zeroUpperBits: 32},
  1275  		{name: "LoweredAtomicOr64", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ORQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1276  		{name: "LoweredAtomicOr32", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ORL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true, zeroUpperBits: 32},
  1277  
  1278  		// Prefetch instructions
  1279  		// Do prefetch arg0 address. arg0=addr, arg1=memory. Instruction variant selects locality hint
  1280  		{name: "PrefetchT0", argLength: 2, reg: prefreg, asm: "PREFETCHT0", hasSideEffects: true},
  1281  		{name: "PrefetchNTA", argLength: 2, reg: prefreg, asm: "PREFETCHNTA", hasSideEffects: true},
  1282  
  1283  		// CPUID feature: BMI1.
  1284  		{name: "ANDNQ", argLength: 2, reg: gp21, asm: "ANDNQ", clobberFlags: true},                            // arg0 &^ arg1
  1285  		{name: "ANDNL", argLength: 2, reg: gp21, asm: "ANDNL", clobberFlags: true, zeroUpperBits: 32},         // arg0 &^ arg1
  1286  		{name: "BLSIQ", argLength: 1, reg: gp11, asm: "BLSIQ", clobberFlags: true},                            // arg0 & -arg0
  1287  		{name: "BLSIL", argLength: 1, reg: gp11, asm: "BLSIL", clobberFlags: true, zeroUpperBits: 32},         // arg0 & -arg0
  1288  		{name: "BLSMSKQ", argLength: 1, reg: gp11, asm: "BLSMSKQ", clobberFlags: true},                        // arg0 ^ (arg0 - 1)
  1289  		{name: "BLSMSKL", argLength: 1, reg: gp11, asm: "BLSMSKL", clobberFlags: true, zeroUpperBits: 32},     // arg0 ^ (arg0 - 1)
  1290  		{name: "BLSRQ", argLength: 1, reg: gp11flags, asm: "BLSRQ", typ: "(UInt64,Flags)"},                    // arg0 & (arg0 - 1)
  1291  		{name: "BLSRL", argLength: 1, reg: gp11flags, asm: "BLSRL", typ: "(UInt32,Flags)", zeroUpperBits: 32}, // arg0 & (arg0 - 1)
  1292  		// count the number of trailing zero bits, prefer TZCNTQ over BSFQ, as TZCNTQ(0)==64
  1293  		// and BSFQ(0) is undefined. Same for TZCNTL(0)==32
  1294  		//
  1295  		// TZCNT/LZCNT deliberately carry no zeroUpperBits: their result is
  1296  		// bounded only as long as every rule producing them stays gated on
  1297  		// GOAMD64 >= 3. On older parts their REP BSF/BSR encodings silently
  1298  		// decode as legacy BSF/BSR, which leave the destination unmodified
  1299  		// on zero input — a guarantee too easy to break silently.
  1300  		{name: "TZCNTQ", argLength: 1, reg: gp11, asm: "TZCNTQ", clobberFlags: true},
  1301  		{name: "TZCNTL", argLength: 1, reg: gp11, asm: "TZCNTL", clobberFlags: true},
  1302  
  1303  		// CPUID feature: LZCNT.
  1304  		// count the number of leading zero bits.
  1305  		{name: "LZCNTQ", argLength: 1, reg: gp11, asm: "LZCNTQ", typ: "UInt64", clobberFlags: true},
  1306  		{name: "LZCNTL", argLength: 1, reg: gp11, asm: "LZCNTL", typ: "UInt32", clobberFlags: true},
  1307  
  1308  		// CPUID feature: MOVBE
  1309  		// MOVBEWload does not satisfy zero extended, so only use MOVBEWstore
  1310  		{name: "MOVBEWstore", argLength: 3, reg: gpstore, asm: "MOVBEW", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                    // swap and store 2 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1311  		{name: "MOVBELload", argLength: 2, reg: gpload, asm: "MOVBEL", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // load and swap 4 bytes from arg0+auxint+aux. arg1=mem.  Zero extend.
  1312  		{name: "MOVBELstore", argLength: 3, reg: gpstore, asm: "MOVBEL", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                    // swap and store 4 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1313  		{name: "MOVBEQload", argLength: 2, reg: gpload, asm: "MOVBEQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // load and swap 8 bytes from arg0+auxint+aux. arg1=mem
  1314  		{name: "MOVBEQstore", argLength: 3, reg: gpstore, asm: "MOVBEQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                    // swap and store 8 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1315  		// indexed MOVBE loads
  1316  		{name: "MOVBELloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBEL", scale: 1, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 32}, // load and swap 4 bytes from arg0+arg1+auxint+aux. arg2=mem. Zero extend.
  1317  		{name: "MOVBELloadidx4", argLength: 3, reg: gploadidx, asm: "MOVBEL", scale: 4, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},                                        // load and swap 4 bytes from arg0+4*arg1+auxint+aux. arg2=mem. Zero extend.
  1318  		{name: "MOVBELloadidx8", argLength: 3, reg: gploadidx, asm: "MOVBEL", scale: 8, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},                                        // load and swap 4 bytes from arg0+8*arg1+auxint+aux. arg2=mem. Zero extend.
  1319  		{name: "MOVBEQloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBEQ", scale: 1, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},                    // load and swap 8 bytes from arg0+arg1+auxint+aux. arg2=mem
  1320  		{name: "MOVBEQloadidx8", argLength: 3, reg: gploadidx, asm: "MOVBEQ", scale: 8, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true},                                                           // load and swap 8 bytes from arg0+8*arg1+auxint+aux. arg2=mem
  1321  		// indexed MOVBE stores
  1322  		{name: "MOVBEWstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEW", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 2 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1323  		{name: "MOVBEWstoreidx2", argLength: 4, reg: gpstoreidx, asm: "MOVBEW", scale: 2, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 2 bytes in arg2 to arg0+2*arg1+auxint+aux. arg3=mem
  1324  		{name: "MOVBELstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEL", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 4 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1325  		{name: "MOVBELstoreidx4", argLength: 4, reg: gpstoreidx, asm: "MOVBEL", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 4 bytes in arg2 to arg0+4*arg1+auxint+aux. arg3=mem
  1326  		{name: "MOVBELstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVBEL", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 4 bytes in arg2 to arg0+8*arg1+auxint+aux. arg3=mem
  1327  		{name: "MOVBEQstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEQ", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 8 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1328  		{name: "MOVBEQstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVBEQ", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 8 bytes in arg2 to arg0+8*arg1+auxint+aux. arg3=mem
  1329  
  1330  		// CPUID feature: BMI2.
  1331  		{name: "SARXQ", argLength: 2, reg: gp21, asm: "SARXQ"},                    // signed arg0 >> arg1, shift amount is mod 64
  1332  		{name: "SARXL", argLength: 2, reg: gp21, asm: "SARXL", zeroUpperBits: 32}, // signed int32(arg0) >> arg1, shift amount is mod 32
  1333  		{name: "SHLXQ", argLength: 2, reg: gp21, asm: "SHLXQ"},                    // arg0 << arg1, shift amount is mod 64
  1334  		{name: "SHLXL", argLength: 2, reg: gp21, asm: "SHLXL", zeroUpperBits: 32}, // arg0 << arg1, shift amount is mod 32
  1335  		{name: "SHRXQ", argLength: 2, reg: gp21, asm: "SHRXQ"},                    // unsigned arg0 >> arg1, shift amount is mod 64
  1336  		{name: "SHRXL", argLength: 2, reg: gp21, asm: "SHRXL", zeroUpperBits: 32}, // unsigned uint32(arg0) >> arg1, shift amount is mod 32
  1337  
  1338  		{name: "SARXLload", argLength: 3, reg: gp21shxload, asm: "SARXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 32
  1339  		{name: "SARXQload", argLength: 3, reg: gp21shxload, asm: "SARXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // signed *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 64
  1340  		{name: "SHLXLload", argLength: 3, reg: gp21shxload, asm: "SHLXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+auxint+aux) << arg1, arg2=mem, shift amount is mod 32
  1341  		{name: "SHLXQload", argLength: 3, reg: gp21shxload, asm: "SHLXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // *(arg0+auxint+aux) << arg1, arg2=mem, shift amount is mod 64
  1342  		{name: "SHRXLload", argLength: 3, reg: gp21shxload, asm: "SHRXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 32
  1343  		{name: "SHRXQload", argLength: 3, reg: gp21shxload, asm: "SHRXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // unsigned *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 64
  1344  
  1345  		{name: "SARXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1346  		{name: "SARXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+4*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1347  		{name: "SARXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1348  		{name: "SARXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SARXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // signed *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1349  		{name: "SARXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SARXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // signed *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1350  		{name: "SHLXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+1*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1351  		{name: "SHLXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+4*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1352  		{name: "SHLXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+8*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1353  		{name: "SHLXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHLXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // *(arg0+1*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 64
  1354  		{name: "SHLXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHLXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // *(arg0+8*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 64
  1355  		{name: "SHRXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1356  		{name: "SHRXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+4*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1357  		{name: "SHRXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1358  		{name: "SHRXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHRXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // unsigned *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1359  		{name: "SHRXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHRXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // unsigned *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1360  
  1361  		// Unpack bytes, low 64-bits.
  1362  		//
  1363  		// Input/output registers treated as [8]uint8.
  1364  		//
  1365  		// output = {in1[0], in2[0], in1[1], in2[1], in1[2], in2[2], in1[3], in2[3]}
  1366  		{name: "PUNPCKLBW", argLength: 2, reg: fp21, resultInArg0: true, asm: "PUNPCKLBW"},
  1367  
  1368  		// Shuffle 16-bit words, low 64-bits.
  1369  		//
  1370  		// Input/output registers treated as [4]uint16.
  1371  		// aux=source word index for each destination word, 2 bits per index.
  1372  		//
  1373  		// output[i] = input[(aux>>2*i)&3].
  1374  		{name: "PSHUFLW", argLength: 1, reg: fp11, aux: "Int8", asm: "PSHUFLW"},
  1375  
  1376  		// Broadcast input byte.
  1377  		//
  1378  		// Input treated as uint8, output treated as [16]uint8.
  1379  		//
  1380  		// output[i] = input.
  1381  		{name: "PSHUFBbroadcast", argLength: 1, reg: fp11, resultInArg0: true, asm: "PSHUFB"}, // PSHUFB with mask zero, (GOAMD64=v1)
  1382  		{name: "VPBROADCASTB", argLength: 1, reg: gpfp, asm: "VPBROADCASTB"},                  // Broadcast input byte from gp (GOAMD64=v3)
  1383  
  1384  		// Byte negate/zero/preserve (GOAMD64=v2).
  1385  		//
  1386  		// Input/output registers treated as [16]uint8.
  1387  		//
  1388  		// if in2[i] > 0 {
  1389  		//   output[i] = in1[i]
  1390  		// } else if in2[i] == 0 {
  1391  		//   output[i] = 0
  1392  		// } else {
  1393  		//   output[i] = -1 * in1[i]
  1394  		// }
  1395  		{name: "PSIGNB", argLength: 2, reg: fp21, resultInArg0: true, asm: "PSIGNB"},
  1396  
  1397  		// Byte compare.
  1398  		//
  1399  		// Input/output registers treated as [16]uint8.
  1400  		//
  1401  		// if in1[i] == in2[i] {
  1402  		//   output[i] = 0xff
  1403  		// } else {
  1404  		//   output[i] = 0
  1405  		// }
  1406  		{name: "PCMPEQB", argLength: 2, reg: fp21, resultInArg0: true, asm: "PCMPEQB", commutative: true},
  1407  
  1408  		// Byte sign mask. Output is a bitmap of sign bits from each input byte.
  1409  		//
  1410  		// Input treated as [16]uint8. Output is [16]bit (uint16 bitmap).
  1411  		//
  1412  		// output[i] = (input[i] >> 7) & 1
  1413  		{name: "PMOVMSKB", argLength: 1, reg: fpgp, asm: "PMOVMSKB", zeroUpperBits: 48},
  1414  
  1415  		// SIMD ops
  1416  		{name: "VMOVDQUload128", argLength: 2, reg: vload, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1417  		{name: "VMOVDQUstore128", argLength: 3, reg: vstore, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1418  
  1419  		{name: "VMOVDQUload256", argLength: 2, reg: vload, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1420  		{name: "VMOVDQUstore256", argLength: 3, reg: vstore, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1421  
  1422  		{name: "VMOVDQUload512", argLength: 2, reg: wload, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1423  		{name: "VMOVDQUstore512", argLength: 3, reg: wstore, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1424  
  1425  		// AVX2 32 and 64-bit element int-vector masked moves.
  1426  		{name: "VPMASK32load128", argLength: 3, reg: vloadv, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1427  		{name: "VPMASK32store128", argLength: 4, reg: vstorev, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1428  		{name: "VPMASK64load128", argLength: 3, reg: vloadv, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1429  		{name: "VPMASK64store128", argLength: 4, reg: vstorev, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1430  
  1431  		{name: "VPMASK32load256", argLength: 3, reg: vloadv, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1432  		{name: "VPMASK32store256", argLength: 4, reg: vstorev, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1433  		{name: "VPMASK64load256", argLength: 3, reg: vloadv, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1434  		{name: "VPMASK64store256", argLength: 4, reg: vstorev, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1435  
  1436  		// AVX512 8-64-bit element mask-register masked moves
  1437  		{name: "VPMASK8load512", argLength: 3, reg: wloadk, asm: "VMOVDQU8", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},      // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1438  		{name: "VPMASK8store512", argLength: 4, reg: wstorek, asm: "VMOVDQU8", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},   // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1439  		{name: "VPMASK16load512", argLength: 3, reg: wloadk, asm: "VMOVDQU16", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1440  		{name: "VPMASK16store512", argLength: 4, reg: wstorek, asm: "VMOVDQU16", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1441  		{name: "VPMASK32load512", argLength: 3, reg: wloadk, asm: "VMOVDQU32", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1442  		{name: "VPMASK32store512", argLength: 4, reg: wstorek, asm: "VMOVDQU32", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1443  		{name: "VPMASK64load512", argLength: 3, reg: wloadk, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1444  		{name: "VPMASK64store512", argLength: 4, reg: wstorek, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1445  
  1446  		// AVX512 moves between int-vector and mask registers
  1447  		{name: "VPMOVMToVec8x16", argLength: 1, reg: kv, asm: "VPMOVM2B"},
  1448  		{name: "VPMOVMToVec8x32", argLength: 1, reg: kv, asm: "VPMOVM2B"},
  1449  		{name: "VPMOVMToVec8x64", argLength: 1, reg: kw, asm: "VPMOVM2B"},
  1450  
  1451  		{name: "VPMOVMToVec16x8", argLength: 1, reg: kv, asm: "VPMOVM2W"},
  1452  		{name: "VPMOVMToVec16x16", argLength: 1, reg: kv, asm: "VPMOVM2W"},
  1453  		{name: "VPMOVMToVec16x32", argLength: 1, reg: kw, asm: "VPMOVM2W"},
  1454  
  1455  		{name: "VPMOVMToVec32x4", argLength: 1, reg: kv, asm: "VPMOVM2D"},
  1456  		{name: "VPMOVMToVec32x8", argLength: 1, reg: kv, asm: "VPMOVM2D"},
  1457  		{name: "VPMOVMToVec32x16", argLength: 1, reg: kw, asm: "VPMOVM2D"},
  1458  
  1459  		{name: "VPMOVMToVec64x2", argLength: 1, reg: kv, asm: "VPMOVM2Q"},
  1460  		{name: "VPMOVMToVec64x4", argLength: 1, reg: kv, asm: "VPMOVM2Q"},
  1461  		{name: "VPMOVMToVec64x8", argLength: 1, reg: kw, asm: "VPMOVM2Q"},
  1462  
  1463  		{name: "VPMOVVec8x16ToM", argLength: 1, reg: vk, asm: "VPMOVB2M"},
  1464  		{name: "VPMOVVec8x32ToM", argLength: 1, reg: vk, asm: "VPMOVB2M"},
  1465  		{name: "VPMOVVec8x64ToM", argLength: 1, reg: wk, asm: "VPMOVB2M"},
  1466  
  1467  		{name: "VPMOVVec16x8ToM", argLength: 1, reg: vk, asm: "VPMOVW2M"},
  1468  		{name: "VPMOVVec16x16ToM", argLength: 1, reg: vk, asm: "VPMOVW2M"},
  1469  		{name: "VPMOVVec16x32ToM", argLength: 1, reg: wk, asm: "VPMOVW2M"},
  1470  
  1471  		{name: "VPMOVVec32x4ToM", argLength: 1, reg: vk, asm: "VPMOVD2M"},
  1472  		{name: "VPMOVVec32x8ToM", argLength: 1, reg: vk, asm: "VPMOVD2M"},
  1473  		{name: "VPMOVVec32x16ToM", argLength: 1, reg: wk, asm: "VPMOVD2M"},
  1474  
  1475  		{name: "VPMOVVec64x2ToM", argLength: 1, reg: vk, asm: "VPMOVQ2M"},
  1476  		{name: "VPMOVVec64x4ToM", argLength: 1, reg: vk, asm: "VPMOVQ2M"},
  1477  		{name: "VPMOVVec64x8ToM", argLength: 1, reg: wk, asm: "VPMOVQ2M"},
  1478  
  1479  		// AVX1/2 moves from int-vector to bitmask (extracting sign bits)
  1480  		{name: "VPMOVMSKB128", argLength: 1, reg: vgp, asm: "VPMOVMSKB", zeroUpperBits: 48},
  1481  		{name: "VPMOVMSKB256", argLength: 1, reg: vgp, asm: "VPMOVMSKB", zeroUpperBits: 32},
  1482  		{name: "VMOVMSKPS128", argLength: 1, reg: vgp, asm: "VMOVMSKPS", zeroUpperBits: 56},
  1483  		{name: "VMOVMSKPS256", argLength: 1, reg: vgp, asm: "VMOVMSKPS", zeroUpperBits: 56},
  1484  		{name: "VMOVMSKPD128", argLength: 1, reg: vgp, asm: "VMOVMSKPD", zeroUpperBits: 56},
  1485  		{name: "VMOVMSKPD256", argLength: 1, reg: vgp, asm: "VMOVMSKPD", zeroUpperBits: 56},
  1486  
  1487  		{name: "Zero128", argLength: 0, reg: x15only, zeroWidth: true, fixedReg: true},
  1488  		{name: "Zero256", argLength: 0, reg: x15only, zeroWidth: true, fixedReg: true},
  1489  		{name: "Zero512", argLength: 0, reg: x15only, zeroWidth: true, fixedReg: true},
  1490  
  1491  		// Move a 32/64 bit float to a 128-bit SIMD register.
  1492  		{name: "VMOVSDf2v", argLength: 1, reg: fpv, asm: "VMOVSD"},
  1493  		{name: "VMOVSSf2v", argLength: 1, reg: fpv, asm: "VMOVSS"},
  1494  
  1495  		{name: "VMOVQ", argLength: 1, reg: gpv, asm: "VMOVQ"},
  1496  		{name: "VMOVD", argLength: 1, reg: gpv, asm: "VMOVD"},
  1497  
  1498  		{name: "VMOVQload", argLength: 2, reg: fpload, asm: "VMOVQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read"},
  1499  		{name: "VMOVDload", argLength: 2, reg: fpload, asm: "VMOVD", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read"},
  1500  		{name: "VMOVSSload", argLength: 2, reg: fpload, asm: "VMOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1501  		{name: "VMOVSDload", argLength: 2, reg: fpload, asm: "VMOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1502  
  1503  		{name: "VMOVSSconst", reg: fp01, asm: "VMOVSS", aux: "Float32", rematerializeable: true},
  1504  		{name: "VMOVSDconst", reg: fp01, asm: "VMOVSD", aux: "Float64", rematerializeable: true},
  1505  
  1506  		{name: "VZEROUPPER", argLength: 1, reg: regInfo{clobbers: v}, asm: "VZEROUPPER"}, // arg=mem, returns mem
  1507  		{name: "VZEROALL", argLength: 1, reg: regInfo{clobbers: v}, asm: "VZEROALL"},     // arg=mem, returns mem
  1508  
  1509  		// KMOVxload: loads masks
  1510  		// Load (Q=8,D=4,W=2,B=1) bytes from (arg0+auxint+aux), arg1=mem.
  1511  		// "+auxint+aux" == add auxint and the offset of the symbol in aux (if any) to the effective address
  1512  		{name: "KMOVBload", argLength: 2, reg: kload, asm: "KMOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1513  		{name: "KMOVWload", argLength: 2, reg: kload, asm: "KMOVW", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1514  		{name: "KMOVDload", argLength: 2, reg: kload, asm: "KMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1515  		{name: "KMOVQload", argLength: 2, reg: kload, asm: "KMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1516  
  1517  		// KMOVxstore: stores masks
  1518  		// Store (Q=8,D=4,W=2,B=1) low bytes of arg1.
  1519  		// Does *(arg0+auxint+aux) = arg1, arg2=mem.
  1520  		{name: "KMOVBstore", argLength: 3, reg: kstore, asm: "KMOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1521  		{name: "KMOVWstore", argLength: 3, reg: kstore, asm: "KMOVW", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1522  		{name: "KMOVDstore", argLength: 3, reg: kstore, asm: "KMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1523  		{name: "KMOVQstore", argLength: 3, reg: kstore, asm: "KMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1524  
  1525  		// Move GP directly to mask register
  1526  		{name: "KMOVQk", argLength: 1, reg: gpk, asm: "KMOVQ"},
  1527  		{name: "KMOVDk", argLength: 1, reg: gpk, asm: "KMOVD"},
  1528  		{name: "KMOVWk", argLength: 1, reg: gpk, asm: "KMOVW"},
  1529  		{name: "KMOVBk", argLength: 1, reg: gpk, asm: "KMOVB"},
  1530  		{name: "KMOVQi", argLength: 1, reg: kgp, asm: "KMOVQ"},
  1531  		{name: "KMOVDi", argLength: 1, reg: kgp, asm: "KMOVD", zeroUpperBits: 32},
  1532  		{name: "KMOVWi", argLength: 1, reg: kgp, asm: "KMOVW", zeroUpperBits: 48},
  1533  		{name: "KMOVBi", argLength: 1, reg: kgp, asm: "KMOVB", zeroUpperBits: 56},
  1534  
  1535  		// Mask logical operations
  1536  		{name: "KANDB", argLength: 2, reg: k2k, asm: "KANDB", typ: "Mask"},
  1537  		{name: "KANDW", argLength: 2, reg: k2k, asm: "KANDW", typ: "Mask"},
  1538  		{name: "KANDD", argLength: 2, reg: k2k, asm: "KANDD", typ: "Mask"},
  1539  		{name: "KANDQ", argLength: 2, reg: k2k, asm: "KANDQ", typ: "Mask"},
  1540  
  1541  		{name: "KORB", argLength: 2, reg: k2k, asm: "KORB", typ: "Mask"},
  1542  		{name: "KORW", argLength: 2, reg: k2k, asm: "KORW", typ: "Mask"},
  1543  		{name: "KORD", argLength: 2, reg: k2k, asm: "KORD", typ: "Mask"},
  1544  		{name: "KORQ", argLength: 2, reg: k2k, asm: "KORQ", typ: "Mask"},
  1545  
  1546  		{name: "KXORB", argLength: 2, reg: k2k, asm: "KXORB", typ: "Mask"},
  1547  		{name: "KXORW", argLength: 2, reg: k2k, asm: "KXORW", typ: "Mask"},
  1548  		{name: "KXORD", argLength: 2, reg: k2k, asm: "KXORD", typ: "Mask"},
  1549  		{name: "KXORQ", argLength: 2, reg: k2k, asm: "KXORQ", typ: "Mask"},
  1550  
  1551  		// Following Intel convention, we call it XNOR instead of EQ.
  1552  		{name: "KXNORB", argLength: 2, reg: k2k, asm: "KXNORB", typ: "Mask"},
  1553  		{name: "KXNORW", argLength: 2, reg: k2k, asm: "KXNORW", typ: "Mask"},
  1554  		{name: "KXNORD", argLength: 2, reg: k2k, asm: "KXNORD", typ: "Mask"},
  1555  		{name: "KXNORQ", argLength: 2, reg: k2k, asm: "KXNORQ", typ: "Mask"},
  1556  
  1557  		// VPTEST
  1558  		{name: "VPTEST", asm: "VPTEST", argLength: 2, reg: v2flags, clobberFlags: true, typ: "Flags"},
  1559  	}
  1560  
  1561  	AMD64blocks := []blockData{
  1562  		{name: "EQ", controls: 1},
  1563  		{name: "NE", controls: 1},
  1564  		{name: "LT", controls: 1},
  1565  		{name: "LE", controls: 1},
  1566  		{name: "GT", controls: 1},
  1567  		{name: "GE", controls: 1},
  1568  		{name: "OS", controls: 1},
  1569  		{name: "OC", controls: 1},
  1570  		{name: "ULT", controls: 1},
  1571  		{name: "ULE", controls: 1},
  1572  		{name: "UGT", controls: 1},
  1573  		{name: "UGE", controls: 1},
  1574  		{name: "EQF", controls: 1},
  1575  		{name: "NEF", controls: 1},
  1576  		{name: "ORD", controls: 1}, // FP, ordered comparison (parity zero)
  1577  		{name: "NAN", controls: 1}, // FP, unordered comparison (parity one)
  1578  
  1579  		// JUMPTABLE implements jump tables.
  1580  		// Aux is the symbol (an *obj.LSym) for the jump table.
  1581  		// control[0] is the index into the jump table.
  1582  		// control[1] is the address of the jump table (the address of the symbol stored in Aux).
  1583  		{name: "JUMPTABLE", controls: 2, aux: "Sym"},
  1584  	}
  1585  
  1586  	archs = append(archs, arch{
  1587  		name:        "AMD64",
  1588  		pkg:         "cmd/internal/obj/x86",
  1589  		genfile:     "../../amd64/ssa.go",
  1590  		genSIMDfile: "../../amd64/simdssa.go",
  1591  		ops: append(AMD64ops, simdAMD64Ops(v11, v21, v2k, vkv, v2kv, v2kk, v31, v3kv, vgpv, vgp, vfpv, vfpkv,
  1592  			w11, w21, w2k, wkw, w2kw, w2kk, w31, w3kw, wgpw, wgp, wfpw, wfpkw, wkwload, v21load, v31load, v11load,
  1593  			w21load, w31load, w2kload, w2kwload, w11load, w3kwload, w2kkload, v31x0AtIn2)...), // AMD64ops,
  1594  		blocks:             AMD64blocks,
  1595  		regnames:           regNamesAMD64,
  1596  		ParamIntRegNames:   "AX BX CX DI SI R8 R9 R10 R11",
  1597  		ParamFloatRegNames: "X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14",
  1598  		gpregmask:          gp,
  1599  		fpregmask:          fp,
  1600  		specialregmask:     mask.union(w.minus(v)),
  1601  		simdregmask:        v,
  1602  		framepointerreg:    int8(num["BP"]),
  1603  		linkreg:            -1, // not used
  1604  	})
  1605  }
  1606  

View as plain text