Source file src/cmd/compile/internal/ssa/_gen/AMD64Ops.go

     1  // Copyright 2015 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package main
     6  
     7  import "strings"
     8  
     9  // Notes:
    10  //  - Integer types live in the low portion of registers. Upper portions are junk.
    11  //  - Boolean types use the low-order byte of a register. 0=false, 1=true.
    12  //    Upper bytes are junk.
    13  //  - Floating-point types live in the low natural slot of an sse2 register.
    14  //    Unused portions are junk.
    15  //  - We do not use AH,BH,CH,DH registers.
    16  //  - When doing sub-register operations, we try to write the whole
    17  //    destination register to avoid a partial-register write.
    18  //  - Unused portions of AuxInt (or the Val portion of ValAndOff) are
    19  //    filled by sign-extending the used portion.  Users of AuxInt which interpret
    20  //    AuxInt as unsigned (e.g. shifts) must be careful.
    21  //  - All SymOff opcodes require their offset to fit in an int32.
    22  
    23  // Suffixes encode the bit width of various instructions.
    24  // Q (quad word) = 64 bit
    25  // L (long word) = 32 bit
    26  // W (word)      = 16 bit
    27  // B (byte)      = 8 bit
    28  // D (double)    = 64 bit float
    29  // S (single)    = 32 bit float
    30  
    31  // copied from ../../amd64/reg.go
    32  var regNamesAMD64 = []string{
    33  	"AX",
    34  	"CX",
    35  	"DX",
    36  	"BX",
    37  	"SP",
    38  	"BP",
    39  	"SI",
    40  	"DI",
    41  	"R8",
    42  	"R9",
    43  	"R10",
    44  	"R11",
    45  	"R12",
    46  	"R13",
    47  	"g", // a.k.a. R14
    48  	"R15",
    49  	"X0",
    50  	"X1",
    51  	"X2",
    52  	"X3",
    53  	"X4",
    54  	"X5",
    55  	"X6",
    56  	"X7",
    57  	"X8",
    58  	"X9",
    59  	"X10",
    60  	"X11",
    61  	"X12",
    62  	"X13",
    63  	"X14",
    64  	"X15", // constant 0 in ABIInternal
    65  	"X16",
    66  	"X17",
    67  	"X18",
    68  	"X19",
    69  	"X20",
    70  	"X21",
    71  	"X22",
    72  	"X23",
    73  	"X24",
    74  	"X25",
    75  	"X26",
    76  	"X27",
    77  	"X28",
    78  	"X29",
    79  	"X30",
    80  	"X31",
    81  
    82  	// TODO: update asyncPreempt for K registers.
    83  	// asyncPreempt also needs to store Z0-Z15 properly.
    84  	"K0",
    85  	"K1",
    86  	"K2",
    87  	"K3",
    88  	"K4",
    89  	"K5",
    90  	"K6",
    91  	"K7",
    92  	// If you add registers, update asyncPreempt in runtime
    93  
    94  	// pseudo-registers
    95  	"SB",
    96  }
    97  
    98  func init() {
    99  	// Make map from reg names to reg integers.
   100  	if len(regNamesAMD64) > 64 {
   101  		panic("too many registers")
   102  	}
   103  	num := map[string]int{}
   104  	for i, name := range regNamesAMD64 {
   105  		num[name] = i
   106  	}
   107  	buildReg := func(s string) regMask {
   108  		m := regMask{}
   109  		for _, r := range strings.Split(s, " ") {
   110  			if n, ok := num[r]; ok {
   111  				m = m.addReg(uint(n))
   112  				continue
   113  			}
   114  			panic("register " + r + " not found")
   115  		}
   116  		return m
   117  	}
   118  
   119  	// Common individual register masks
   120  	var (
   121  		ax         = buildReg("AX")
   122  		cx         = buildReg("CX")
   123  		dx         = buildReg("DX")
   124  		gp         = buildReg("AX CX DX BX BP SI DI R8 R9 R10 R11 R12 R13 R15")
   125  		g          = buildReg("g")
   126  		fp         = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14")
   127  		v          = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14")
   128  		w          = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14 X16 X17 X18 X19 X20 X21 X22 X23 X24 X25 X26 X27 X28 X29 X30 X31")
   129  		x15        = buildReg("X15")
   130  		mask       = buildReg("K1 K2 K3 K4 K5 K6 K7")
   131  		gpsp       = gp.union(buildReg("SP"))
   132  		gpspsb     = gpsp.union(buildReg("SB"))
   133  		gpspsbg    = gpspsb.union(g)
   134  		callerSave = gp.union(w).union(mask).union(g) // runtime.setg (and anything calling it) may clobber g
   135  
   136  		vz = v.union(x15)
   137  		wz = w.union(x15)
   138  		x0 = buildReg("X0")
   139  	)
   140  	// Common slices of register masks
   141  	var (
   142  		gponly   = []regMask{gp}
   143  		fponly   = []regMask{fp}
   144  		vonly    = []regMask{v}
   145  		wonly    = []regMask{w}
   146  		maskonly = []regMask{mask}
   147  		vzonly   = []regMask{vz}
   148  		wzonly   = []regMask{wz}
   149  	)
   150  
   151  	// Common regInfo
   152  	var (
   153  		gp01           = regInfo{inputs: nil, outputs: gponly}
   154  		gp11           = regInfo{inputs: []regMask{gp}, outputs: gponly}
   155  		gp11sp         = regInfo{inputs: []regMask{gpsp}, outputs: gponly}
   156  		gp11sb         = regInfo{inputs: []regMask{gpspsbg}, outputs: gponly}
   157  		gp21           = regInfo{inputs: []regMask{gp, gp}, outputs: gponly}
   158  		gp21sp         = regInfo{inputs: []regMask{gpsp, gp}, outputs: gponly}
   159  		gp21sp2        = regInfo{inputs: []regMask{gp, gpsp}, outputs: gponly}
   160  		gp21sb         = regInfo{inputs: []regMask{gpspsbg, gpsp}, outputs: gponly}
   161  		gp21shift      = regInfo{inputs: []regMask{gp, cx}, outputs: []regMask{gp}}
   162  		gp11div        = regInfo{inputs: []regMask{ax, gpsp.minus(dx)}, outputs: []regMask{ax, dx}}
   163  		gp21hmul       = regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{dx}, clobbers: ax}
   164  		gp21flags      = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp, regMask{}}}
   165  		gp2flags1flags = regInfo{inputs: []regMask{gp, gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   166  
   167  		gp2flags     = regInfo{inputs: []regMask{gpsp, gpsp}}
   168  		gp1flags     = regInfo{inputs: []regMask{gpsp}}
   169  		gp0flagsLoad = regInfo{inputs: []regMask{gpspsbg, regMask{}}}
   170  		gp1flagsLoad = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   171  		gp2flagsLoad = regInfo{inputs: []regMask{gpspsbg, gpsp, gpsp, regMask{}}}
   172  		flagsgp      = regInfo{inputs: nil, outputs: gponly}
   173  
   174  		gp11flags      = regInfo{inputs: []regMask{gp}, outputs: []regMask{gp, regMask{}}}
   175  		gp1flags1flags = regInfo{inputs: []regMask{gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   176  
   177  		readflags = regInfo{inputs: nil, outputs: gponly}
   178  
   179  		gpload         = regInfo{inputs: []regMask{gpspsbg, regMask{}}, outputs: gponly}
   180  		gp21load       = regInfo{inputs: []regMask{gp, gpspsbg, regMask{}}, outputs: gponly}
   181  		gploadidx      = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}, outputs: gponly}
   182  		gp21loadidx    = regInfo{inputs: []regMask{gp, gpspsbg, gpsp, regMask{}}, outputs: gponly}
   183  		gp21shxload    = regInfo{inputs: []regMask{gpspsbg, gp, regMask{}}, outputs: gponly}
   184  		gp21shxloadidx = regInfo{inputs: []regMask{gpspsbg, gpsp, gp, regMask{}}, outputs: gponly}
   185  
   186  		gpstore         = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   187  		gpstoreconst    = regInfo{inputs: []regMask{gpspsbg, regMask{}}}
   188  		gpstoreidx      = regInfo{inputs: []regMask{gpspsbg, gpsp, gpsp, regMask{}}}
   189  		gpstoreconstidx = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   190  		gpstorexchg     = regInfo{inputs: []regMask{gp, gpspsbg, regMask{}}, outputs: []regMask{gp}}
   191  		cmpxchg         = regInfo{inputs: []regMask{gp, ax, gp, regMask{}}, outputs: []regMask{gp, regMask{}}, clobbers: ax}
   192  		atomicLogic     = regInfo{inputs: []regMask{gp.minus(ax), gp.minus(ax), regMask{}}, outputs: []regMask{ax, regMask{}}}
   193  
   194  		fp01        = regInfo{inputs: nil, outputs: fponly}
   195  		fp21        = regInfo{inputs: []regMask{fp, fp}, outputs: fponly}
   196  		fp31        = regInfo{inputs: []regMask{fp, fp, fp}, outputs: fponly}
   197  		fp21load    = regInfo{inputs: []regMask{fp, gpspsbg, regMask{}}, outputs: fponly}
   198  		fp21loadidx = regInfo{inputs: []regMask{fp, gpspsbg, gpspsb, regMask{}}, outputs: fponly}
   199  		fpgp        = regInfo{inputs: fponly, outputs: gponly}
   200  		gpfp        = regInfo{inputs: gponly, outputs: fponly}
   201  		fp11        = regInfo{inputs: fponly, outputs: fponly}
   202  		fp2flags    = regInfo{inputs: []regMask{fp, fp}}
   203  
   204  		fpload    = regInfo{inputs: []regMask{gpspsb, {}}, outputs: fponly}
   205  		fploadidx = regInfo{inputs: []regMask{gpspsb, gpsp, {}}, outputs: fponly}
   206  		vload     = regInfo{inputs: []regMask{gpspsb, {}}, outputs: vonly}
   207  		wload     = regInfo{inputs: []regMask{gpspsb, {}}, outputs: wonly}
   208  
   209  		fpstore    = regInfo{inputs: []regMask{gpspsb, fp, {}}}
   210  		fpstoreidx = regInfo{inputs: []regMask{gpspsb, gpsp, fp, {}}}
   211  		vstore     = regInfo{inputs: []regMask{gpspsb, vz, {}}}
   212  		wstore     = regInfo{inputs: []regMask{gpspsb, wz, {}}}
   213  
   214  		// masked loads/stores, vector register or mask register
   215  		vloadv  = regInfo{inputs: []regMask{gpspsb, v, {}}, outputs: vonly}
   216  		vstorev = regInfo{inputs: []regMask{gpspsb, v, vz, {}}}
   217  		wloadk  = regInfo{inputs: []regMask{gpspsb, mask, {}}, outputs: wonly}
   218  		wstorek = regInfo{inputs: []regMask{gpspsb, mask, wz, {}}}
   219  
   220  		v01     = regInfo{inputs: nil, outputs: vonly}
   221  		v11     = regInfo{inputs: vonly, outputs: vonly}            // used in resultInArg0 ops, arg0 must not be x15
   222  		v21     = regInfo{inputs: []regMask{v, vz}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   223  		vk      = regInfo{inputs: vzonly, outputs: maskonly}
   224  		kv      = regInfo{inputs: maskonly, outputs: vonly}
   225  		v2k     = regInfo{inputs: []regMask{vz, vz}, outputs: maskonly}
   226  		vkv     = regInfo{inputs: []regMask{vz, mask}, outputs: vonly}
   227  		v2kv    = regInfo{inputs: []regMask{vz, vz, mask}, outputs: vonly}
   228  		v2kk    = regInfo{inputs: []regMask{vz, vz, mask}, outputs: maskonly}
   229  		v31     = regInfo{inputs: []regMask{v, vz, vz}, outputs: vonly}       // used in resultInArg0 ops, arg0 must not be x15
   230  		v3kv    = regInfo{inputs: []regMask{v, vz, vz, mask}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   231  		vgpv    = regInfo{inputs: []regMask{vz, gp}, outputs: vonly}
   232  		vgp     = regInfo{inputs: vonly, outputs: gponly}
   233  		vfpv    = regInfo{inputs: []regMask{vz, fp}, outputs: vonly}
   234  		vfpkv   = regInfo{inputs: []regMask{vz, fp, mask}, outputs: vonly}
   235  		fpv     = regInfo{inputs: []regMask{fp}, outputs: vonly}
   236  		gpv     = regInfo{inputs: []regMask{gp}, outputs: vonly}
   237  		v2flags = regInfo{inputs: []regMask{vz, vz}}
   238  
   239  		w01   = regInfo{inputs: nil, outputs: wonly}
   240  		w11   = regInfo{inputs: wonly, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   241  		w21   = regInfo{inputs: []regMask{wz, wz}, outputs: wonly}
   242  		wk    = regInfo{inputs: wzonly, outputs: maskonly}
   243  		kw    = regInfo{inputs: maskonly, outputs: wonly}
   244  		w2k   = regInfo{inputs: []regMask{wz, wz}, outputs: maskonly}
   245  		wkw   = regInfo{inputs: []regMask{wz, mask}, outputs: wonly}
   246  		w2kw  = regInfo{inputs: []regMask{w, wz, mask}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   247  		w2kk  = regInfo{inputs: []regMask{wz, wz, mask}, outputs: maskonly}
   248  		w31   = regInfo{inputs: []regMask{w, wz, wz}, outputs: wonly}       // used in resultInArg0 ops, arg0 must not be x15
   249  		w3kw  = regInfo{inputs: []regMask{w, wz, wz, mask}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   250  		wgpw  = regInfo{inputs: []regMask{wz, gp}, outputs: wonly}
   251  		wgp   = regInfo{inputs: wzonly, outputs: gponly}
   252  		wfpw  = regInfo{inputs: []regMask{wz, fp}, outputs: wonly}
   253  		wfpkw = regInfo{inputs: []regMask{wz, fp, mask}, outputs: wonly}
   254  
   255  		// These register masks are used by SIMD only, they follow the pattern:
   256  		// Mem last, k mask second to last (if any), address right before mem and k mask.
   257  		wkwload    = regInfo{inputs: []regMask{gpspsb, mask, regMask{}}, outputs: wonly}
   258  		v21load    = regInfo{inputs: []regMask{v, gpspsb, regMask{}}, outputs: vonly}     // used in resultInArg0 ops, arg0 must not be x15
   259  		v31load    = regInfo{inputs: []regMask{v, vz, gpspsb, regMask{}}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   260  		v11load    = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: vonly}
   261  		w21load    = regInfo{inputs: []regMask{wz, gpspsb, regMask{}}, outputs: wonly}
   262  		w31load    = regInfo{inputs: []regMask{w, wz, gpspsb, regMask{}}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   263  		w2kload    = regInfo{inputs: []regMask{wz, gpspsb, regMask{}}, outputs: maskonly}
   264  		w2kwload   = regInfo{inputs: []regMask{wz, gpspsb, mask, regMask{}}, outputs: wonly}
   265  		w11load    = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: wonly}
   266  		w3kwload   = regInfo{inputs: []regMask{w, wz, gpspsb, mask, regMask{}}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   267  		w2kkload   = regInfo{inputs: []regMask{wz, gpspsb, mask, regMask{}}, outputs: maskonly}
   268  		v31x0AtIn2 = regInfo{inputs: []regMask{v, vz, x0}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   269  
   270  		kload  = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: maskonly}
   271  		kstore = regInfo{inputs: []regMask{gpspsb, mask, regMask{}}}
   272  		gpk    = regInfo{inputs: gponly, outputs: maskonly}
   273  		kgp    = regInfo{inputs: maskonly, outputs: gponly}
   274  		k2k    = regInfo{inputs: []regMask{mask, mask}, outputs: maskonly}
   275  
   276  		x15only = regInfo{inputs: nil, outputs: []regMask{x15}}
   277  
   278  		prefreg = regInfo{inputs: []regMask{gpspsbg}}
   279  	)
   280  
   281  	var AMD64ops = []opData{
   282  		// {ADD,SUB,MUL,DIV}Sx: floating-point arithmetic
   283  		// x==S for float32, x==D for float64
   284  		// computes arg0 OP arg1
   285  		{name: "ADDSS", argLength: 2, reg: fp21, asm: "ADDSS", commutative: true, resultInArg0: true, earlyOk: true},
   286  		{name: "ADDSD", argLength: 2, reg: fp21, asm: "ADDSD", commutative: true, resultInArg0: true, earlyOk: true},
   287  		{name: "SUBSS", argLength: 2, reg: fp21, asm: "SUBSS", resultInArg0: true, earlyOk: true},
   288  		{name: "SUBSD", argLength: 2, reg: fp21, asm: "SUBSD", resultInArg0: true, earlyOk: true},
   289  		{name: "MULSS", argLength: 2, reg: fp21, asm: "MULSS", commutative: true, resultInArg0: true, earlyOk: true},
   290  		{name: "MULSD", argLength: 2, reg: fp21, asm: "MULSD", commutative: true, resultInArg0: true, earlyOk: true},
   291  		{name: "DIVSS", argLength: 2, reg: fp21, asm: "DIVSS", resultInArg0: true, earlyOk: true},
   292  		{name: "DIVSD", argLength: 2, reg: fp21, asm: "DIVSD", resultInArg0: true, earlyOk: true},
   293  
   294  		// MOVSxload: floating-point loads
   295  		// x==S for float32, x==D for float64
   296  		// load from arg0+auxint+aux, arg1 = mem
   297  		{name: "MOVSSload", argLength: 2, reg: fpload, asm: "MOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   298  		{name: "MOVSDload", argLength: 2, reg: fpload, asm: "MOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   299  
   300  		// MOVSxconst: floatint-point constants
   301  		// x==S for float32, x==D for float64
   302  		{name: "MOVSSconst", reg: fp01, asm: "MOVSS", aux: "Float32", rematerializeable: true, earlyOk: true},
   303  		{name: "MOVSDconst", reg: fp01, asm: "MOVSD", aux: "Float64", rematerializeable: true, earlyOk: true},
   304  
   305  		// MOVSxloadidx: floating-point indexed loads
   306  		// x==S for float32, x==D for float64
   307  		// load from arg0 + scale*arg1+auxint+aux, arg2 = mem
   308  		{name: "MOVSSloadidx1", argLength: 3, reg: fploadidx, asm: "MOVSS", scale: 1, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   309  		{name: "MOVSSloadidx4", argLength: 3, reg: fploadidx, asm: "MOVSS", scale: 4, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   310  		{name: "MOVSDloadidx1", argLength: 3, reg: fploadidx, asm: "MOVSD", scale: 1, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   311  		{name: "MOVSDloadidx8", argLength: 3, reg: fploadidx, asm: "MOVSD", scale: 8, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   312  
   313  		// MOVSxstore: floating-point stores
   314  		// x==S for float32, x==D for float64
   315  		// does *(arg0+auxint+aux) = arg1, arg2 = mem
   316  		{name: "MOVSSstore", argLength: 3, reg: fpstore, asm: "MOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   317  		{name: "MOVSDstore", argLength: 3, reg: fpstore, asm: "MOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   318  
   319  		// MOVSxstoreidx: floating-point indexed stores
   320  		// x==S for float32, x==D for float64
   321  		// does *(arg0+scale*arg1+auxint+aux) = arg2, arg3 = mem
   322  		{name: "MOVSSstoreidx1", argLength: 4, reg: fpstoreidx, asm: "MOVSS", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   323  		{name: "MOVSSstoreidx4", argLength: 4, reg: fpstoreidx, asm: "MOVSS", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   324  		{name: "MOVSDstoreidx1", argLength: 4, reg: fpstoreidx, asm: "MOVSD", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   325  		{name: "MOVSDstoreidx8", argLength: 4, reg: fpstoreidx, asm: "MOVSD", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   326  
   327  		// {ADD,SUB,MUL,DIV}Sxload: floating-point load / op combo
   328  		// x==S for float32, x==D for float64
   329  		// computes arg0 OP *(arg1+auxint+aux), arg2=mem
   330  		{name: "ADDSSload", argLength: 3, reg: fp21load, asm: "ADDSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   331  		{name: "ADDSDload", argLength: 3, reg: fp21load, asm: "ADDSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   332  		{name: "SUBSSload", argLength: 3, reg: fp21load, asm: "SUBSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   333  		{name: "SUBSDload", argLength: 3, reg: fp21load, asm: "SUBSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   334  		{name: "MULSSload", argLength: 3, reg: fp21load, asm: "MULSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   335  		{name: "MULSDload", argLength: 3, reg: fp21load, asm: "MULSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   336  		{name: "DIVSSload", argLength: 3, reg: fp21load, asm: "DIVSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   337  		{name: "DIVSDload", argLength: 3, reg: fp21load, asm: "DIVSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   338  
   339  		// {ADD,SUB,MUL,DIV}Sxloadidx: floating-point indexed load / op combo
   340  		// x==S for float32, x==D for float64
   341  		// computes arg0 OP *(arg1+scale*arg2+auxint+aux), arg3=mem
   342  		{name: "ADDSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "ADDSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   343  		{name: "ADDSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "ADDSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   344  		{name: "ADDSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "ADDSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   345  		{name: "ADDSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "ADDSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   346  		{name: "SUBSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "SUBSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   347  		{name: "SUBSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "SUBSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   348  		{name: "SUBSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "SUBSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   349  		{name: "SUBSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "SUBSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   350  		{name: "MULSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "MULSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   351  		{name: "MULSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "MULSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   352  		{name: "MULSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "MULSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   353  		{name: "MULSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "MULSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   354  		{name: "DIVSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "DIVSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   355  		{name: "DIVSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "DIVSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   356  		{name: "DIVSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "DIVSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   357  		{name: "DIVSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "DIVSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   358  
   359  		// {ADD,SUB,MUL,DIV,AND,OR,XOR}x: binary integer ops
   360  		//   unadorned versions compute arg0 OP arg1
   361  		//       const versions compute arg0 OP auxint (auxint is a sign-extended 32-bit value)
   362  		// constmodify versions compute *(arg0+ValAndOff(AuxInt).Off().aux) OP= ValAndOff(AuxInt).Val(), arg1 = mem
   363  		// x==L operations zero the upper 4 bytes of the destination register (not meaningful for constmodify versions).
   364  		{name: "ADDQ", argLength: 2, reg: gp21sp, asm: "ADDQ", commutative: true, clobberFlags: true, earlyOk: true},
   365  		{name: "ADDL", argLength: 2, reg: gp21sp, asm: "ADDL", commutative: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   366  		{name: "ADDQconst", argLength: 1, reg: gp11sp, asm: "ADDQ", aux: "Int32", typ: "UInt64", clobberFlags: true, earlyOk: true},
   367  		{name: "ADDLconst", argLength: 1, reg: gp11sp, asm: "ADDL", aux: "Int32", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   368  		{name: "ADDQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   369  		{name: "ADDLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   370  		{name: "ADDWconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   371  		{name: "ADDBconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   372  
   373  		{name: "SUBQ", argLength: 2, reg: gp21sp2, asm: "SUBQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   374  		{name: "SUBL", argLength: 2, reg: gp21, asm: "SUBL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   375  		{name: "SUBQconst", argLength: 1, reg: gp11, asm: "SUBQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},
   376  		{name: "SUBLconst", argLength: 1, reg: gp11, asm: "SUBL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   377  
   378  		{name: "MULQ", argLength: 2, reg: gp21, asm: "IMULQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},
   379  		{name: "MULL", argLength: 2, reg: gp21, asm: "IMULL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   380  		{name: "MULQconst", argLength: 1, reg: gp11, asm: "IMUL3Q", aux: "Int32", clobberFlags: true, earlyOk: true},
   381  		{name: "MULLconst", argLength: 1, reg: gp11, asm: "IMUL3L", aux: "Int32", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   382  
   383  		// Let x = arg0*arg1 (full 32x32->64  unsigned multiply). Returns uint32(x), and flags set to overflow if uint32(x) != x.
   384  		{name: "MULLU", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{ax, regMask{}}, clobbers: dx}, typ: "(UInt32,Flags)", asm: "MULL", commutative: true, clobberFlags: true, zeroUpperBits: 32},
   385  		// Let x = arg0*arg1 (full 64x64->128 unsigned multiply). Returns uint64(x), and flags set to overflow if uint64(x) != x.
   386  		{name: "MULQU", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{ax, regMask{}}, clobbers: dx}, typ: "(UInt64,Flags)", asm: "MULQ", commutative: true, clobberFlags: true},
   387  
   388  		// HMULx[U]: computes the high bits of an integer multiply.
   389  		// computes arg0 * arg1 >> (x==L?32:64)
   390  		// The multiply is unsigned for the U versions, signed for the non-U versions.
   391  		// HMULx[U] are intentionally not marked as commutative, even though they are.
   392  		// This is because they have asymmetric register requirements.
   393  		// There are rewrite rules to try to place arguments in preferable slots.
   394  		{name: "HMULQ", argLength: 2, reg: gp21hmul, asm: "IMULQ", clobberFlags: true, earlyOk: true},
   395  		{name: "HMULL", argLength: 2, reg: gp21hmul, asm: "IMULL", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   396  		{name: "HMULQU", argLength: 2, reg: gp21hmul, asm: "MULQ", clobberFlags: true, earlyOk: true},
   397  		{name: "HMULLU", argLength: 2, reg: gp21hmul, asm: "MULL", clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   398  
   399  		// (arg0 + arg1) / 2 as unsigned, all 64 result bits
   400  		{name: "AVGQU", argLength: 2, reg: gp21, commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},
   401  
   402  		// DIVx[U] computes [arg0 / arg1, arg0 % arg1]
   403  		// For signed versions, AuxInt non-zero means that the divisor has been proved to be not -1.
   404  		{name: "DIVQ", argLength: 2, reg: gp11div, typ: "(Int64,Int64)", asm: "IDIVQ", aux: "Bool", clobberFlags: true},
   405  		{name: "DIVL", argLength: 2, reg: gp11div, typ: "(Int32,Int32)", asm: "IDIVL", aux: "Bool", clobberFlags: true, zeroUpperBits: 32},
   406  		{name: "DIVW", argLength: 2, reg: gp11div, typ: "(Int16,Int16)", asm: "IDIVW", aux: "Bool", clobberFlags: true},
   407  		{name: "DIVQU", argLength: 2, reg: gp11div, typ: "(UInt64,UInt64)", asm: "DIVQ", clobberFlags: true},
   408  		{name: "DIVLU", argLength: 2, reg: gp11div, typ: "(UInt32,UInt32)", asm: "DIVL", clobberFlags: true, zeroUpperBits: 32},
   409  		{name: "DIVWU", argLength: 2, reg: gp11div, typ: "(UInt16,UInt16)", asm: "DIVW", clobberFlags: true},
   410  
   411  		// computes -arg0, flags set for 0-arg0.
   412  		{name: "NEGLflags", argLength: 1, reg: gp11flags, typ: "(UInt32,Flags)", asm: "NEGL", resultInArg0: true, zeroUpperBits: 32},
   413  		// compute arg0+auxint. flags set for arg0+auxint.
   414  		// NOTE: we pretend the CF/OF flags are undefined for these instructions,
   415  		// so we can use INC/DEC instead of ADDQconst if auxint is +/-1. (INC/DEC don't modify CF.)
   416  		{name: "ADDQconstflags", argLength: 1, reg: gp11flags, aux: "Int32", asm: "ADDQ", resultInArg0: true},
   417  		{name: "ADDLconstflags", argLength: 1, reg: gp11flags, aux: "Int32", asm: "ADDL", resultInArg0: true, zeroUpperBits: 32},
   418  
   419  		// The following 4 add opcodes return the low 64 bits of the sum in the first result and
   420  		// the carry (the 65th bit) in the carry flag.
   421  		{name: "ADDQcarry", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "ADDQ", commutative: true, resultInArg0: true}, // r = arg0+arg1
   422  		{name: "ADCQ", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "ADCQ", commutative: true, resultInArg0: true}, // r = arg0+arg1+carry(arg2)
   423  		{name: "ADDQconstcarry", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "ADDQ", aux: "Int32", resultInArg0: true}, // r = arg0+auxint
   424  		{name: "ADCQconst", argLength: 2, reg: gp1flags1flags, typ: "(UInt64,Flags)", asm: "ADCQ", aux: "Int32", resultInArg0: true}, // r = arg0+auxint+carry(arg1)
   425  
   426  		// The following 4 add opcodes return the low 64 bits of the difference in the first result and
   427  		// the borrow (if the result is negative) in the carry flag.
   428  		{name: "SUBQborrow", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "SUBQ", resultInArg0: true},                    // r = arg0-arg1
   429  		{name: "SBBQ", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "SBBQ", resultInArg0: true},                     // r = arg0-(arg1+carry(arg2))
   430  		{name: "SUBQconstborrow", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "SUBQ", aux: "Int32", resultInArg0: true}, // r = arg0-auxint
   431  		{name: "SBBQconst", argLength: 2, reg: gp1flags1flags, typ: "(UInt64,Flags)", asm: "SBBQ", aux: "Int32", resultInArg0: true},  // r = arg0-(auxint+carry(arg1))
   432  
   433  		{name: "MULQU2", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{dx, ax}}, commutative: true, asm: "MULQ", clobberFlags: true, earlyOk: true}, // arg0 * arg1, returns (hi, lo)
   434  		// MULXQ is the BMI2 unsigned 64x64->128 multiply. arg0 must be in DX
   435  		// (the implicit operand); arg1 is any register or memory. Outputs are
   436  		// (hi, lo) and may be placed in any general-purpose registers (they
   437  		// must be different from each other; the assembler/register allocator
   438  		// arranges this). Unlike MULQ, MULXQ does not affect the flags, which
   439  		// makes it interleavable with ADCX/ADOX carry chains.
   440  		{name: "MULXQ", argLength: 2, reg: regInfo{inputs: []regMask{dx, gpsp}, outputs: []regMask{gp, gp}}, commutative: true, asm: "MULXQ"},      // arg0 * arg1, returns (hi, lo); does not affect flags
   441  		{name: "DIVQU2", argLength: 3, reg: regInfo{inputs: []regMask{dx, ax, gpsp}, outputs: []regMask{ax, dx}}, asm: "DIVQ", clobberFlags: true}, // arg0:arg1 / arg2 (128-bit divided by 64-bit), returns (q, r)
   442  
   443  		{name: "ANDQ", argLength: 2, reg: gp21, asm: "ANDQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & arg1
   444  		{name: "ANDL", argLength: 2, reg: gp21, asm: "ANDL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 & arg1
   445  		{name: "ANDQconst", argLength: 1, reg: gp11, asm: "ANDQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & auxint
   446  		{name: "ANDLconst", argLength: 1, reg: gp11, asm: "ANDL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 & auxint
   447  		{name: "ANDQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   448  		{name: "ANDLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   449  		{name: "ANDWconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   450  		{name: "ANDBconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   451  
   452  		{name: "ORQ", argLength: 2, reg: gp21, asm: "ORQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | arg1
   453  		{name: "ORL", argLength: 2, reg: gp21, asm: "ORL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 | arg1
   454  		{name: "ORQconst", argLength: 1, reg: gp11, asm: "ORQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | auxint
   455  		{name: "ORLconst", argLength: 1, reg: gp11, asm: "ORL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 | auxint
   456  		{name: "ORQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   457  		{name: "ORLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   458  		{name: "ORWconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   459  		{name: "ORBconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   460  
   461  		{name: "XORQ", argLength: 2, reg: gp21, asm: "XORQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ arg1
   462  		{name: "XORL", argLength: 2, reg: gp21, asm: "XORL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 ^ arg1
   463  		{name: "XORQconst", argLength: 1, reg: gp11, asm: "XORQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ auxint
   464  		{name: "XORLconst", argLength: 1, reg: gp11, asm: "XORL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},                                   // arg0 ^ auxint
   465  		{name: "XORQconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   466  		{name: "XORLconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   467  		{name: "XORWconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORW", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   468  		{name: "XORBconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORB", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   469  
   470  		// CMPx: compare arg0 to arg1.
   471  		{name: "CMPQ", argLength: 2, reg: gp2flags, asm: "CMPQ", typ: "Flags"},
   472  		{name: "CMPL", argLength: 2, reg: gp2flags, asm: "CMPL", typ: "Flags"},
   473  		{name: "CMPW", argLength: 2, reg: gp2flags, asm: "CMPW", typ: "Flags"},
   474  		{name: "CMPB", argLength: 2, reg: gp2flags, asm: "CMPB", typ: "Flags"},
   475  
   476  		// CMPxconst: compare arg0 to auxint.
   477  		{name: "CMPQconst", argLength: 1, reg: gp1flags, asm: "CMPQ", typ: "Flags", aux: "Int32"},
   478  		{name: "CMPLconst", argLength: 1, reg: gp1flags, asm: "CMPL", typ: "Flags", aux: "Int32"},
   479  		{name: "CMPWconst", argLength: 1, reg: gp1flags, asm: "CMPW", typ: "Flags", aux: "Int16"},
   480  		{name: "CMPBconst", argLength: 1, reg: gp1flags, asm: "CMPB", typ: "Flags", aux: "Int8"},
   481  
   482  		// CMPxload: compare *(arg0+auxint+aux) to arg1 (in that order). arg2=mem.
   483  		{name: "CMPQload", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   484  		{name: "CMPLload", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   485  		{name: "CMPWload", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   486  		{name: "CMPBload", argLength: 3, reg: gp1flagsLoad, asm: "CMPB", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   487  
   488  		// CMPxconstload: compare *(arg0+ValAndOff(AuxInt).Off()+aux) to ValAndOff(AuxInt).Val() (in that order). arg1=mem.
   489  		{name: "CMPQconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPQ", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   490  		{name: "CMPLconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPL", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   491  		{name: "CMPWconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPW", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   492  		{name: "CMPBconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPB", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   493  
   494  		// CMPxloadidx: compare *(arg0+N*arg1+auxint+aux) to arg2 (in that order). arg3=mem.
   495  		{name: "CMPQloadidx8", argLength: 4, reg: gp2flagsLoad, asm: "CMPQ", scale: 8, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   496  		{name: "CMPQloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPQ", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   497  		{name: "CMPLloadidx4", argLength: 4, reg: gp2flagsLoad, asm: "CMPL", scale: 4, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   498  		{name: "CMPLloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPL", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   499  		{name: "CMPWloadidx2", argLength: 4, reg: gp2flagsLoad, asm: "CMPW", scale: 2, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   500  		{name: "CMPWloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPW", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   501  		{name: "CMPBloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPB", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   502  
   503  		// CMPxconstloadidx: compare *(arg0+N*arg1+ValAndOff(AuxInt).Off()+aux) to ValAndOff(AuxInt).Val() (in that order). arg2=mem.
   504  		{name: "CMPQconstloadidx8", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", scale: 8, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   505  		{name: "CMPQconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   506  		{name: "CMPLconstloadidx4", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", scale: 4, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   507  		{name: "CMPLconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   508  		{name: "CMPWconstloadidx2", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", scale: 2, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   509  		{name: "CMPWconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   510  		{name: "CMPBconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPB", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   511  
   512  		// UCOMISx: floating-point compare arg0 to arg1
   513  		// x==S for float32, x==D for float64
   514  		{name: "UCOMISS", argLength: 2, reg: fp2flags, asm: "UCOMISS", typ: "Flags"},
   515  		{name: "UCOMISD", argLength: 2, reg: fp2flags, asm: "UCOMISD", typ: "Flags"},
   516  
   517  		// bit test/set/clear operations
   518  		{name: "BTL", argLength: 2, reg: gp2flags, asm: "BTL", typ: "Flags"},                                                           // test whether bit arg0%32 in arg1 is set
   519  		{name: "BTQ", argLength: 2, reg: gp2flags, asm: "BTQ", typ: "Flags"},                                                           // test whether bit arg0%64 in arg1 is set
   520  		{name: "BTCL", argLength: 2, reg: gp21, asm: "BTCL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32}, // complement bit arg1%32 in arg0
   521  		{name: "BTCQ", argLength: 2, reg: gp21, asm: "BTCQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                    // complement bit arg1%64 in arg0
   522  		{name: "BTRL", argLength: 2, reg: gp21, asm: "BTRL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32}, // reset bit arg1%32 in arg0
   523  		{name: "BTRQ", argLength: 2, reg: gp21, asm: "BTRQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                    // reset bit arg1%64 in arg0
   524  		{name: "BTSL", argLength: 2, reg: gp21, asm: "BTSL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32}, // set bit arg1%32 in arg0
   525  		{name: "BTSQ", argLength: 2, reg: gp21, asm: "BTSQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                    // set bit arg1%64 in arg0
   526  		{name: "BTLconst", argLength: 1, reg: gp1flags, asm: "BTL", typ: "Flags", aux: "Int8", earlyOk: true},                          // test whether bit auxint in arg0 is set, 0 <= auxint < 32
   527  		{name: "BTQconst", argLength: 1, reg: gp1flags, asm: "BTQ", typ: "Flags", aux: "Int8", earlyOk: true},                          // test whether bit auxint in arg0 is set, 0 <= auxint < 64
   528  		{name: "BTCQconst", argLength: 1, reg: gp11, asm: "BTCQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true},  // complement bit auxint in arg0, 31 <= auxint < 64
   529  		{name: "BTRQconst", argLength: 1, reg: gp11, asm: "BTRQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true},  // reset bit auxint in arg0, 31 <= auxint < 64
   530  		{name: "BTSQconst", argLength: 1, reg: gp11, asm: "BTSQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true},  // set bit auxint in arg0, 31 <= auxint < 64
   531  
   532  		// BT[SRC]Qconstmodify
   533  		//
   534  		//  S: set bit
   535  		//  R: reset (clear) bit
   536  		//  C: complement bit
   537  		//
   538  		// Apply operation to bit ValAndOff(AuxInt).Val() in the 64 bits at
   539  		// memory address arg0+ValAndOff(AuxInt).Off()+aux
   540  		// Bit index must be in range (31-63).
   541  		// (We use OR/AND/XOR for thinner targets and lower bit indexes.)
   542  		// arg1=mem, returns mem
   543  		//
   544  		// Note that there aren't non-const versions of these instructions.
   545  		// Well, there are such instructions, but they are slow and weird so we don't use them.
   546  		{name: "BTSQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTSQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   547  		{name: "BTRQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTRQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   548  		{name: "BTCQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTCQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   549  
   550  		// TESTx: compare (arg0 & arg1) to 0
   551  		{name: "TESTQ", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTQ", typ: "Flags"},
   552  		{name: "TESTL", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTL", typ: "Flags"},
   553  		{name: "TESTW", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTW", typ: "Flags"},
   554  		{name: "TESTB", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTB", typ: "Flags"},
   555  
   556  		// TESTxconst: compare (arg0 & auxint) to 0
   557  		{name: "TESTQconst", argLength: 1, reg: gp1flags, asm: "TESTQ", typ: "Flags", aux: "Int32"},
   558  		{name: "TESTLconst", argLength: 1, reg: gp1flags, asm: "TESTL", typ: "Flags", aux: "Int32"},
   559  		{name: "TESTWconst", argLength: 1, reg: gp1flags, asm: "TESTW", typ: "Flags", aux: "Int16"},
   560  		{name: "TESTBconst", argLength: 1, reg: gp1flags, asm: "TESTB", typ: "Flags", aux: "Int8"},
   561  
   562  		// S{HL, HR, AR}x: shift operations
   563  		// SHL: shift left
   564  		// SHR: shift right logical (0s are shifted in from beyond the word size)
   565  		// SAR: shift right arithmetic (sign bit is shifted in from beyond the word size)
   566  		// arg0 is the value being shifted
   567  		// arg1 is the amount to shift, interpreted mod (Q=64,L=32,W=32,B=32)
   568  		// (Note: x86 is weird, the 16 and 8 byte shifts still use all 5 bits of shift amount!)
   569  		// For *const versions, use auxint instead of arg1 as the shift amount. auxint must be in the range 0 to (Q=63,L=31,W=15,B=7) inclusive.
   570  		{name: "SHLQ", argLength: 2, reg: gp21shift, asm: "SHLQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   571  		{name: "SHLL", argLength: 2, reg: gp21shift, asm: "SHLL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   572  		{name: "SHLQconst", argLength: 1, reg: gp11, asm: "SHLQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   573  		{name: "SHLLconst", argLength: 1, reg: gp11, asm: "SHLL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   574  
   575  		{name: "SHRQ", argLength: 2, reg: gp21shift, asm: "SHRQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   576  		{name: "SHRL", argLength: 2, reg: gp21shift, asm: "SHRL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   577  		{name: "SHRW", argLength: 2, reg: gp21shift, asm: "SHRW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   578  		{name: "SHRB", argLength: 2, reg: gp21shift, asm: "SHRB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   579  		{name: "SHRQconst", argLength: 1, reg: gp11, asm: "SHRQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   580  		{name: "SHRLconst", argLength: 1, reg: gp11, asm: "SHRL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   581  		{name: "SHRWconst", argLength: 1, reg: gp11, asm: "SHRW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   582  		{name: "SHRBconst", argLength: 1, reg: gp11, asm: "SHRB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   583  
   584  		{name: "SARQ", argLength: 2, reg: gp21shift, asm: "SARQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   585  		{name: "SARL", argLength: 2, reg: gp21shift, asm: "SARL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   586  		{name: "SARW", argLength: 2, reg: gp21shift, asm: "SARW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   587  		{name: "SARB", argLength: 2, reg: gp21shift, asm: "SARB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   588  		{name: "SARQconst", argLength: 1, reg: gp11, asm: "SARQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   589  		{name: "SARLconst", argLength: 1, reg: gp11, asm: "SARL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   590  		{name: "SARWconst", argLength: 1, reg: gp11, asm: "SARW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   591  		{name: "SARBconst", argLength: 1, reg: gp11, asm: "SARB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   592  
   593  		// RO{L,R}x: rotate instructions
   594  		// computes arg0 rotate (L=left,R=right) arg1 bits.
   595  		// Bits are rotated within the low (Q=64,L=32,W=16,B=8) bits of the register.
   596  		// For *const versions use auxint instead of arg1 as the rotate amount. auxint must be in the range 0 to (Q=63,L=31,W=15,B=7) inclusive.
   597  		// x==L versions zero the upper 32 bits of the destination register.
   598  		// x==W and x==B versions leave the upper bits unspecified.
   599  		{name: "ROLQ", argLength: 2, reg: gp21shift, asm: "ROLQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   600  		{name: "ROLL", argLength: 2, reg: gp21shift, asm: "ROLL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   601  		{name: "ROLW", argLength: 2, reg: gp21shift, asm: "ROLW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   602  		{name: "ROLB", argLength: 2, reg: gp21shift, asm: "ROLB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   603  		{name: "RORQ", argLength: 2, reg: gp21shift, asm: "RORQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   604  		{name: "RORL", argLength: 2, reg: gp21shift, asm: "RORL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   605  		{name: "RORW", argLength: 2, reg: gp21shift, asm: "RORW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   606  		{name: "RORB", argLength: 2, reg: gp21shift, asm: "RORB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   607  		{name: "ROLQconst", argLength: 1, reg: gp11, asm: "ROLQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   608  		{name: "ROLLconst", argLength: 1, reg: gp11, asm: "ROLL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   609  		{name: "ROLWconst", argLength: 1, reg: gp11, asm: "ROLW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   610  		{name: "ROLBconst", argLength: 1, reg: gp11, asm: "ROLB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   611  
   612  		// [ADD,SUB,AND,OR]xload: integer load/op combo
   613  		// L = int32, Q = int64
   614  		// x==L operations zero the upper 4 bytes of the destination register.
   615  		// computes arg0 op *(arg1+auxint+aux), arg2=mem
   616  		{name: "ADDLload", argLength: 3, reg: gp21load, asm: "ADDL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   617  		{name: "ADDQload", argLength: 3, reg: gp21load, asm: "ADDQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   618  		{name: "SUBQload", argLength: 3, reg: gp21load, asm: "SUBQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   619  		{name: "SUBLload", argLength: 3, reg: gp21load, asm: "SUBL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   620  		{name: "ANDLload", argLength: 3, reg: gp21load, asm: "ANDL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   621  		{name: "ANDQload", argLength: 3, reg: gp21load, asm: "ANDQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   622  		{name: "ORQload", argLength: 3, reg: gp21load, asm: "ORQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   623  		{name: "ORLload", argLength: 3, reg: gp21load, asm: "ORL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   624  		{name: "XORQload", argLength: 3, reg: gp21load, asm: "XORQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   625  		{name: "XORLload", argLength: 3, reg: gp21load, asm: "XORL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   626  
   627  		// integer indexed load/op combo
   628  		// L = int32, Q = int64
   629  		// L operations zero the upper 4 bytes of the destination register.
   630  		// computes arg0 op *(arg1+scale*arg2+auxint+aux), arg3=mem
   631  		{name: "ADDLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   632  		{name: "ADDLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   633  		{name: "ADDLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   634  		{name: "ADDQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ADDQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   635  		{name: "ADDQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ADDQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   636  		{name: "SUBLloadidx1", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   637  		{name: "SUBLloadidx4", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   638  		{name: "SUBLloadidx8", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   639  		{name: "SUBQloadidx1", argLength: 4, reg: gp21loadidx, asm: "SUBQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   640  		{name: "SUBQloadidx8", argLength: 4, reg: gp21loadidx, asm: "SUBQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   641  		{name: "ANDLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   642  		{name: "ANDLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   643  		{name: "ANDLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   644  		{name: "ANDQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ANDQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   645  		{name: "ANDQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ANDQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   646  		{name: "ORLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   647  		{name: "ORLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   648  		{name: "ORLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   649  		{name: "ORQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ORQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   650  		{name: "ORQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ORQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   651  		{name: "XORLloadidx1", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   652  		{name: "XORLloadidx4", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   653  		{name: "XORLloadidx8", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true, zeroUpperBits: 32},
   654  		{name: "XORQloadidx1", argLength: 4, reg: gp21loadidx, asm: "XORQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   655  		{name: "XORQloadidx8", argLength: 4, reg: gp21loadidx, asm: "XORQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   656  
   657  		// direct binary op on memory (read-modify-write)
   658  		// L = int32, Q = int64
   659  		// does *(arg0+auxint+aux) op= arg1, arg2=mem
   660  		{name: "ADDQmodify", argLength: 3, reg: gpstore, asm: "ADDQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   661  		{name: "SUBQmodify", argLength: 3, reg: gpstore, asm: "SUBQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   662  		{name: "ANDQmodify", argLength: 3, reg: gpstore, asm: "ANDQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   663  		{name: "ORQmodify", argLength: 3, reg: gpstore, asm: "ORQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   664  		{name: "XORQmodify", argLength: 3, reg: gpstore, asm: "XORQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   665  		{name: "ADDLmodify", argLength: 3, reg: gpstore, asm: "ADDL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   666  		{name: "SUBLmodify", argLength: 3, reg: gpstore, asm: "SUBL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   667  		{name: "ANDLmodify", argLength: 3, reg: gpstore, asm: "ANDL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   668  		{name: "ORLmodify", argLength: 3, reg: gpstore, asm: "ORL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   669  		{name: "XORLmodify", argLength: 3, reg: gpstore, asm: "XORL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   670  
   671  		// indexed direct binary op on memory.
   672  		// does *(arg0+scale*arg1+auxint+aux) op= arg2, arg3=mem
   673  		{name: "ADDQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ADDQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   674  		{name: "ADDQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ADDQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   675  		{name: "SUBQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "SUBQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   676  		{name: "SUBQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "SUBQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   677  		{name: "ANDQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ANDQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   678  		{name: "ANDQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ANDQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   679  		{name: "ORQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ORQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   680  		{name: "ORQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ORQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   681  		{name: "XORQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "XORQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   682  		{name: "XORQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "XORQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   683  		{name: "ADDLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   684  		{name: "ADDLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   685  		{name: "ADDLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   686  		{name: "SUBLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   687  		{name: "SUBLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   688  		{name: "SUBLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   689  		{name: "ANDLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   690  		{name: "ANDLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   691  		{name: "ANDLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   692  		{name: "ORLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   693  		{name: "ORLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   694  		{name: "ORLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   695  		{name: "XORLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   696  		{name: "XORLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   697  		{name: "XORLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   698  
   699  		// indexed direct binary op on memory with constant argument.
   700  		// does *(arg0+scale*arg1+ValAndOff(AuxInt).Off()+aux) op= ValAndOff(AuxInt).Val(), arg2=mem
   701  		{name: "ADDQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   702  		{name: "ADDQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ADDQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   703  		{name: "ANDQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   704  		{name: "ANDQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ANDQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   705  		{name: "ORQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   706  		{name: "ORQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ORQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   707  		{name: "XORQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   708  		{name: "XORQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "XORQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   709  		{name: "ADDLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   710  		{name: "ADDLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   711  		{name: "ADDLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   712  		{name: "ADDWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   713  		{name: "ADDWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "ADDW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   714  		{name: "ADDBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   715  		{name: "ANDLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   716  		{name: "ANDLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   717  		{name: "ANDLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   718  		{name: "ANDWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   719  		{name: "ANDWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "ANDW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   720  		{name: "ANDBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   721  		{name: "ORLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   722  		{name: "ORLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   723  		{name: "ORLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   724  		{name: "ORWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   725  		{name: "ORWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "ORW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   726  		{name: "ORBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   727  		{name: "XORLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   728  		{name: "XORLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   729  		{name: "XORLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   730  		{name: "XORWconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORW", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   731  		{name: "XORWconstmodifyidx2", argLength: 3, reg: gpstoreconstidx, asm: "XORW", scale: 2, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   732  		{name: "XORBconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORB", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   733  
   734  		// {NEG,NOT}x: unary ops
   735  		// computes [NEG:-,NOT:^]arg0
   736  		// L = int32, Q = int64
   737  		// L operations zero the upper 4 bytes of the destination register.
   738  		{name: "NEGQ", argLength: 1, reg: gp11, asm: "NEGQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   739  		{name: "NEGL", argLength: 1, reg: gp11, asm: "NEGL", resultInArg0: true, clobberFlags: true, earlyOk: true, zeroUpperBits: 32},
   740  		{name: "NOTQ", argLength: 1, reg: gp11, asm: "NOTQ", resultInArg0: true, earlyOk: true},
   741  		{name: "NOTL", argLength: 1, reg: gp11, asm: "NOTL", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   742  
   743  		// BS{F,R}Q returns a tuple [result, flags]
   744  		// result is undefined if the input is zero.
   745  		// flags are set to "equal" if the input is zero, "not equal" otherwise.
   746  		// BS{F,R}L returns only the result.
   747  		{name: "BSFQ", argLength: 1, reg: gp11flags, asm: "BSFQ", typ: "(UInt64,Flags)", earlyOk: true},        // # of low-order zeroes in 64-bit arg
   748  		{name: "BSFL", argLength: 1, reg: gp11, asm: "BSFL", typ: "UInt32", clobberFlags: true, earlyOk: true}, // # of low-order zeroes in 32-bit arg
   749  		{name: "BSRQ", argLength: 1, reg: gp11flags, asm: "BSRQ", typ: "(UInt64,Flags)", earlyOk: true},        // # of high-order zeroes in 64-bit arg
   750  		{name: "BSRL", argLength: 1, reg: gp11, asm: "BSRL", typ: "UInt32", clobberFlags: true, earlyOk: true}, // # of high-order zeroes in 32-bit arg
   751  
   752  		// CMOV instructions: 64, 32 and 16-bit sizes.
   753  		// if arg2 encodes a true result, return arg1, else arg0
   754  		{name: "CMOVQEQ", argLength: 3, reg: gp21, asm: "CMOVQEQ", resultInArg0: true, earlyOk: true},
   755  		{name: "CMOVQNE", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, earlyOk: true},
   756  		{name: "CMOVQLT", argLength: 3, reg: gp21, asm: "CMOVQLT", resultInArg0: true, earlyOk: true},
   757  		{name: "CMOVQGT", argLength: 3, reg: gp21, asm: "CMOVQGT", resultInArg0: true, earlyOk: true},
   758  		{name: "CMOVQLE", argLength: 3, reg: gp21, asm: "CMOVQLE", resultInArg0: true, earlyOk: true},
   759  		{name: "CMOVQGE", argLength: 3, reg: gp21, asm: "CMOVQGE", resultInArg0: true, earlyOk: true},
   760  		{name: "CMOVQLS", argLength: 3, reg: gp21, asm: "CMOVQLS", resultInArg0: true, earlyOk: true},
   761  		{name: "CMOVQHI", argLength: 3, reg: gp21, asm: "CMOVQHI", resultInArg0: true, earlyOk: true},
   762  		{name: "CMOVQCC", argLength: 3, reg: gp21, asm: "CMOVQCC", resultInArg0: true, earlyOk: true},
   763  		{name: "CMOVQCS", argLength: 3, reg: gp21, asm: "CMOVQCS", resultInArg0: true, earlyOk: true},
   764  
   765  		{name: "CMOVLEQ", argLength: 3, reg: gp21, asm: "CMOVLEQ", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   766  		{name: "CMOVLNE", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   767  		{name: "CMOVLLT", argLength: 3, reg: gp21, asm: "CMOVLLT", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   768  		{name: "CMOVLGT", argLength: 3, reg: gp21, asm: "CMOVLGT", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   769  		{name: "CMOVLLE", argLength: 3, reg: gp21, asm: "CMOVLLE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   770  		{name: "CMOVLGE", argLength: 3, reg: gp21, asm: "CMOVLGE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   771  		{name: "CMOVLLS", argLength: 3, reg: gp21, asm: "CMOVLLS", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   772  		{name: "CMOVLHI", argLength: 3, reg: gp21, asm: "CMOVLHI", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   773  		{name: "CMOVLCC", argLength: 3, reg: gp21, asm: "CMOVLCC", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   774  		{name: "CMOVLCS", argLength: 3, reg: gp21, asm: "CMOVLCS", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   775  
   776  		{name: "CMOVWEQ", argLength: 3, reg: gp21, asm: "CMOVWEQ", resultInArg0: true, earlyOk: true},
   777  		{name: "CMOVWNE", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, earlyOk: true},
   778  		{name: "CMOVWLT", argLength: 3, reg: gp21, asm: "CMOVWLT", resultInArg0: true, earlyOk: true},
   779  		{name: "CMOVWGT", argLength: 3, reg: gp21, asm: "CMOVWGT", resultInArg0: true, earlyOk: true},
   780  		{name: "CMOVWLE", argLength: 3, reg: gp21, asm: "CMOVWLE", resultInArg0: true, earlyOk: true},
   781  		{name: "CMOVWGE", argLength: 3, reg: gp21, asm: "CMOVWGE", resultInArg0: true, earlyOk: true},
   782  		{name: "CMOVWLS", argLength: 3, reg: gp21, asm: "CMOVWLS", resultInArg0: true, earlyOk: true},
   783  		{name: "CMOVWHI", argLength: 3, reg: gp21, asm: "CMOVWHI", resultInArg0: true, earlyOk: true},
   784  		{name: "CMOVWCC", argLength: 3, reg: gp21, asm: "CMOVWCC", resultInArg0: true, earlyOk: true},
   785  		{name: "CMOVWCS", argLength: 3, reg: gp21, asm: "CMOVWCS", resultInArg0: true, earlyOk: true},
   786  
   787  		// CMOV with floating point instructions. We need separate pseudo-op to handle
   788  		// InvertFlags correctly, and to generate special code that handles NaN (unordered flag).
   789  		// NOTE: the fact that CMOV*EQF here is marked to generate CMOV*NE is not a bug. See
   790  		// code generation in amd64/ssa.go.
   791  		{name: "CMOVQEQF", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, needIntTemp: true, earlyOk: true},
   792  		{name: "CMOVQNEF", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, earlyOk: true},
   793  		{name: "CMOVQGTF", argLength: 3, reg: gp21, asm: "CMOVQHI", resultInArg0: true, earlyOk: true},
   794  		{name: "CMOVQGEF", argLength: 3, reg: gp21, asm: "CMOVQCC", resultInArg0: true, earlyOk: true},
   795  		{name: "CMOVLEQF", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, needIntTemp: true, earlyOk: true, zeroUpperBits: 32},
   796  		{name: "CMOVLNEF", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   797  		{name: "CMOVLGTF", argLength: 3, reg: gp21, asm: "CMOVLHI", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   798  		{name: "CMOVLGEF", argLength: 3, reg: gp21, asm: "CMOVLCC", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   799  		{name: "CMOVWEQF", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, needIntTemp: true, earlyOk: true},
   800  		{name: "CMOVWNEF", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, earlyOk: true},
   801  		{name: "CMOVWGTF", argLength: 3, reg: gp21, asm: "CMOVWHI", resultInArg0: true, earlyOk: true},
   802  		{name: "CMOVWGEF", argLength: 3, reg: gp21, asm: "CMOVWCC", resultInArg0: true, earlyOk: true},
   803  
   804  		// BSWAPx swaps the low-order (L=4,Q=8) bytes of arg0.
   805  		// Q: abcdefgh -> hgfedcba
   806  		// L: abcdefgh -> 0000hgfe (L zeros the upper 4 bytes)
   807  		{name: "BSWAPQ", argLength: 1, reg: gp11, asm: "BSWAPQ", resultInArg0: true, earlyOk: true},
   808  		{name: "BSWAPL", argLength: 1, reg: gp11, asm: "BSWAPL", resultInArg0: true, earlyOk: true, zeroUpperBits: 32},
   809  
   810  		// POPCNTx counts the number of set bits in the low-order (L=32,Q=64) bits of arg0.
   811  		// POPCNTx instructions are only guaranteed to be available if GOAMD64>=v2.
   812  		// For GOAMD64<v2, any use must be preceded by a successful runtime check of runtime.x86HasPOPCNT.
   813  		{name: "POPCNTQ", argLength: 1, reg: gp11, asm: "POPCNTQ", clobberFlags: true, zeroUpperBits: 56},
   814  		{name: "POPCNTL", argLength: 1, reg: gp11, asm: "POPCNTL", clobberFlags: true, zeroUpperBits: 56},
   815  
   816  		// SQRTSx computes sqrt(arg0)
   817  		// S = float32, D = float64
   818  		{name: "SQRTSD", argLength: 1, reg: fp11, asm: "SQRTSD", earlyOk: true},
   819  		{name: "SQRTSS", argLength: 1, reg: fp11, asm: "SQRTSS", earlyOk: true},
   820  
   821  		// ROUNDSD rounds arg0 to an integer depending on auxint
   822  		// 0 means math.RoundToEven, 1 means math.Floor, 2 math.Ceil, 3 math.Trunc
   823  		// (The result is still a float64.)
   824  		// ROUNDSD instruction is only guaraneteed to be available if GOAMD64>=v2.
   825  		// For GOAMD64<v2, any use must be preceded by a successful check of runtime.x86HasSSE41.
   826  		{name: "ROUNDSD", argLength: 1, reg: fp11, aux: "Int8", asm: "ROUNDSD"},
   827  		{name: "ROUNDSS", argLength: 1, reg: fp11, aux: "Int8", asm: "ROUNDSS"},
   828  		// See why we need those in issue #71204
   829  		{name: "LoweredRound32F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   830  		{name: "LoweredRound64F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   831  
   832  		// VFMADD231Sx only exist on platforms with the FMA3 instruction set.
   833  		// Any use must be preceded by a successful check of runtime.x86HasFMA or a check of GOAMD64>=v3.
   834  		// x==S for float32, x==D for float64
   835  		// arg0 + arg1*arg2, with no intermediate rounding.
   836  		{name: "VFMADD231SS", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMADD231SS"},
   837  		{name: "VFMADD231SD", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMADD231SD"},
   838  		// arg1*arg2 - arg0, with no intermediate rounding.
   839  		{name: "VFMSUB231SS", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMSUB231SS"},
   840  		{name: "VFMSUB231SD", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMSUB231SD"},
   841  		// arg0 - arg1*arg2, with no intermediate rounding.
   842  		{name: "VFNMADD231SS", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFNMADD231SS"},
   843  		{name: "VFNMADD231SD", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFNMADD231SD"},
   844  
   845  		// Note that these operations don't exactly match the semantics of Go's
   846  		// builtin min. In particular, these aren't commutative, because on various
   847  		// special cases the 2nd argument is preferred.
   848  		{name: "MINSD", argLength: 2, reg: fp21, resultInArg0: true, asm: "MINSD", earlyOk: true}, // min(arg0,arg1)
   849  		{name: "MINSS", argLength: 2, reg: fp21, resultInArg0: true, asm: "MINSS", earlyOk: true}, // min(arg0,arg1)
   850  
   851  		{name: "SBBQcarrymask", argLength: 1, reg: flagsgp, asm: "SBBQ", earlyOk: true},                    // (int64)(-1) if carry is set, 0 if carry is clear.
   852  		{name: "SBBLcarrymask", argLength: 1, reg: flagsgp, asm: "SBBL", earlyOk: true, zeroUpperBits: 32}, // (int32)(-1) if carry is set, 0 if carry is clear.
   853  		// Note: SBBW and SBBB are subsumed by SBBL
   854  
   855  		{name: "SETEQ", argLength: 1, reg: readflags, asm: "SETEQ", earlyOk: true}, // extract == condition from arg0
   856  		{name: "SETNE", argLength: 1, reg: readflags, asm: "SETNE", earlyOk: true}, // extract != condition from arg0
   857  		{name: "SETL", argLength: 1, reg: readflags, asm: "SETLT", earlyOk: true},  // extract signed < condition from arg0
   858  		{name: "SETLE", argLength: 1, reg: readflags, asm: "SETLE", earlyOk: true}, // extract signed <= condition from arg0
   859  		{name: "SETG", argLength: 1, reg: readflags, asm: "SETGT", earlyOk: true},  // extract signed > condition from arg0
   860  		{name: "SETGE", argLength: 1, reg: readflags, asm: "SETGE", earlyOk: true}, // extract signed >= condition from arg0
   861  		{name: "SETB", argLength: 1, reg: readflags, asm: "SETCS", earlyOk: true},  // extract unsigned < condition from arg0
   862  		{name: "SETBE", argLength: 1, reg: readflags, asm: "SETLS", earlyOk: true}, // extract unsigned <= condition from arg0
   863  		{name: "SETA", argLength: 1, reg: readflags, asm: "SETHI", earlyOk: true},  // extract unsigned > condition from arg0
   864  		{name: "SETAE", argLength: 1, reg: readflags, asm: "SETCC", earlyOk: true}, // extract unsigned >= condition from arg0
   865  		{name: "SETO", argLength: 1, reg: readflags, asm: "SETOS", earlyOk: true},  // extract if overflow flag is set from arg0
   866  		// Variants that store result to memory
   867  		{name: "SETEQstore", argLength: 3, reg: gpstoreconst, asm: "SETEQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract == condition from arg1 to arg0+auxint+aux, arg2=mem
   868  		{name: "SETNEstore", argLength: 3, reg: gpstoreconst, asm: "SETNE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract != condition from arg1 to arg0+auxint+aux, arg2=mem
   869  		{name: "SETLstore", argLength: 3, reg: gpstoreconst, asm: "SETLT", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract signed < condition from arg1 to arg0+auxint+aux, arg2=mem
   870  		{name: "SETLEstore", argLength: 3, reg: gpstoreconst, asm: "SETLE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract signed <= condition from arg1 to arg0+auxint+aux, arg2=mem
   871  		{name: "SETGstore", argLength: 3, reg: gpstoreconst, asm: "SETGT", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract signed > condition from arg1 to arg0+auxint+aux, arg2=mem
   872  		{name: "SETGEstore", argLength: 3, reg: gpstoreconst, asm: "SETGE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract signed >= condition from arg1 to arg0+auxint+aux, arg2=mem
   873  		{name: "SETBstore", argLength: 3, reg: gpstoreconst, asm: "SETCS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract unsigned < condition from arg1 to arg0+auxint+aux, arg2=mem
   874  		{name: "SETBEstore", argLength: 3, reg: gpstoreconst, asm: "SETLS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract unsigned <= condition from arg1 to arg0+auxint+aux, arg2=mem
   875  		{name: "SETAstore", argLength: 3, reg: gpstoreconst, asm: "SETHI", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract unsigned > condition from arg1 to arg0+auxint+aux, arg2=mem
   876  		{name: "SETAEstore", argLength: 3, reg: gpstoreconst, asm: "SETCC", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract unsigned >= condition from arg1 to arg0+auxint+aux, arg2=mem
   877  		{name: "SETEQstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETEQ", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract == condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   878  		{name: "SETNEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETNE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract != condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   879  		{name: "SETLstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLT", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract signed < condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   880  		{name: "SETLEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract signed <= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   881  		{name: "SETGstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETGT", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract signed > condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   882  		{name: "SETGEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETGE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract signed >= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   883  		{name: "SETBstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETCS", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract unsigned < condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   884  		{name: "SETBEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLS", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract unsigned <= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   885  		{name: "SETAstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETHI", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract unsigned > condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   886  		{name: "SETAEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETCC", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract unsigned >= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   887  
   888  		// Need different opcodes for floating point conditions because
   889  		// any comparison involving a NaN is always FALSE and thus
   890  		// the patterns for inverting conditions cannot be used.
   891  		{name: "SETEQF", argLength: 1, reg: flagsgp, asm: "SETEQ", clobberFlags: true, needIntTemp: true, earlyOk: true}, // extract == condition from arg0
   892  		{name: "SETNEF", argLength: 1, reg: flagsgp, asm: "SETNE", clobberFlags: true, needIntTemp: true, earlyOk: true}, // extract != condition from arg0
   893  		{name: "SETORD", argLength: 1, reg: flagsgp, asm: "SETPC", earlyOk: true},                                        // extract "ordered" (No Nan present) condition from arg0
   894  		{name: "SETNAN", argLength: 1, reg: flagsgp, asm: "SETPS", earlyOk: true},                                        // extract "unordered" (Nan present) condition from arg0
   895  
   896  		{name: "SETGF", argLength: 1, reg: flagsgp, asm: "SETHI", earlyOk: true},  // extract floating > condition from arg0
   897  		{name: "SETGEF", argLength: 1, reg: flagsgp, asm: "SETCC", earlyOk: true}, // extract floating >= condition from arg0
   898  
   899  		{name: "MOVBQSX", argLength: 1, reg: gp11, asm: "MOVBQSX", earlyOk: true},                    // sign extend arg0 from int8 to int64
   900  		{name: "MOVBQZX", argLength: 1, reg: gp11, asm: "MOVBLZX", earlyOk: true, zeroUpperBits: 56}, // zero extend arg0 from int8 to int64
   901  		{name: "MOVWQSX", argLength: 1, reg: gp11, asm: "MOVWQSX", earlyOk: true},                    // sign extend arg0 from int16 to int64
   902  		{name: "MOVWQZX", argLength: 1, reg: gp11, asm: "MOVWLZX", earlyOk: true, zeroUpperBits: 48}, // zero extend arg0 from int16 to int64
   903  		{name: "MOVLQSX", argLength: 1, reg: gp11, asm: "MOVLQSX", earlyOk: true},                    // sign extend arg0 from int32 to int64
   904  		{name: "MOVLQZX", argLength: 1, reg: gp11, asm: "MOVL", earlyOk: true, zeroUpperBits: 32},    // zero extend arg0 from int32 to int64
   905  
   906  		{name: "MOVLconst", reg: gp01, asm: "MOVL", typ: "UInt32", aux: "Int32", rematerializeable: true, earlyOk: true, zeroUpperBits: 32}, // 32 low bits of auxint (upper 32 are zeroed)
   907  		{name: "MOVQconst", reg: gp01, asm: "MOVQ", typ: "UInt64", aux: "Int64", rematerializeable: true, earlyOk: true},                    // auxint
   908  
   909  		{name: "CVTTSD2SL", argLength: 1, reg: fpgp, asm: "CVTTSD2SL", earlyOk: true, zeroUpperBits: 32}, // convert float64 to int32
   910  		{name: "CVTTSD2SQ", argLength: 1, reg: fpgp, asm: "CVTTSD2SQ", earlyOk: true},                    // convert float64 to int64
   911  		{name: "CVTTSS2SL", argLength: 1, reg: fpgp, asm: "CVTTSS2SL", earlyOk: true, zeroUpperBits: 32}, // convert float32 to int32
   912  		{name: "CVTTSS2SQ", argLength: 1, reg: fpgp, asm: "CVTTSS2SQ", earlyOk: true},                    // convert float32 to int64
   913  		{name: "CVTSL2SS", argLength: 1, reg: gpfp, asm: "CVTSL2SS", earlyOk: true},                      // convert int32 to float32
   914  		{name: "CVTSL2SD", argLength: 1, reg: gpfp, asm: "CVTSL2SD", earlyOk: true},                      // convert int32 to float64
   915  		{name: "CVTSQ2SS", argLength: 1, reg: gpfp, asm: "CVTSQ2SS", earlyOk: true},                      // convert int64 to float32
   916  		{name: "CVTSQ2SD", argLength: 1, reg: gpfp, asm: "CVTSQ2SD", earlyOk: true},                      // convert int64 to float64
   917  		{name: "CVTSD2SS", argLength: 1, reg: fp11, asm: "CVTSD2SS", earlyOk: true},                      // convert float64 to float32
   918  		{name: "CVTSS2SD", argLength: 1, reg: fp11, asm: "CVTSS2SD", earlyOk: true},                      // convert float32 to float64
   919  
   920  		// Move values between int and float registers, with no conversion.
   921  		// TODO: should we have generic versions of these?
   922  		{name: "MOVQi2f", argLength: 1, reg: gpfp, typ: "Float64", earlyOk: true},                   // move 64 bits from int to float reg
   923  		{name: "MOVQf2i", argLength: 1, reg: fpgp, typ: "UInt64", earlyOk: true},                    // move 64 bits from float to int reg
   924  		{name: "MOVLi2f", argLength: 1, reg: gpfp, typ: "Float32", earlyOk: true},                   // move 32 bits from int to float reg
   925  		{name: "MOVLf2i", argLength: 1, reg: fpgp, typ: "UInt32", earlyOk: true, zeroUpperBits: 32}, // move 32 bits from float to int reg, zero extend
   926  
   927  		{name: "PXOR", argLength: 2, reg: fp21, asm: "PXOR", commutative: true, resultInArg0: true, earlyOk: true}, // exclusive or, applied to X regs (for float negation).
   928  		{name: "POR", argLength: 2, reg: fp21, asm: "POR", commutative: true, resultInArg0: true, earlyOk: true},   // inclusive or, applied to X regs (for float min/max).
   929  
   930  		{name: "LEAQ", argLength: 1, reg: gp11sb, asm: "LEAQ", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true},                    // arg0 + auxint + offset encoded in aux
   931  		{name: "LEAL", argLength: 1, reg: gp11sb, asm: "LEAL", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true, zeroUpperBits: 32}, // arg0 + auxint + offset encoded in aux
   932  		{name: "LEAW", argLength: 1, reg: gp11sb, asm: "LEAW", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true},                    // arg0 + auxint + offset encoded in aux
   933  
   934  		// LEAxn computes arg0 + n*arg1 + auxint + aux
   935  		// x==L zeroes the upper 4 bytes.
   936  		{name: "LEAQ1", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + arg1 + auxint + aux
   937  		{name: "LEAL1", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32}, // arg0 + arg1 + auxint + aux
   938  		{name: "LEAW1", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + arg1 + auxint + aux
   939  		{name: "LEAQ2", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 2*arg1 + auxint + aux
   940  		{name: "LEAL2", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32},                    // arg0 + 2*arg1 + auxint + aux
   941  		{name: "LEAW2", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 2*arg1 + auxint + aux
   942  		{name: "LEAQ4", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 4*arg1 + auxint + aux
   943  		{name: "LEAL4", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32},                    // arg0 + 4*arg1 + auxint + aux
   944  		{name: "LEAW4", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 4*arg1 + auxint + aux
   945  		{name: "LEAQ8", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 8*arg1 + auxint + aux
   946  		{name: "LEAL8", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true, zeroUpperBits: 32},                    // arg0 + 8*arg1 + auxint + aux
   947  		{name: "LEAW8", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true},                                       // arg0 + 8*arg1 + auxint + aux
   948  		// Note: LEAx{1,2,4,8} must not have OpSB as either argument.
   949  
   950  		// MOVxload: loads
   951  		// Load (Q=8,L=4,W=2,B=1) bytes from (arg0+auxint+aux), arg1=mem.
   952  		// "+auxint+aux" == add auxint and the offset of the symbol in aux (if any) to the effective address
   953  		// Standard versions zero extend the result. SX versions sign extend the result.
   954  		{name: "MOVBload", argLength: 2, reg: gpload, asm: "MOVBLZX", aux: "SymOff", typ: "UInt8", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 56},
   955  		{name: "MOVBQSXload", argLength: 2, reg: gpload, asm: "MOVBQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   956  		{name: "MOVWload", argLength: 2, reg: gpload, asm: "MOVWLZX", aux: "SymOff", typ: "UInt16", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 48},
   957  		{name: "MOVWQSXload", argLength: 2, reg: gpload, asm: "MOVWQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   958  		{name: "MOVLload", argLength: 2, reg: gpload, asm: "MOVL", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},
   959  		{name: "MOVLQSXload", argLength: 2, reg: gpload, asm: "MOVLQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   960  		{name: "MOVQload", argLength: 2, reg: gpload, asm: "MOVQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   961  
   962  		// MOVxstore: stores
   963  		// Store (Q=8,L=4,W=2,B=1) low bytes of arg1.
   964  		// Does *(arg0+auxint+aux) = arg1, arg2=mem.
   965  		{name: "MOVBstore", argLength: 3, reg: gpstore, asm: "MOVB", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   966  		{name: "MOVWstore", argLength: 3, reg: gpstore, asm: "MOVW", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   967  		{name: "MOVLstore", argLength: 3, reg: gpstore, asm: "MOVL", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   968  		{name: "MOVQstore", argLength: 3, reg: gpstore, asm: "MOVQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   969  
   970  		// MOVOload/store: 16 byte load/store
   971  		// These operations are only used to move data around: there is no *O arithmetic, for example.
   972  		{name: "MOVOload", argLength: 2, reg: fpload, asm: "MOVUPS", aux: "SymOff", typ: "Int128", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // load 16 bytes from arg0+auxint+aux. arg1=mem
   973  		{name: "MOVOstore", argLength: 3, reg: fpstore, asm: "MOVUPS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 16 bytes in arg1 to arg0+auxint+aux. arg2=mem
   974  
   975  		// MOVxloadidx: indexed loads
   976  		// load (Q=8,L=4,W=2,B=1) bytes from (arg0+scale*arg1+auxint+aux), arg2=mem.
   977  		// Results are zero-extended. (TODO: sign-extending indexed loads)
   978  		{name: "MOVBloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBLZX", scale: 1, aux: "SymOff", typ: "UInt8", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 56},
   979  		{name: "MOVWloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVWLZX", scale: 1, aux: "SymOff", typ: "UInt16", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 48},
   980  		{name: "MOVWloadidx2", argLength: 3, reg: gploadidx, asm: "MOVWLZX", scale: 2, aux: "SymOff", typ: "UInt16", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 48},
   981  		{name: "MOVLloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 32},
   982  		{name: "MOVLloadidx4", argLength: 3, reg: gploadidx, asm: "MOVL", scale: 4, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},
   983  		{name: "MOVLloadidx8", argLength: 3, reg: gploadidx, asm: "MOVL", scale: 8, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},
   984  		{name: "MOVQloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   985  		{name: "MOVQloadidx8", argLength: 3, reg: gploadidx, asm: "MOVQ", scale: 8, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true},
   986  
   987  		// MOVxstoreidx: indexed stores
   988  		// Store (Q=8,L=4,W=2,B=1) low bytes of arg2.
   989  		// Does *(arg0+scale*arg1+auxint+aux) = arg2, arg3=mem.
   990  		{name: "MOVBstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVB", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   991  		{name: "MOVWstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVW", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   992  		{name: "MOVWstoreidx2", argLength: 4, reg: gpstoreidx, asm: "MOVW", scale: 2, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   993  		{name: "MOVLstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   994  		{name: "MOVLstoreidx4", argLength: 4, reg: gpstoreidx, asm: "MOVL", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   995  		{name: "MOVLstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVL", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   996  		{name: "MOVQstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   997  		{name: "MOVQstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVQ", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   998  
   999  		// TODO: add size-mismatched indexed loads/stores, like MOVBstoreidx4?
  1000  
  1001  		// MOVxstoreconst: constant stores
  1002  		// Store (O=16,Q=8,L=4,W=2,B=1) constant bytes.
  1003  		// Does *(arg0+ValAndOff(AuxInt).Off()+aux) = ValAndOff(AuxInt).Val(), arg1=mem.
  1004  		// O version can only store the constant 0.
  1005  		{name: "MOVBstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVB", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1006  		{name: "MOVWstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVW", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1007  		{name: "MOVLstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVL", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1008  		{name: "MOVQstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVQ", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1009  		{name: "MOVOstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVUPS", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
  1010  
  1011  		// MOVxstoreconstidx: constant indexed stores
  1012  		// Store (Q=8,L=4,W=2,B=1) constant bytes.
  1013  		// Does *(arg0+scale*arg1+ValAndOff(AuxInt).Off()+aux) = ValAndOff(AuxInt).Val(), arg2=mem.
  1014  		{name: "MOVBstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVB", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1015  		{name: "MOVWstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVW", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1016  		{name: "MOVWstoreconstidx2", argLength: 3, reg: gpstoreconstidx, asm: "MOVW", scale: 2, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
  1017  		{name: "MOVLstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1018  		{name: "MOVLstoreconstidx4", argLength: 3, reg: gpstoreconstidx, asm: "MOVL", scale: 4, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
  1019  		{name: "MOVQstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
  1020  		{name: "MOVQstoreconstidx8", argLength: 3, reg: gpstoreconstidx, asm: "MOVQ", scale: 8, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
  1021  
  1022  		// arg0 = pointer to start of memory to zero
  1023  		// arg1 = mem
  1024  		// auxint = # of bytes to zero
  1025  		// returns mem
  1026  		{
  1027  			name:      "LoweredZero",
  1028  			aux:       "Int64",
  1029  			argLength: 2,
  1030  			reg: regInfo{
  1031  				inputs: []regMask{gp},
  1032  			},
  1033  			faultOnNilArg0: true,
  1034  			addrSinkArg0:   true,
  1035  		},
  1036  
  1037  		// arg0 = pointer to start of memory to zero
  1038  		// arg1 = mem
  1039  		// auxint = # of bytes to zero
  1040  		// returns mem
  1041  		{
  1042  			name:      "LoweredZeroLoop",
  1043  			aux:       "Int64",
  1044  			argLength: 2,
  1045  			reg: regInfo{
  1046  				inputs:       []regMask{gp},
  1047  				clobbersArg0: true,
  1048  			},
  1049  			clobberFlags:   true,
  1050  			faultOnNilArg0: true,
  1051  			addrSinkArg0:   true,
  1052  			needIntTemp:    true,
  1053  		},
  1054  
  1055  		// arg0 = address of memory to zero
  1056  		// arg1 = # of 8-byte words to zero
  1057  		// arg2 = value to store (will always be zero)
  1058  		// arg3 = mem
  1059  		// returns mem
  1060  		{
  1061  			name:      "REPSTOSQ",
  1062  			argLength: 4,
  1063  			reg: regInfo{
  1064  				inputs:   []regMask{buildReg("DI"), buildReg("CX"), buildReg("AX")},
  1065  				clobbers: buildReg("DI CX"),
  1066  			},
  1067  			faultOnNilArg0: true,
  1068  			addrSinkArg0:   true,
  1069  		},
  1070  
  1071  		// With a register ABI, the actual register info for these instructions (i.e., what is used in regalloc) is augmented with per-call-site bindings of additional arguments to specific in and out registers.
  1072  		{name: "CALLstatic", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                                      // call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
  1073  		{name: "CALLtail", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},                                        // tail call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
  1074  		{name: "CALLtailinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},            // tail call fn by pointer. arg0=codeptr, last arg=mem, auxint=argsize, returns mem
  1075  		{name: "CALLclosure", argLength: -1, reg: regInfo{inputs: []regMask{gpsp, buildReg("DX"), regMask{}}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true}, // call function via closure.  arg0=codeptr, arg1=closure, last arg=mem, auxint=argsize, returns mem
  1076  		{name: "CALLinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                // call fn by pointer.  arg0=codeptr, last arg=mem, auxint=argsize, returns mem
  1077  
  1078  		// arg0 = destination pointer
  1079  		// arg1 = source pointer
  1080  		// arg2 = mem
  1081  		// auxint = # of bytes to copy
  1082  		// returns memory
  1083  		{
  1084  			name:      "LoweredMove",
  1085  			aux:       "Int64",
  1086  			argLength: 3,
  1087  			reg: regInfo{
  1088  				inputs:   []regMask{gp, gp},
  1089  				clobbers: buildReg("X14"), // uses X14 as a temporary
  1090  			},
  1091  			faultOnNilArg0: true,
  1092  			faultOnNilArg1: true,
  1093  			addrSinkArg0:   true,
  1094  			addrSinkArg1:   true,
  1095  		},
  1096  		// arg0 = destination pointer
  1097  		// arg1 = source pointer
  1098  		// arg2 = mem
  1099  		// auxint = # of bytes to copy
  1100  		// returns memory
  1101  		{
  1102  			name:      "LoweredMoveLoop",
  1103  			aux:       "Int64",
  1104  			argLength: 3,
  1105  			reg: regInfo{
  1106  				inputs:       []regMask{gp, gp},
  1107  				clobbers:     buildReg("X14"), // uses X14 as a temporary
  1108  				clobbersArg0: true,
  1109  				clobbersArg1: true,
  1110  			},
  1111  			clobberFlags:   true,
  1112  			faultOnNilArg0: true,
  1113  			faultOnNilArg1: true,
  1114  			addrSinkArg0:   true,
  1115  			addrSinkArg1:   true,
  1116  			needIntTemp:    true,
  1117  		},
  1118  
  1119  		// arg0 = destination pointer
  1120  		// arg1 = source pointer
  1121  		// arg2 = # of 8-byte words to copy
  1122  		// arg3 = mem
  1123  		// returns memory
  1124  		{
  1125  			name:      "REPMOVSQ",
  1126  			argLength: 4,
  1127  			reg: regInfo{
  1128  				inputs:   []regMask{buildReg("DI"), buildReg("SI"), buildReg("CX")},
  1129  				clobbers: buildReg("DI SI CX"),
  1130  			},
  1131  			faultOnNilArg0: true,
  1132  			faultOnNilArg1: true,
  1133  			addrSinkArg0:   true,
  1134  			addrSinkArg1:   true,
  1135  		},
  1136  
  1137  		// (InvertFlags (CMPQ a b)) == (CMPQ b a)
  1138  		// So if we want (SETL (CMPQ a b)) but we can't do that because a is a constant,
  1139  		// then we do (SETL (InvertFlags (CMPQ b a))) instead.
  1140  		// Rewrites will convert this to (SETG (CMPQ b a)).
  1141  		// InvertFlags is a pseudo-op which can't appear in assembly output.
  1142  		{name: "InvertFlags", argLength: 1}, // reverse direction of arg0
  1143  
  1144  		// Pseudo-ops
  1145  		{name: "LoweredGetG", argLength: 1, reg: gp01}, // arg0=mem
  1146  		// Scheduler ensures LoweredGetClosurePtr occurs only in entry block,
  1147  		// and sorts it to the very beginning of the block to prevent other
  1148  		// use of DX (the closure pointer)
  1149  		{name: "LoweredGetClosurePtr", reg: regInfo{outputs: []regMask{buildReg("DX")}}, zeroWidth: true},
  1150  		// LoweredGetCallerPC evaluates to the PC to which its "caller" will return.
  1151  		// I.e., if f calls g "calls" sys.GetCallerPC,
  1152  		// the result should be the PC within f that g will return to.
  1153  		// See runtime/stubs.go for a more detailed discussion.
  1154  		{name: "LoweredGetCallerPC", reg: gp01, rematerializeable: true},
  1155  		// LoweredGetCallerSP returns the SP of the caller of the current function. arg0=mem
  1156  		{name: "LoweredGetCallerSP", argLength: 1, reg: gp01, rematerializeable: true},
  1157  		//arg0=ptr,arg1=mem, returns void.  Faults if ptr is nil.
  1158  		{name: "LoweredNilCheck", argLength: 2, reg: regInfo{inputs: []regMask{gpsp}}, clobberFlags: true, nilCheck: true, faultOnNilArg0: true},
  1159  		// LoweredWB invokes runtime.gcWriteBarrier{auxint}. arg0=mem, auxint=# of buffer entries needed.
  1160  		// It saves all GP registers if necessary, but may clobber others.
  1161  		// Returns a pointer to a write barrier buffer in R11.
  1162  		{name: "LoweredWB", argLength: 1, reg: regInfo{clobbers: callerSave.minus(gp.union(g)), outputs: []regMask{buildReg("R11")}}, clobberFlags: true, aux: "Int64"},
  1163  
  1164  		{name: "LoweredHasCPUFeature", argLength: 0, reg: gp01, rematerializeable: true, typ: "UInt64", aux: "Sym", symEffect: "None", zeroUpperBits: 56},
  1165  
  1166  		// LoweredPanicBoundsRR takes x and y, two values that caused a bounds check to fail.
  1167  		// the RC and CR versions are used when one of the arguments is a constant. CC is used
  1168  		// when both are constant (normally both 0, as prove derives the fact that a [0] bounds
  1169  		// failure means the length must have also been 0).
  1170  		// AuxInt contains a report code (see PanicBounds in genericOps.go).
  1171  		{name: "LoweredPanicBoundsRR", argLength: 3, aux: "Int64", reg: regInfo{inputs: []regMask{gp, gp}}, typ: "Mem", call: true},    // arg0=x, arg1=y, arg2=mem, returns memory.
  1172  		{name: "LoweredPanicBoundsRC", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{gp}}, typ: "Mem", call: true}, // arg0=x, arg1=mem, returns memory.
  1173  		{name: "LoweredPanicBoundsCR", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{gp}}, typ: "Mem", call: true}, // arg0=y, arg1=mem, returns memory.
  1174  		{name: "LoweredPanicBoundsCC", argLength: 1, aux: "PanicBoundsCC", reg: regInfo{}, typ: "Mem", call: true},                     // arg0=mem, returns memory.
  1175  
  1176  		// Constant flag values. For any comparison, there are 5 possible
  1177  		// outcomes: the three from the signed total order (<,==,>) and the
  1178  		// three from the unsigned total order. The == cases overlap.
  1179  		// Note: there's a sixth "unordered" outcome for floating-point
  1180  		// comparisons, but we don't use such a beast yet.
  1181  		// These ops are for temporary use by rewrite rules. They
  1182  		// cannot appear in the generated assembly.
  1183  		{name: "FlagEQ"},     // equal
  1184  		{name: "FlagLT_ULT"}, // signed < and unsigned <
  1185  		{name: "FlagLT_UGT"}, // signed < and unsigned >
  1186  		{name: "FlagGT_UGT"}, // signed > and unsigned >
  1187  		{name: "FlagGT_ULT"}, // signed > and unsigned <
  1188  
  1189  		// Atomic loads.  These are just normal loads but return <value,memory> tuples
  1190  		// so they can be properly ordered with other loads.
  1191  		// load from arg0+auxint+aux.  arg1=mem.
  1192  		{name: "MOVBatomicload", argLength: 2, reg: gpload, asm: "MOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1193  		{name: "MOVLatomicload", argLength: 2, reg: gpload, asm: "MOVL", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", zeroUpperBits: 32},
  1194  		{name: "MOVQatomicload", argLength: 2, reg: gpload, asm: "MOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1195  
  1196  		// Atomic stores and exchanges.  Stores use XCHG to get the right memory ordering semantics.
  1197  		// store arg0 to arg1+auxint+aux, arg2=mem.
  1198  		// These ops return a tuple of <old contents of *(arg1+auxint+aux), memory>.
  1199  		// Note: arg0 and arg1 are backwards compared to MOVLstore (to facilitate resultInArg0)!
  1200  		{name: "XCHGB", argLength: 3, reg: gpstorexchg, asm: "XCHGB", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1201  		{name: "XCHGL", argLength: 3, reg: gpstorexchg, asm: "XCHGL", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr", zeroUpperBits: 32},
  1202  		{name: "XCHGQ", argLength: 3, reg: gpstorexchg, asm: "XCHGQ", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1203  
  1204  		// Atomic adds.
  1205  		// *(arg1+auxint+aux) += arg0.  arg2=mem.
  1206  		// Returns a tuple of <old contents of *(arg1+auxint+aux), memory>.
  1207  		// Note: arg0 and arg1 are backwards compared to MOVLstore (to facilitate resultInArg0)!
  1208  		{name: "XADDLlock", argLength: 3, reg: gpstorexchg, asm: "XADDL", typ: "(UInt32,Mem)", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr", zeroUpperBits: 32},
  1209  		{name: "XADDQlock", argLength: 3, reg: gpstorexchg, asm: "XADDQ", typ: "(UInt64,Mem)", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1210  		{name: "AddTupleFirst32", argLength: 2}, // arg1=tuple <x,y>.  Returns <x+arg0,y>.
  1211  		{name: "AddTupleFirst64", argLength: 2}, // arg1=tuple <x,y>.  Returns <x+arg0,y>.
  1212  
  1213  		// Compare and swap.
  1214  		// arg0 = pointer, arg1 = old value, arg2 = new value, arg3 = memory.
  1215  		// if *(arg0+auxint+aux) == arg1 {
  1216  		//   *(arg0+auxint+aux) = arg2
  1217  		//   return (true, memory)
  1218  		// } else {
  1219  		//   return (false, memory)
  1220  		// }
  1221  		// Note that these instructions also return the old value in AX, but we ignore it.
  1222  		// TODO: have these return flags instead of bool.  The current system generates:
  1223  		//    CMPXCHGQ ...
  1224  		//    SETEQ AX
  1225  		//    CMPB  AX, $0
  1226  		//    JNE ...
  1227  		// instead of just
  1228  		//    CMPXCHGQ ...
  1229  		//    JEQ ...
  1230  		// but we can't do that because memory-using ops can't generate flags yet
  1231  		// (flagalloc wants to move flag-generating instructions around).
  1232  		{name: "CMPXCHGLlock", argLength: 4, reg: cmpxchg, asm: "CMPXCHGL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1233  		{name: "CMPXCHGQlock", argLength: 4, reg: cmpxchg, asm: "CMPXCHGQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1234  
  1235  		// Atomic memory updates using logical operations.
  1236  		// Old style that just returns the memory state.
  1237  		{name: "ANDBlock", argLength: 3, reg: gpstore, asm: "ANDB", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1238  		{name: "ANDLlock", argLength: 3, reg: gpstore, asm: "ANDL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1239  		{name: "ANDQlock", argLength: 3, reg: gpstore, asm: "ANDQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1240  		{name: "ORBlock", argLength: 3, reg: gpstore, asm: "ORB", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1241  		{name: "ORLlock", argLength: 3, reg: gpstore, asm: "ORL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1242  		{name: "ORQlock", argLength: 3, reg: gpstore, asm: "ORQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1243  
  1244  		// Atomic memory updates using logical operations.
  1245  		// *(arg0+auxint+aux) op= arg1. arg2=mem.
  1246  		// New style that returns a tuple of <old contents of *(arg0+auxint+aux), memory>.
  1247  		{name: "LoweredAtomicAnd64", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ANDQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1248  		{name: "LoweredAtomicAnd32", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ANDL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true, zeroUpperBits: 32},
  1249  		{name: "LoweredAtomicOr64", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ORQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1250  		{name: "LoweredAtomicOr32", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ORL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true, zeroUpperBits: 32},
  1251  
  1252  		// Prefetch instructions
  1253  		// Do prefetch arg0 address. arg0=addr, arg1=memory. Instruction variant selects locality hint
  1254  		{name: "PrefetchT0", argLength: 2, reg: prefreg, asm: "PREFETCHT0", hasSideEffects: true},
  1255  		{name: "PrefetchNTA", argLength: 2, reg: prefreg, asm: "PREFETCHNTA", hasSideEffects: true},
  1256  
  1257  		// CPUID feature: BMI1.
  1258  		{name: "ANDNQ", argLength: 2, reg: gp21, asm: "ANDNQ", clobberFlags: true},                            // arg0 &^ arg1
  1259  		{name: "ANDNL", argLength: 2, reg: gp21, asm: "ANDNL", clobberFlags: true, zeroUpperBits: 32},         // arg0 &^ arg1
  1260  		{name: "BLSIQ", argLength: 1, reg: gp11, asm: "BLSIQ", clobberFlags: true},                            // arg0 & -arg0
  1261  		{name: "BLSIL", argLength: 1, reg: gp11, asm: "BLSIL", clobberFlags: true, zeroUpperBits: 32},         // arg0 & -arg0
  1262  		{name: "BLSMSKQ", argLength: 1, reg: gp11, asm: "BLSMSKQ", clobberFlags: true},                        // arg0 ^ (arg0 - 1)
  1263  		{name: "BLSMSKL", argLength: 1, reg: gp11, asm: "BLSMSKL", clobberFlags: true, zeroUpperBits: 32},     // arg0 ^ (arg0 - 1)
  1264  		{name: "BLSRQ", argLength: 1, reg: gp11flags, asm: "BLSRQ", typ: "(UInt64,Flags)"},                    // arg0 & (arg0 - 1)
  1265  		{name: "BLSRL", argLength: 1, reg: gp11flags, asm: "BLSRL", typ: "(UInt32,Flags)", zeroUpperBits: 32}, // arg0 & (arg0 - 1)
  1266  		// count the number of trailing zero bits, prefer TZCNTQ over BSFQ, as TZCNTQ(0)==64
  1267  		// and BSFQ(0) is undefined. Same for TZCNTL(0)==32
  1268  		//
  1269  		// TZCNT/LZCNT deliberately carry no zeroUpperBits: their result is
  1270  		// bounded only as long as every rule producing them stays gated on
  1271  		// GOAMD64 >= 3. On older parts their REP BSF/BSR encodings silently
  1272  		// decode as legacy BSF/BSR, which leave the destination unmodified
  1273  		// on zero input — a guarantee too easy to break silently.
  1274  		{name: "TZCNTQ", argLength: 1, reg: gp11, asm: "TZCNTQ", clobberFlags: true},
  1275  		{name: "TZCNTL", argLength: 1, reg: gp11, asm: "TZCNTL", clobberFlags: true},
  1276  
  1277  		// CPUID feature: LZCNT.
  1278  		// count the number of leading zero bits.
  1279  		{name: "LZCNTQ", argLength: 1, reg: gp11, asm: "LZCNTQ", typ: "UInt64", clobberFlags: true},
  1280  		{name: "LZCNTL", argLength: 1, reg: gp11, asm: "LZCNTL", typ: "UInt32", clobberFlags: true},
  1281  
  1282  		// CPUID feature: MOVBE
  1283  		// MOVBEWload does not satisfy zero extended, so only use MOVBEWstore
  1284  		{name: "MOVBEWstore", argLength: 3, reg: gpstore, asm: "MOVBEW", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                    // swap and store 2 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1285  		{name: "MOVBELload", argLength: 2, reg: gpload, asm: "MOVBEL", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // load and swap 4 bytes from arg0+auxint+aux. arg1=mem.  Zero extend.
  1286  		{name: "MOVBELstore", argLength: 3, reg: gpstore, asm: "MOVBEL", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                    // swap and store 4 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1287  		{name: "MOVBEQload", argLength: 2, reg: gpload, asm: "MOVBEQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // load and swap 8 bytes from arg0+auxint+aux. arg1=mem
  1288  		{name: "MOVBEQstore", argLength: 3, reg: gpstore, asm: "MOVBEQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                    // swap and store 8 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1289  		// indexed MOVBE loads
  1290  		{name: "MOVBELloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBEL", scale: 1, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true, zeroUpperBits: 32}, // load and swap 4 bytes from arg0+arg1+auxint+aux. arg2=mem. Zero extend.
  1291  		{name: "MOVBELloadidx4", argLength: 3, reg: gploadidx, asm: "MOVBEL", scale: 4, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},                                        // load and swap 4 bytes from arg0+4*arg1+auxint+aux. arg2=mem. Zero extend.
  1292  		{name: "MOVBELloadidx8", argLength: 3, reg: gploadidx, asm: "MOVBEL", scale: 8, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32},                                        // load and swap 4 bytes from arg0+8*arg1+auxint+aux. arg2=mem. Zero extend.
  1293  		{name: "MOVBEQloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBEQ", scale: 1, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},                    // load and swap 8 bytes from arg0+arg1+auxint+aux. arg2=mem
  1294  		{name: "MOVBEQloadidx8", argLength: 3, reg: gploadidx, asm: "MOVBEQ", scale: 8, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true},                                                           // load and swap 8 bytes from arg0+8*arg1+auxint+aux. arg2=mem
  1295  		// indexed MOVBE stores
  1296  		{name: "MOVBEWstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEW", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 2 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1297  		{name: "MOVBEWstoreidx2", argLength: 4, reg: gpstoreidx, asm: "MOVBEW", scale: 2, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 2 bytes in arg2 to arg0+2*arg1+auxint+aux. arg3=mem
  1298  		{name: "MOVBELstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEL", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 4 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1299  		{name: "MOVBELstoreidx4", argLength: 4, reg: gpstoreidx, asm: "MOVBEL", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 4 bytes in arg2 to arg0+4*arg1+auxint+aux. arg3=mem
  1300  		{name: "MOVBELstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVBEL", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 4 bytes in arg2 to arg0+8*arg1+auxint+aux. arg3=mem
  1301  		{name: "MOVBEQstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEQ", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 8 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1302  		{name: "MOVBEQstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVBEQ", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 8 bytes in arg2 to arg0+8*arg1+auxint+aux. arg3=mem
  1303  
  1304  		// CPUID feature: BMI2.
  1305  		{name: "SARXQ", argLength: 2, reg: gp21, asm: "SARXQ"},                    // signed arg0 >> arg1, shift amount is mod 64
  1306  		{name: "SARXL", argLength: 2, reg: gp21, asm: "SARXL", zeroUpperBits: 32}, // signed int32(arg0) >> arg1, shift amount is mod 32
  1307  		{name: "SHLXQ", argLength: 2, reg: gp21, asm: "SHLXQ"},                    // arg0 << arg1, shift amount is mod 64
  1308  		{name: "SHLXL", argLength: 2, reg: gp21, asm: "SHLXL", zeroUpperBits: 32}, // arg0 << arg1, shift amount is mod 32
  1309  		{name: "SHRXQ", argLength: 2, reg: gp21, asm: "SHRXQ"},                    // unsigned arg0 >> arg1, shift amount is mod 64
  1310  		{name: "SHRXL", argLength: 2, reg: gp21, asm: "SHRXL", zeroUpperBits: 32}, // unsigned uint32(arg0) >> arg1, shift amount is mod 32
  1311  
  1312  		{name: "SARXLload", argLength: 3, reg: gp21shxload, asm: "SARXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 32
  1313  		{name: "SARXQload", argLength: 3, reg: gp21shxload, asm: "SARXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // signed *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 64
  1314  		{name: "SHLXLload", argLength: 3, reg: gp21shxload, asm: "SHLXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+auxint+aux) << arg1, arg2=mem, shift amount is mod 32
  1315  		{name: "SHLXQload", argLength: 3, reg: gp21shxload, asm: "SHLXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // *(arg0+auxint+aux) << arg1, arg2=mem, shift amount is mod 64
  1316  		{name: "SHRXLload", argLength: 3, reg: gp21shxload, asm: "SHRXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 32
  1317  		{name: "SHRXQload", argLength: 3, reg: gp21shxload, asm: "SHRXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // unsigned *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 64
  1318  
  1319  		{name: "SARXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1320  		{name: "SARXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+4*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1321  		{name: "SARXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // signed *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1322  		{name: "SARXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SARXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // signed *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1323  		{name: "SARXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SARXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // signed *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1324  		{name: "SHLXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+1*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1325  		{name: "SHLXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+4*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1326  		{name: "SHLXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // *(arg0+8*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1327  		{name: "SHLXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHLXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // *(arg0+1*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 64
  1328  		{name: "SHLXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHLXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // *(arg0+8*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 64
  1329  		{name: "SHRXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1330  		{name: "SHRXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+4*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1331  		{name: "SHRXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true, zeroUpperBits: 32}, // unsigned *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1332  		{name: "SHRXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHRXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // unsigned *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1333  		{name: "SHRXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHRXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},                    // unsigned *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1334  
  1335  		// Unpack bytes, low 64-bits.
  1336  		//
  1337  		// Input/output registers treated as [8]uint8.
  1338  		//
  1339  		// output = {in1[0], in2[0], in1[1], in2[1], in1[2], in2[2], in1[3], in2[3]}
  1340  		{name: "PUNPCKLBW", argLength: 2, reg: fp21, resultInArg0: true, asm: "PUNPCKLBW"},
  1341  
  1342  		// Shuffle 16-bit words, low 64-bits.
  1343  		//
  1344  		// Input/output registers treated as [4]uint16.
  1345  		// aux=source word index for each destination word, 2 bits per index.
  1346  		//
  1347  		// output[i] = input[(aux>>2*i)&3].
  1348  		{name: "PSHUFLW", argLength: 1, reg: fp11, aux: "Int8", asm: "PSHUFLW"},
  1349  
  1350  		// Broadcast input byte.
  1351  		//
  1352  		// Input treated as uint8, output treated as [16]uint8.
  1353  		//
  1354  		// output[i] = input.
  1355  		{name: "PSHUFBbroadcast", argLength: 1, reg: fp11, resultInArg0: true, asm: "PSHUFB"}, // PSHUFB with mask zero, (GOAMD64=v1)
  1356  		{name: "VPBROADCASTB", argLength: 1, reg: gpfp, asm: "VPBROADCASTB"},                  // Broadcast input byte from gp (GOAMD64=v3)
  1357  
  1358  		// Byte negate/zero/preserve (GOAMD64=v2).
  1359  		//
  1360  		// Input/output registers treated as [16]uint8.
  1361  		//
  1362  		// if in2[i] > 0 {
  1363  		//   output[i] = in1[i]
  1364  		// } else if in2[i] == 0 {
  1365  		//   output[i] = 0
  1366  		// } else {
  1367  		//   output[i] = -1 * in1[i]
  1368  		// }
  1369  		{name: "PSIGNB", argLength: 2, reg: fp21, resultInArg0: true, asm: "PSIGNB"},
  1370  
  1371  		// Byte compare.
  1372  		//
  1373  		// Input/output registers treated as [16]uint8.
  1374  		//
  1375  		// if in1[i] == in2[i] {
  1376  		//   output[i] = 0xff
  1377  		// } else {
  1378  		//   output[i] = 0
  1379  		// }
  1380  		{name: "PCMPEQB", argLength: 2, reg: fp21, resultInArg0: true, asm: "PCMPEQB", commutative: true},
  1381  
  1382  		// Byte sign mask. Output is a bitmap of sign bits from each input byte.
  1383  		//
  1384  		// Input treated as [16]uint8. Output is [16]bit (uint16 bitmap).
  1385  		//
  1386  		// output[i] = (input[i] >> 7) & 1
  1387  		{name: "PMOVMSKB", argLength: 1, reg: fpgp, asm: "PMOVMSKB", zeroUpperBits: 48},
  1388  
  1389  		// SIMD ops
  1390  		{name: "VMOVDQUload128", argLength: 2, reg: vload, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1391  		{name: "VMOVDQUstore128", argLength: 3, reg: vstore, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1392  
  1393  		{name: "VMOVDQUload256", argLength: 2, reg: vload, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1394  		{name: "VMOVDQUstore256", argLength: 3, reg: vstore, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1395  
  1396  		{name: "VMOVDQUload512", argLength: 2, reg: wload, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1397  		{name: "VMOVDQUstore512", argLength: 3, reg: wstore, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1398  
  1399  		// AVX2 32 and 64-bit element int-vector masked moves.
  1400  		{name: "VPMASK32load128", argLength: 3, reg: vloadv, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1401  		{name: "VPMASK32store128", argLength: 4, reg: vstorev, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1402  		{name: "VPMASK64load128", argLength: 3, reg: vloadv, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1403  		{name: "VPMASK64store128", argLength: 4, reg: vstorev, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1404  
  1405  		{name: "VPMASK32load256", argLength: 3, reg: vloadv, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1406  		{name: "VPMASK32store256", argLength: 4, reg: vstorev, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1407  		{name: "VPMASK64load256", argLength: 3, reg: vloadv, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1408  		{name: "VPMASK64store256", argLength: 4, reg: vstorev, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1409  
  1410  		// AVX512 8-64-bit element mask-register masked moves
  1411  		{name: "VPMASK8load512", argLength: 3, reg: wloadk, asm: "VMOVDQU8", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},      // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1412  		{name: "VPMASK8store512", argLength: 4, reg: wstorek, asm: "VMOVDQU8", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},   // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1413  		{name: "VPMASK16load512", argLength: 3, reg: wloadk, asm: "VMOVDQU16", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1414  		{name: "VPMASK16store512", argLength: 4, reg: wstorek, asm: "VMOVDQU16", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1415  		{name: "VPMASK32load512", argLength: 3, reg: wloadk, asm: "VMOVDQU32", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1416  		{name: "VPMASK32store512", argLength: 4, reg: wstorek, asm: "VMOVDQU32", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1417  		{name: "VPMASK64load512", argLength: 3, reg: wloadk, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1418  		{name: "VPMASK64store512", argLength: 4, reg: wstorek, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1419  
  1420  		// AVX512 moves between int-vector and mask registers
  1421  		{name: "VPMOVMToVec8x16", argLength: 1, reg: kv, asm: "VPMOVM2B"},
  1422  		{name: "VPMOVMToVec8x32", argLength: 1, reg: kv, asm: "VPMOVM2B"},
  1423  		{name: "VPMOVMToVec8x64", argLength: 1, reg: kw, asm: "VPMOVM2B"},
  1424  
  1425  		{name: "VPMOVMToVec16x8", argLength: 1, reg: kv, asm: "VPMOVM2W"},
  1426  		{name: "VPMOVMToVec16x16", argLength: 1, reg: kv, asm: "VPMOVM2W"},
  1427  		{name: "VPMOVMToVec16x32", argLength: 1, reg: kw, asm: "VPMOVM2W"},
  1428  
  1429  		{name: "VPMOVMToVec32x4", argLength: 1, reg: kv, asm: "VPMOVM2D"},
  1430  		{name: "VPMOVMToVec32x8", argLength: 1, reg: kv, asm: "VPMOVM2D"},
  1431  		{name: "VPMOVMToVec32x16", argLength: 1, reg: kw, asm: "VPMOVM2D"},
  1432  
  1433  		{name: "VPMOVMToVec64x2", argLength: 1, reg: kv, asm: "VPMOVM2Q"},
  1434  		{name: "VPMOVMToVec64x4", argLength: 1, reg: kv, asm: "VPMOVM2Q"},
  1435  		{name: "VPMOVMToVec64x8", argLength: 1, reg: kw, asm: "VPMOVM2Q"},
  1436  
  1437  		{name: "VPMOVVec8x16ToM", argLength: 1, reg: vk, asm: "VPMOVB2M"},
  1438  		{name: "VPMOVVec8x32ToM", argLength: 1, reg: vk, asm: "VPMOVB2M"},
  1439  		{name: "VPMOVVec8x64ToM", argLength: 1, reg: wk, asm: "VPMOVB2M"},
  1440  
  1441  		{name: "VPMOVVec16x8ToM", argLength: 1, reg: vk, asm: "VPMOVW2M"},
  1442  		{name: "VPMOVVec16x16ToM", argLength: 1, reg: vk, asm: "VPMOVW2M"},
  1443  		{name: "VPMOVVec16x32ToM", argLength: 1, reg: wk, asm: "VPMOVW2M"},
  1444  
  1445  		{name: "VPMOVVec32x4ToM", argLength: 1, reg: vk, asm: "VPMOVD2M"},
  1446  		{name: "VPMOVVec32x8ToM", argLength: 1, reg: vk, asm: "VPMOVD2M"},
  1447  		{name: "VPMOVVec32x16ToM", argLength: 1, reg: wk, asm: "VPMOVD2M"},
  1448  
  1449  		{name: "VPMOVVec64x2ToM", argLength: 1, reg: vk, asm: "VPMOVQ2M"},
  1450  		{name: "VPMOVVec64x4ToM", argLength: 1, reg: vk, asm: "VPMOVQ2M"},
  1451  		{name: "VPMOVVec64x8ToM", argLength: 1, reg: wk, asm: "VPMOVQ2M"},
  1452  
  1453  		// AVX1/2 moves from int-vector to bitmask (extracting sign bits)
  1454  		{name: "VPMOVMSKB128", argLength: 1, reg: vgp, asm: "VPMOVMSKB", zeroUpperBits: 48},
  1455  		{name: "VPMOVMSKB256", argLength: 1, reg: vgp, asm: "VPMOVMSKB", zeroUpperBits: 32},
  1456  		{name: "VMOVMSKPS128", argLength: 1, reg: vgp, asm: "VMOVMSKPS", zeroUpperBits: 56},
  1457  		{name: "VMOVMSKPS256", argLength: 1, reg: vgp, asm: "VMOVMSKPS", zeroUpperBits: 56},
  1458  		{name: "VMOVMSKPD128", argLength: 1, reg: vgp, asm: "VMOVMSKPD", zeroUpperBits: 56},
  1459  		{name: "VMOVMSKPD256", argLength: 1, reg: vgp, asm: "VMOVMSKPD", zeroUpperBits: 56},
  1460  
  1461  		// X15 is the zero register up to 128-bit. For larger values, we zero it on the fly.
  1462  		{name: "Zero128", argLength: 0, reg: x15only, zeroWidth: true, fixedReg: true},
  1463  		{name: "Zero256", argLength: 0, reg: v01, asm: "VPXOR"},
  1464  		{name: "Zero512", argLength: 0, reg: w01, asm: "VPXORQ"},
  1465  
  1466  		// Move a 32/64 bit float to a 128-bit SIMD register.
  1467  		{name: "VMOVSDf2v", argLength: 1, reg: fpv, asm: "VMOVSD"},
  1468  		{name: "VMOVSSf2v", argLength: 1, reg: fpv, asm: "VMOVSS"},
  1469  
  1470  		{name: "VMOVQ", argLength: 1, reg: gpv, asm: "VMOVQ"},
  1471  		{name: "VMOVD", argLength: 1, reg: gpv, asm: "VMOVD"},
  1472  
  1473  		{name: "VMOVQload", argLength: 2, reg: fpload, asm: "VMOVQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read"},
  1474  		{name: "VMOVDload", argLength: 2, reg: fpload, asm: "VMOVD", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read"},
  1475  		{name: "VMOVSSload", argLength: 2, reg: fpload, asm: "VMOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1476  		{name: "VMOVSDload", argLength: 2, reg: fpload, asm: "VMOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1477  
  1478  		{name: "VMOVSSconst", reg: fp01, asm: "VMOVSS", aux: "Float32", rematerializeable: true},
  1479  		{name: "VMOVSDconst", reg: fp01, asm: "VMOVSD", aux: "Float64", rematerializeable: true},
  1480  
  1481  		{name: "VZEROUPPER", argLength: 1, reg: regInfo{clobbers: v}, asm: "VZEROUPPER"}, // arg=mem, returns mem
  1482  		{name: "VZEROALL", argLength: 1, reg: regInfo{clobbers: v}, asm: "VZEROALL"},     // arg=mem, returns mem
  1483  
  1484  		// KMOVxload: loads masks
  1485  		// Load (Q=8,D=4,W=2,B=1) bytes from (arg0+auxint+aux), arg1=mem.
  1486  		// "+auxint+aux" == add auxint and the offset of the symbol in aux (if any) to the effective address
  1487  		{name: "KMOVBload", argLength: 2, reg: kload, asm: "KMOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1488  		{name: "KMOVWload", argLength: 2, reg: kload, asm: "KMOVW", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1489  		{name: "KMOVDload", argLength: 2, reg: kload, asm: "KMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1490  		{name: "KMOVQload", argLength: 2, reg: kload, asm: "KMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1491  
  1492  		// KMOVxstore: stores masks
  1493  		// Store (Q=8,D=4,W=2,B=1) low bytes of arg1.
  1494  		// Does *(arg0+auxint+aux) = arg1, arg2=mem.
  1495  		{name: "KMOVBstore", argLength: 3, reg: kstore, asm: "KMOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1496  		{name: "KMOVWstore", argLength: 3, reg: kstore, asm: "KMOVW", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1497  		{name: "KMOVDstore", argLength: 3, reg: kstore, asm: "KMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1498  		{name: "KMOVQstore", argLength: 3, reg: kstore, asm: "KMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1499  
  1500  		// Move GP directly to mask register
  1501  		{name: "KMOVQk", argLength: 1, reg: gpk, asm: "KMOVQ"},
  1502  		{name: "KMOVDk", argLength: 1, reg: gpk, asm: "KMOVD"},
  1503  		{name: "KMOVWk", argLength: 1, reg: gpk, asm: "KMOVW"},
  1504  		{name: "KMOVBk", argLength: 1, reg: gpk, asm: "KMOVB"},
  1505  		{name: "KMOVQi", argLength: 1, reg: kgp, asm: "KMOVQ"},
  1506  		{name: "KMOVDi", argLength: 1, reg: kgp, asm: "KMOVD", zeroUpperBits: 32},
  1507  		{name: "KMOVWi", argLength: 1, reg: kgp, asm: "KMOVW", zeroUpperBits: 48},
  1508  		{name: "KMOVBi", argLength: 1, reg: kgp, asm: "KMOVB", zeroUpperBits: 56},
  1509  
  1510  		// Mask logical operations
  1511  		{name: "KANDB", argLength: 2, reg: k2k, asm: "KANDB", typ: "Mask"},
  1512  		{name: "KANDW", argLength: 2, reg: k2k, asm: "KANDW", typ: "Mask"},
  1513  		{name: "KANDD", argLength: 2, reg: k2k, asm: "KANDD", typ: "Mask"},
  1514  		{name: "KANDQ", argLength: 2, reg: k2k, asm: "KANDQ", typ: "Mask"},
  1515  
  1516  		{name: "KORB", argLength: 2, reg: k2k, asm: "KORB", typ: "Mask"},
  1517  		{name: "KORW", argLength: 2, reg: k2k, asm: "KORW", typ: "Mask"},
  1518  		{name: "KORD", argLength: 2, reg: k2k, asm: "KORD", typ: "Mask"},
  1519  		{name: "KORQ", argLength: 2, reg: k2k, asm: "KORQ", typ: "Mask"},
  1520  
  1521  		{name: "KXORB", argLength: 2, reg: k2k, asm: "KXORB", typ: "Mask"},
  1522  		{name: "KXORW", argLength: 2, reg: k2k, asm: "KXORW", typ: "Mask"},
  1523  		{name: "KXORD", argLength: 2, reg: k2k, asm: "KXORD", typ: "Mask"},
  1524  		{name: "KXORQ", argLength: 2, reg: k2k, asm: "KXORQ", typ: "Mask"},
  1525  
  1526  		// Following Intel convention, we call it XNOR instead of EQ.
  1527  		{name: "KXNORB", argLength: 2, reg: k2k, asm: "KXNORB", typ: "Mask"},
  1528  		{name: "KXNORW", argLength: 2, reg: k2k, asm: "KXNORW", typ: "Mask"},
  1529  		{name: "KXNORD", argLength: 2, reg: k2k, asm: "KXNORD", typ: "Mask"},
  1530  		{name: "KXNORQ", argLength: 2, reg: k2k, asm: "KXNORQ", typ: "Mask"},
  1531  
  1532  		// VPTEST
  1533  		{name: "VPTEST", asm: "VPTEST", argLength: 2, reg: v2flags, clobberFlags: true, typ: "Flags"},
  1534  	}
  1535  
  1536  	AMD64blocks := []blockData{
  1537  		{name: "EQ", controls: 1},
  1538  		{name: "NE", controls: 1},
  1539  		{name: "LT", controls: 1},
  1540  		{name: "LE", controls: 1},
  1541  		{name: "GT", controls: 1},
  1542  		{name: "GE", controls: 1},
  1543  		{name: "OS", controls: 1},
  1544  		{name: "OC", controls: 1},
  1545  		{name: "ULT", controls: 1},
  1546  		{name: "ULE", controls: 1},
  1547  		{name: "UGT", controls: 1},
  1548  		{name: "UGE", controls: 1},
  1549  		{name: "EQF", controls: 1},
  1550  		{name: "NEF", controls: 1},
  1551  		{name: "ORD", controls: 1}, // FP, ordered comparison (parity zero)
  1552  		{name: "NAN", controls: 1}, // FP, unordered comparison (parity one)
  1553  
  1554  		// JUMPTABLE implements jump tables.
  1555  		// Aux is the symbol (an *obj.LSym) for the jump table.
  1556  		// control[0] is the index into the jump table.
  1557  		// control[1] is the address of the jump table (the address of the symbol stored in Aux).
  1558  		{name: "JUMPTABLE", controls: 2, aux: "Sym"},
  1559  	}
  1560  
  1561  	archs = append(archs, arch{
  1562  		name:        "AMD64",
  1563  		pkg:         "cmd/internal/obj/x86",
  1564  		genfile:     "../../amd64/ssa.go",
  1565  		genSIMDfile: "../../amd64/simdssa.go",
  1566  		ops: append(AMD64ops, simdAMD64Ops(v11, v21, v2k, vkv, v2kv, v2kk, v31, v3kv, vgpv, vgp, vfpv, vfpkv,
  1567  			w11, w21, w2k, wkw, w2kw, w2kk, w31, w3kw, wgpw, wgp, wfpw, wfpkw, wkwload, v21load, v31load, v11load,
  1568  			w21load, w31load, w2kload, w2kwload, w11load, w3kwload, w2kkload, v31x0AtIn2)...), // AMD64ops,
  1569  		blocks:             AMD64blocks,
  1570  		regnames:           regNamesAMD64,
  1571  		ParamIntRegNames:   "AX BX CX DI SI R8 R9 R10 R11",
  1572  		ParamFloatRegNames: "X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14",
  1573  		gpregmask:          gp,
  1574  		fpregmask:          fp,
  1575  		specialregmask:     mask.union(w.minus(v)),
  1576  		simdregmask:        v,
  1577  		framepointerreg:    int8(num["BP"]),
  1578  		linkreg:            -1, // not used
  1579  	})
  1580  }
  1581  

View as plain text