Source file src/cmd/compile/internal/ssa/_gen/AMD64Ops.go

     1  // Copyright 2015 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package main
     6  
     7  import "strings"
     8  
     9  // Notes:
    10  //  - Integer types live in the low portion of registers. Upper portions are junk.
    11  //  - Boolean types use the low-order byte of a register. 0=false, 1=true.
    12  //    Upper bytes are junk.
    13  //  - Floating-point types live in the low natural slot of an sse2 register.
    14  //    Unused portions are junk.
    15  //  - We do not use AH,BH,CH,DH registers.
    16  //  - When doing sub-register operations, we try to write the whole
    17  //    destination register to avoid a partial-register write.
    18  //  - Unused portions of AuxInt (or the Val portion of ValAndOff) are
    19  //    filled by sign-extending the used portion.  Users of AuxInt which interpret
    20  //    AuxInt as unsigned (e.g. shifts) must be careful.
    21  //  - All SymOff opcodes require their offset to fit in an int32.
    22  
    23  // Suffixes encode the bit width of various instructions.
    24  // Q (quad word) = 64 bit
    25  // L (long word) = 32 bit
    26  // W (word)      = 16 bit
    27  // B (byte)      = 8 bit
    28  // D (double)    = 64 bit float
    29  // S (single)    = 32 bit float
    30  
    31  // copied from ../../amd64/reg.go
    32  var regNamesAMD64 = []string{
    33  	"AX",
    34  	"CX",
    35  	"DX",
    36  	"BX",
    37  	"SP",
    38  	"BP",
    39  	"SI",
    40  	"DI",
    41  	"R8",
    42  	"R9",
    43  	"R10",
    44  	"R11",
    45  	"R12",
    46  	"R13",
    47  	"g", // a.k.a. R14
    48  	"R15",
    49  	"X0",
    50  	"X1",
    51  	"X2",
    52  	"X3",
    53  	"X4",
    54  	"X5",
    55  	"X6",
    56  	"X7",
    57  	"X8",
    58  	"X9",
    59  	"X10",
    60  	"X11",
    61  	"X12",
    62  	"X13",
    63  	"X14",
    64  	"X15", // constant 0 in ABIInternal
    65  	"X16",
    66  	"X17",
    67  	"X18",
    68  	"X19",
    69  	"X20",
    70  	"X21",
    71  	"X22",
    72  	"X23",
    73  	"X24",
    74  	"X25",
    75  	"X26",
    76  	"X27",
    77  	"X28",
    78  	"X29",
    79  	"X30",
    80  	"X31",
    81  
    82  	// TODO: update asyncPreempt for K registers.
    83  	// asyncPreempt also needs to store Z0-Z15 properly.
    84  	"K0",
    85  	"K1",
    86  	"K2",
    87  	"K3",
    88  	"K4",
    89  	"K5",
    90  	"K6",
    91  	"K7",
    92  	// If you add registers, update asyncPreempt in runtime
    93  
    94  	// pseudo-registers
    95  	"SB",
    96  }
    97  
    98  func init() {
    99  	// Make map from reg names to reg integers.
   100  	if len(regNamesAMD64) > 64 {
   101  		panic("too many registers")
   102  	}
   103  	num := map[string]int{}
   104  	for i, name := range regNamesAMD64 {
   105  		num[name] = i
   106  	}
   107  	buildReg := func(s string) regMask {
   108  		m := regMask{}
   109  		for _, r := range strings.Split(s, " ") {
   110  			if n, ok := num[r]; ok {
   111  				m = m.addReg(uint(n))
   112  				continue
   113  			}
   114  			panic("register " + r + " not found")
   115  		}
   116  		return m
   117  	}
   118  
   119  	// Common individual register masks
   120  	var (
   121  		ax         = buildReg("AX")
   122  		cx         = buildReg("CX")
   123  		dx         = buildReg("DX")
   124  		gp         = buildReg("AX CX DX BX BP SI DI R8 R9 R10 R11 R12 R13 R15")
   125  		g          = buildReg("g")
   126  		fp         = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14")
   127  		v          = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14")
   128  		w          = buildReg("X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14 X16 X17 X18 X19 X20 X21 X22 X23 X24 X25 X26 X27 X28 X29 X30 X31")
   129  		x15        = buildReg("X15")
   130  		mask       = buildReg("K1 K2 K3 K4 K5 K6 K7")
   131  		gpsp       = gp.union(buildReg("SP"))
   132  		gpspsb     = gpsp.union(buildReg("SB"))
   133  		gpspsbg    = gpspsb.union(g)
   134  		callerSave = gp.union(w).union(mask).union(g) // runtime.setg (and anything calling it) may clobber g
   135  
   136  		vz = v.union(x15)
   137  		wz = w.union(x15)
   138  		x0 = buildReg("X0")
   139  	)
   140  	// Common slices of register masks
   141  	var (
   142  		gponly   = []regMask{gp}
   143  		fponly   = []regMask{fp}
   144  		vonly    = []regMask{v}
   145  		wonly    = []regMask{w}
   146  		maskonly = []regMask{mask}
   147  		vzonly   = []regMask{vz}
   148  		wzonly   = []regMask{wz}
   149  	)
   150  
   151  	// Common regInfo
   152  	var (
   153  		gp01           = regInfo{inputs: nil, outputs: gponly}
   154  		gp11           = regInfo{inputs: []regMask{gp}, outputs: gponly}
   155  		gp11sp         = regInfo{inputs: []regMask{gpsp}, outputs: gponly}
   156  		gp11sb         = regInfo{inputs: []regMask{gpspsbg}, outputs: gponly}
   157  		gp21           = regInfo{inputs: []regMask{gp, gp}, outputs: gponly}
   158  		gp21sp         = regInfo{inputs: []regMask{gpsp, gp}, outputs: gponly}
   159  		gp21sp2        = regInfo{inputs: []regMask{gp, gpsp}, outputs: gponly}
   160  		gp21sb         = regInfo{inputs: []regMask{gpspsbg, gpsp}, outputs: gponly}
   161  		gp21shift      = regInfo{inputs: []regMask{gp, cx}, outputs: []regMask{gp}}
   162  		gp11div        = regInfo{inputs: []regMask{ax, gpsp.minus(dx)}, outputs: []regMask{ax, dx}}
   163  		gp21hmul       = regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{dx}, clobbers: ax}
   164  		gp21flags      = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp, regMask{}}}
   165  		gp2flags1flags = regInfo{inputs: []regMask{gp, gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   166  
   167  		gp2flags     = regInfo{inputs: []regMask{gpsp, gpsp}}
   168  		gp1flags     = regInfo{inputs: []regMask{gpsp}}
   169  		gp0flagsLoad = regInfo{inputs: []regMask{gpspsbg, regMask{}}}
   170  		gp1flagsLoad = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   171  		gp2flagsLoad = regInfo{inputs: []regMask{gpspsbg, gpsp, gpsp, regMask{}}}
   172  		flagsgp      = regInfo{inputs: nil, outputs: gponly}
   173  
   174  		gp11flags      = regInfo{inputs: []regMask{gp}, outputs: []regMask{gp, regMask{}}}
   175  		gp1flags1flags = regInfo{inputs: []regMask{gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   176  
   177  		readflags = regInfo{inputs: nil, outputs: gponly}
   178  
   179  		gpload         = regInfo{inputs: []regMask{gpspsbg, regMask{}}, outputs: gponly}
   180  		gp21load       = regInfo{inputs: []regMask{gp, gpspsbg, regMask{}}, outputs: gponly}
   181  		gploadidx      = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}, outputs: gponly}
   182  		gp21loadidx    = regInfo{inputs: []regMask{gp, gpspsbg, gpsp, regMask{}}, outputs: gponly}
   183  		gp21shxload    = regInfo{inputs: []regMask{gpspsbg, gp, regMask{}}, outputs: gponly}
   184  		gp21shxloadidx = regInfo{inputs: []regMask{gpspsbg, gpsp, gp, regMask{}}, outputs: gponly}
   185  
   186  		gpstore         = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   187  		gpstoreconst    = regInfo{inputs: []regMask{gpspsbg, regMask{}}}
   188  		gpstoreidx      = regInfo{inputs: []regMask{gpspsbg, gpsp, gpsp, regMask{}}}
   189  		gpstoreconstidx = regInfo{inputs: []regMask{gpspsbg, gpsp, regMask{}}}
   190  		gpstorexchg     = regInfo{inputs: []regMask{gp, gpspsbg, regMask{}}, outputs: []regMask{gp}}
   191  		cmpxchg         = regInfo{inputs: []regMask{gp, ax, gp, regMask{}}, outputs: []regMask{gp, regMask{}}, clobbers: ax}
   192  		atomicLogic     = regInfo{inputs: []regMask{gp.minus(ax), gp.minus(ax), regMask{}}, outputs: []regMask{ax, regMask{}}}
   193  
   194  		fp01        = regInfo{inputs: nil, outputs: fponly}
   195  		fp21        = regInfo{inputs: []regMask{fp, fp}, outputs: fponly}
   196  		fp31        = regInfo{inputs: []regMask{fp, fp, fp}, outputs: fponly}
   197  		fp21load    = regInfo{inputs: []regMask{fp, gpspsbg, regMask{}}, outputs: fponly}
   198  		fp21loadidx = regInfo{inputs: []regMask{fp, gpspsbg, gpspsb, regMask{}}, outputs: fponly}
   199  		fpgp        = regInfo{inputs: fponly, outputs: gponly}
   200  		gpfp        = regInfo{inputs: gponly, outputs: fponly}
   201  		fp11        = regInfo{inputs: fponly, outputs: fponly}
   202  		fp2flags    = regInfo{inputs: []regMask{fp, fp}}
   203  
   204  		fpload    = regInfo{inputs: []regMask{gpspsb, {}}, outputs: fponly}
   205  		fploadidx = regInfo{inputs: []regMask{gpspsb, gpsp, {}}, outputs: fponly}
   206  		vload     = regInfo{inputs: []regMask{gpspsb, {}}, outputs: vonly}
   207  		wload     = regInfo{inputs: []regMask{gpspsb, {}}, outputs: wonly}
   208  
   209  		fpstore    = regInfo{inputs: []regMask{gpspsb, fp, {}}}
   210  		fpstoreidx = regInfo{inputs: []regMask{gpspsb, gpsp, fp, {}}}
   211  		vstore     = regInfo{inputs: []regMask{gpspsb, vz, {}}}
   212  		wstore     = regInfo{inputs: []regMask{gpspsb, wz, {}}}
   213  
   214  		// masked loads/stores, vector register or mask register
   215  		vloadv  = regInfo{inputs: []regMask{gpspsb, v, {}}, outputs: vonly}
   216  		vstorev = regInfo{inputs: []regMask{gpspsb, v, vz, {}}}
   217  		wloadk  = regInfo{inputs: []regMask{gpspsb, mask, {}}, outputs: wonly}
   218  		wstorek = regInfo{inputs: []regMask{gpspsb, mask, wz, {}}}
   219  
   220  		v01     = regInfo{inputs: nil, outputs: vonly}
   221  		v11     = regInfo{inputs: vonly, outputs: vonly}            // used in resultInArg0 ops, arg0 must not be x15
   222  		v21     = regInfo{inputs: []regMask{v, vz}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   223  		vk      = regInfo{inputs: vzonly, outputs: maskonly}
   224  		kv      = regInfo{inputs: maskonly, outputs: vonly}
   225  		v2k     = regInfo{inputs: []regMask{vz, vz}, outputs: maskonly}
   226  		vkv     = regInfo{inputs: []regMask{vz, mask}, outputs: vonly}
   227  		v2kv    = regInfo{inputs: []regMask{vz, vz, mask}, outputs: vonly}
   228  		v2kk    = regInfo{inputs: []regMask{vz, vz, mask}, outputs: maskonly}
   229  		v31     = regInfo{inputs: []regMask{v, vz, vz}, outputs: vonly}       // used in resultInArg0 ops, arg0 must not be x15
   230  		v3kv    = regInfo{inputs: []regMask{v, vz, vz, mask}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   231  		vgpv    = regInfo{inputs: []regMask{vz, gp}, outputs: vonly}
   232  		vgp     = regInfo{inputs: vonly, outputs: gponly}
   233  		vfpv    = regInfo{inputs: []regMask{vz, fp}, outputs: vonly}
   234  		vfpkv   = regInfo{inputs: []regMask{vz, fp, mask}, outputs: vonly}
   235  		fpv     = regInfo{inputs: []regMask{fp}, outputs: vonly}
   236  		gpv     = regInfo{inputs: []regMask{gp}, outputs: vonly}
   237  		v2flags = regInfo{inputs: []regMask{vz, vz}}
   238  
   239  		w01   = regInfo{inputs: nil, outputs: wonly}
   240  		w11   = regInfo{inputs: wonly, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   241  		w21   = regInfo{inputs: []regMask{wz, wz}, outputs: wonly}
   242  		wk    = regInfo{inputs: wzonly, outputs: maskonly}
   243  		kw    = regInfo{inputs: maskonly, outputs: wonly}
   244  		w2k   = regInfo{inputs: []regMask{wz, wz}, outputs: maskonly}
   245  		wkw   = regInfo{inputs: []regMask{wz, mask}, outputs: wonly}
   246  		w2kw  = regInfo{inputs: []regMask{w, wz, mask}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   247  		w2kk  = regInfo{inputs: []regMask{wz, wz, mask}, outputs: maskonly}
   248  		w31   = regInfo{inputs: []regMask{w, wz, wz}, outputs: wonly}       // used in resultInArg0 ops, arg0 must not be x15
   249  		w3kw  = regInfo{inputs: []regMask{w, wz, wz, mask}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   250  		wgpw  = regInfo{inputs: []regMask{wz, gp}, outputs: wonly}
   251  		wgp   = regInfo{inputs: wzonly, outputs: gponly}
   252  		wfpw  = regInfo{inputs: []regMask{wz, fp}, outputs: wonly}
   253  		wfpkw = regInfo{inputs: []regMask{wz, fp, mask}, outputs: wonly}
   254  
   255  		// These register masks are used by SIMD only, they follow the pattern:
   256  		// Mem last, k mask second to last (if any), address right before mem and k mask.
   257  		wkwload    = regInfo{inputs: []regMask{gpspsb, mask, regMask{}}, outputs: wonly}
   258  		v21load    = regInfo{inputs: []regMask{v, gpspsb, regMask{}}, outputs: vonly}     // used in resultInArg0 ops, arg0 must not be x15
   259  		v31load    = regInfo{inputs: []regMask{v, vz, gpspsb, regMask{}}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   260  		v11load    = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: vonly}
   261  		w21load    = regInfo{inputs: []regMask{wz, gpspsb, regMask{}}, outputs: wonly}
   262  		w31load    = regInfo{inputs: []regMask{w, wz, gpspsb, regMask{}}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   263  		w2kload    = regInfo{inputs: []regMask{wz, gpspsb, regMask{}}, outputs: maskonly}
   264  		w2kwload   = regInfo{inputs: []regMask{wz, gpspsb, mask, regMask{}}, outputs: wonly}
   265  		w11load    = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: wonly}
   266  		w3kwload   = regInfo{inputs: []regMask{w, wz, gpspsb, mask, regMask{}}, outputs: wonly} // used in resultInArg0 ops, arg0 must not be x15
   267  		w2kkload   = regInfo{inputs: []regMask{wz, gpspsb, mask, regMask{}}, outputs: maskonly}
   268  		v31x0AtIn2 = regInfo{inputs: []regMask{v, vz, x0}, outputs: vonly} // used in resultInArg0 ops, arg0 must not be x15
   269  
   270  		kload  = regInfo{inputs: []regMask{gpspsb, regMask{}}, outputs: maskonly}
   271  		kstore = regInfo{inputs: []regMask{gpspsb, mask, regMask{}}}
   272  		gpk    = regInfo{inputs: gponly, outputs: maskonly}
   273  		kgp    = regInfo{inputs: maskonly, outputs: gponly}
   274  		k2k    = regInfo{inputs: []regMask{mask, mask}, outputs: maskonly}
   275  
   276  		x15only = regInfo{inputs: nil, outputs: []regMask{x15}}
   277  
   278  		prefreg = regInfo{inputs: []regMask{gpspsbg}}
   279  	)
   280  
   281  	var AMD64ops = []opData{
   282  		// {ADD,SUB,MUL,DIV}Sx: floating-point arithmetic
   283  		// x==S for float32, x==D for float64
   284  		// computes arg0 OP arg1
   285  		{name: "ADDSS", argLength: 2, reg: fp21, asm: "ADDSS", commutative: true, resultInArg0: true, earlyOk: true},
   286  		{name: "ADDSD", argLength: 2, reg: fp21, asm: "ADDSD", commutative: true, resultInArg0: true, earlyOk: true},
   287  		{name: "SUBSS", argLength: 2, reg: fp21, asm: "SUBSS", resultInArg0: true, earlyOk: true},
   288  		{name: "SUBSD", argLength: 2, reg: fp21, asm: "SUBSD", resultInArg0: true, earlyOk: true},
   289  		{name: "MULSS", argLength: 2, reg: fp21, asm: "MULSS", commutative: true, resultInArg0: true, earlyOk: true},
   290  		{name: "MULSD", argLength: 2, reg: fp21, asm: "MULSD", commutative: true, resultInArg0: true, earlyOk: true},
   291  		{name: "DIVSS", argLength: 2, reg: fp21, asm: "DIVSS", resultInArg0: true, earlyOk: true},
   292  		{name: "DIVSD", argLength: 2, reg: fp21, asm: "DIVSD", resultInArg0: true, earlyOk: true},
   293  
   294  		// MOVSxload: floating-point loads
   295  		// x==S for float32, x==D for float64
   296  		// load from arg0+auxint+aux, arg1 = mem
   297  		{name: "MOVSSload", argLength: 2, reg: fpload, asm: "MOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   298  		{name: "MOVSDload", argLength: 2, reg: fpload, asm: "MOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   299  
   300  		// MOVSxconst: floatint-point constants
   301  		// x==S for float32, x==D for float64
   302  		{name: "MOVSSconst", reg: fp01, asm: "MOVSS", aux: "Float32", rematerializeable: true, earlyOk: true},
   303  		{name: "MOVSDconst", reg: fp01, asm: "MOVSD", aux: "Float64", rematerializeable: true, earlyOk: true},
   304  
   305  		// MOVSxloadidx: floating-point indexed loads
   306  		// x==S for float32, x==D for float64
   307  		// load from arg0 + scale*arg1+auxint+aux, arg2 = mem
   308  		{name: "MOVSSloadidx1", argLength: 3, reg: fploadidx, asm: "MOVSS", scale: 1, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   309  		{name: "MOVSSloadidx4", argLength: 3, reg: fploadidx, asm: "MOVSS", scale: 4, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   310  		{name: "MOVSDloadidx1", argLength: 3, reg: fploadidx, asm: "MOVSD", scale: 1, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   311  		{name: "MOVSDloadidx8", argLength: 3, reg: fploadidx, asm: "MOVSD", scale: 8, aux: "SymOff", symEffect: "Read", addrSinkArg0: true},
   312  
   313  		// MOVSxstore: floating-point stores
   314  		// x==S for float32, x==D for float64
   315  		// does *(arg0+auxint+aux) = arg1, arg2 = mem
   316  		{name: "MOVSSstore", argLength: 3, reg: fpstore, asm: "MOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   317  		{name: "MOVSDstore", argLength: 3, reg: fpstore, asm: "MOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   318  
   319  		// MOVSxstoreidx: floating-point indexed stores
   320  		// x==S for float32, x==D for float64
   321  		// does *(arg0+scale*arg1+auxint+aux) = arg2, arg3 = mem
   322  		{name: "MOVSSstoreidx1", argLength: 4, reg: fpstoreidx, asm: "MOVSS", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   323  		{name: "MOVSSstoreidx4", argLength: 4, reg: fpstoreidx, asm: "MOVSS", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   324  		{name: "MOVSDstoreidx1", argLength: 4, reg: fpstoreidx, asm: "MOVSD", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   325  		{name: "MOVSDstoreidx8", argLength: 4, reg: fpstoreidx, asm: "MOVSD", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   326  
   327  		// {ADD,SUB,MUL,DIV}Sxload: floating-point load / op combo
   328  		// x==S for float32, x==D for float64
   329  		// computes arg0 OP *(arg1+auxint+aux), arg2=mem
   330  		{name: "ADDSSload", argLength: 3, reg: fp21load, asm: "ADDSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   331  		{name: "ADDSDload", argLength: 3, reg: fp21load, asm: "ADDSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   332  		{name: "SUBSSload", argLength: 3, reg: fp21load, asm: "SUBSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   333  		{name: "SUBSDload", argLength: 3, reg: fp21load, asm: "SUBSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   334  		{name: "MULSSload", argLength: 3, reg: fp21load, asm: "MULSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   335  		{name: "MULSDload", argLength: 3, reg: fp21load, asm: "MULSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   336  		{name: "DIVSSload", argLength: 3, reg: fp21load, asm: "DIVSS", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   337  		{name: "DIVSDload", argLength: 3, reg: fp21load, asm: "DIVSD", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   338  
   339  		// {ADD,SUB,MUL,DIV}Sxloadidx: floating-point indexed load / op combo
   340  		// x==S for float32, x==D for float64
   341  		// computes arg0 OP *(arg1+scale*arg2+auxint+aux), arg3=mem
   342  		{name: "ADDSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "ADDSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   343  		{name: "ADDSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "ADDSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   344  		{name: "ADDSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "ADDSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   345  		{name: "ADDSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "ADDSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   346  		{name: "SUBSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "SUBSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   347  		{name: "SUBSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "SUBSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   348  		{name: "SUBSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "SUBSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   349  		{name: "SUBSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "SUBSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   350  		{name: "MULSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "MULSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   351  		{name: "MULSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "MULSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   352  		{name: "MULSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "MULSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   353  		{name: "MULSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "MULSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   354  		{name: "DIVSSloadidx1", argLength: 4, reg: fp21loadidx, asm: "DIVSS", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   355  		{name: "DIVSSloadidx4", argLength: 4, reg: fp21loadidx, asm: "DIVSS", scale: 4, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   356  		{name: "DIVSDloadidx1", argLength: 4, reg: fp21loadidx, asm: "DIVSD", scale: 1, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   357  		{name: "DIVSDloadidx8", argLength: 4, reg: fp21loadidx, asm: "DIVSD", scale: 8, aux: "SymOff", resultInArg0: true, symEffect: "Read", addrSinkArg1: true},
   358  
   359  		// {ADD,SUB,MUL,DIV,AND,OR,XOR}x: binary integer ops
   360  		//   unadorned versions compute arg0 OP arg1
   361  		//       const versions compute arg0 OP auxint (auxint is a sign-extended 32-bit value)
   362  		// constmodify versions compute *(arg0+ValAndOff(AuxInt).Off().aux) OP= ValAndOff(AuxInt).Val(), arg1 = mem
   363  		// x==L operations zero the upper 4 bytes of the destination register (not meaningful for constmodify versions).
   364  		{name: "ADDQ", argLength: 2, reg: gp21sp, asm: "ADDQ", commutative: true, clobberFlags: true, earlyOk: true},
   365  		{name: "ADDL", argLength: 2, reg: gp21sp, asm: "ADDL", commutative: true, clobberFlags: true, earlyOk: true},
   366  		{name: "ADDQconst", argLength: 1, reg: gp11sp, asm: "ADDQ", aux: "Int32", typ: "UInt64", clobberFlags: true, earlyOk: true},
   367  		{name: "ADDLconst", argLength: 1, reg: gp11sp, asm: "ADDL", aux: "Int32", clobberFlags: true, earlyOk: true},
   368  		{name: "ADDQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   369  		{name: "ADDLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ADDL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   370  
   371  		{name: "SUBQ", argLength: 2, reg: gp21sp2, asm: "SUBQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   372  		{name: "SUBL", argLength: 2, reg: gp21, asm: "SUBL", resultInArg0: true, clobberFlags: true, earlyOk: true},
   373  		{name: "SUBQconst", argLength: 1, reg: gp11, asm: "SUBQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},
   374  		{name: "SUBLconst", argLength: 1, reg: gp11, asm: "SUBL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},
   375  
   376  		{name: "MULQ", argLength: 2, reg: gp21, asm: "IMULQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},
   377  		{name: "MULL", argLength: 2, reg: gp21, asm: "IMULL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},
   378  		{name: "MULQconst", argLength: 1, reg: gp11, asm: "IMUL3Q", aux: "Int32", clobberFlags: true, earlyOk: true},
   379  		{name: "MULLconst", argLength: 1, reg: gp11, asm: "IMUL3L", aux: "Int32", clobberFlags: true, earlyOk: true},
   380  
   381  		// Let x = arg0*arg1 (full 32x32->64  unsigned multiply). Returns uint32(x), and flags set to overflow if uint32(x) != x.
   382  		{name: "MULLU", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{ax, regMask{}}, clobbers: dx}, typ: "(UInt32,Flags)", asm: "MULL", commutative: true, clobberFlags: true},
   383  		// Let x = arg0*arg1 (full 64x64->128 unsigned multiply). Returns uint64(x), and flags set to overflow if uint64(x) != x.
   384  		{name: "MULQU", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{ax, regMask{}}, clobbers: dx}, typ: "(UInt64,Flags)", asm: "MULQ", commutative: true, clobberFlags: true},
   385  
   386  		// HMULx[U]: computes the high bits of an integer multiply.
   387  		// computes arg0 * arg1 >> (x==L?32:64)
   388  		// The multiply is unsigned for the U versions, signed for the non-U versions.
   389  		// HMULx[U] are intentionally not marked as commutative, even though they are.
   390  		// This is because they have asymmetric register requirements.
   391  		// There are rewrite rules to try to place arguments in preferable slots.
   392  		{name: "HMULQ", argLength: 2, reg: gp21hmul, asm: "IMULQ", clobberFlags: true, earlyOk: true},
   393  		{name: "HMULL", argLength: 2, reg: gp21hmul, asm: "IMULL", clobberFlags: true, earlyOk: true},
   394  		{name: "HMULQU", argLength: 2, reg: gp21hmul, asm: "MULQ", clobberFlags: true, earlyOk: true},
   395  		{name: "HMULLU", argLength: 2, reg: gp21hmul, asm: "MULL", clobberFlags: true, earlyOk: true},
   396  
   397  		// (arg0 + arg1) / 2 as unsigned, all 64 result bits
   398  		{name: "AVGQU", argLength: 2, reg: gp21, commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},
   399  
   400  		// DIVx[U] computes [arg0 / arg1, arg0 % arg1]
   401  		// For signed versions, AuxInt non-zero means that the divisor has been proved to be not -1.
   402  		{name: "DIVQ", argLength: 2, reg: gp11div, typ: "(Int64,Int64)", asm: "IDIVQ", aux: "Bool", clobberFlags: true},
   403  		{name: "DIVL", argLength: 2, reg: gp11div, typ: "(Int32,Int32)", asm: "IDIVL", aux: "Bool", clobberFlags: true},
   404  		{name: "DIVW", argLength: 2, reg: gp11div, typ: "(Int16,Int16)", asm: "IDIVW", aux: "Bool", clobberFlags: true},
   405  		{name: "DIVQU", argLength: 2, reg: gp11div, typ: "(UInt64,UInt64)", asm: "DIVQ", clobberFlags: true},
   406  		{name: "DIVLU", argLength: 2, reg: gp11div, typ: "(UInt32,UInt32)", asm: "DIVL", clobberFlags: true},
   407  		{name: "DIVWU", argLength: 2, reg: gp11div, typ: "(UInt16,UInt16)", asm: "DIVW", clobberFlags: true},
   408  
   409  		// computes -arg0, flags set for 0-arg0.
   410  		{name: "NEGLflags", argLength: 1, reg: gp11flags, typ: "(UInt32,Flags)", asm: "NEGL", resultInArg0: true},
   411  		// compute arg0+auxint. flags set for arg0+auxint.
   412  		// NOTE: we pretend the CF/OF flags are undefined for these instructions,
   413  		// so we can use INC/DEC instead of ADDQconst if auxint is +/-1. (INC/DEC don't modify CF.)
   414  		{name: "ADDQconstflags", argLength: 1, reg: gp11flags, aux: "Int32", asm: "ADDQ", resultInArg0: true},
   415  		{name: "ADDLconstflags", argLength: 1, reg: gp11flags, aux: "Int32", asm: "ADDL", resultInArg0: true},
   416  
   417  		// The following 4 add opcodes return the low 64 bits of the sum in the first result and
   418  		// the carry (the 65th bit) in the carry flag.
   419  		{name: "ADDQcarry", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "ADDQ", commutative: true, resultInArg0: true}, // r = arg0+arg1
   420  		{name: "ADCQ", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "ADCQ", commutative: true, resultInArg0: true}, // r = arg0+arg1+carry(arg2)
   421  		{name: "ADDQconstcarry", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "ADDQ", aux: "Int32", resultInArg0: true}, // r = arg0+auxint
   422  		{name: "ADCQconst", argLength: 2, reg: gp1flags1flags, typ: "(UInt64,Flags)", asm: "ADCQ", aux: "Int32", resultInArg0: true}, // r = arg0+auxint+carry(arg1)
   423  
   424  		// The following 4 add opcodes return the low 64 bits of the difference in the first result and
   425  		// the borrow (if the result is negative) in the carry flag.
   426  		{name: "SUBQborrow", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "SUBQ", resultInArg0: true},                    // r = arg0-arg1
   427  		{name: "SBBQ", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "SBBQ", resultInArg0: true},                     // r = arg0-(arg1+carry(arg2))
   428  		{name: "SUBQconstborrow", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "SUBQ", aux: "Int32", resultInArg0: true}, // r = arg0-auxint
   429  		{name: "SBBQconst", argLength: 2, reg: gp1flags1flags, typ: "(UInt64,Flags)", asm: "SBBQ", aux: "Int32", resultInArg0: true},  // r = arg0-(auxint+carry(arg1))
   430  
   431  		{name: "MULQU2", argLength: 2, reg: regInfo{inputs: []regMask{ax, gpsp}, outputs: []regMask{dx, ax}}, commutative: true, asm: "MULQ", clobberFlags: true, earlyOk: true}, // arg0 * arg1, returns (hi, lo)
   432  		{name: "DIVQU2", argLength: 3, reg: regInfo{inputs: []regMask{dx, ax, gpsp}, outputs: []regMask{ax, dx}}, asm: "DIVQ", clobberFlags: true},                               // arg0:arg1 / arg2 (128-bit divided by 64-bit), returns (q, r)
   433  
   434  		{name: "ANDQ", argLength: 2, reg: gp21, asm: "ANDQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & arg1
   435  		{name: "ANDL", argLength: 2, reg: gp21, asm: "ANDL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & arg1
   436  		{name: "ANDQconst", argLength: 1, reg: gp11, asm: "ANDQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & auxint
   437  		{name: "ANDLconst", argLength: 1, reg: gp11, asm: "ANDL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 & auxint
   438  		{name: "ANDQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   439  		{name: "ANDLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ANDL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // and ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   440  
   441  		{name: "ORQ", argLength: 2, reg: gp21, asm: "ORQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | arg1
   442  		{name: "ORL", argLength: 2, reg: gp21, asm: "ORL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | arg1
   443  		{name: "ORQconst", argLength: 1, reg: gp11, asm: "ORQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | auxint
   444  		{name: "ORLconst", argLength: 1, reg: gp11, asm: "ORL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 | auxint
   445  		{name: "ORQconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   446  		{name: "ORLconstmodify", argLength: 2, reg: gpstoreconst, asm: "ORL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // or ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   447  
   448  		{name: "XORQ", argLength: 2, reg: gp21, asm: "XORQ", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ arg1
   449  		{name: "XORL", argLength: 2, reg: gp21, asm: "XORL", commutative: true, resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ arg1
   450  		{name: "XORQconst", argLength: 1, reg: gp11, asm: "XORQ", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ auxint
   451  		{name: "XORLconst", argLength: 1, reg: gp11, asm: "XORL", aux: "Int32", resultInArg0: true, clobberFlags: true, earlyOk: true},                                                      // arg0 ^ auxint
   452  		{name: "XORQconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   453  		{name: "XORLconstmodify", argLength: 2, reg: gpstoreconst, asm: "XORL", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true}, // xor ValAndOff(AuxInt).Val() to arg0+ValAndOff(AuxInt).Off()+aux, arg1=mem
   454  
   455  		// CMPx: compare arg0 to arg1.
   456  		{name: "CMPQ", argLength: 2, reg: gp2flags, asm: "CMPQ", typ: "Flags"},
   457  		{name: "CMPL", argLength: 2, reg: gp2flags, asm: "CMPL", typ: "Flags"},
   458  		{name: "CMPW", argLength: 2, reg: gp2flags, asm: "CMPW", typ: "Flags"},
   459  		{name: "CMPB", argLength: 2, reg: gp2flags, asm: "CMPB", typ: "Flags"},
   460  
   461  		// CMPxconst: compare arg0 to auxint.
   462  		{name: "CMPQconst", argLength: 1, reg: gp1flags, asm: "CMPQ", typ: "Flags", aux: "Int32"},
   463  		{name: "CMPLconst", argLength: 1, reg: gp1flags, asm: "CMPL", typ: "Flags", aux: "Int32"},
   464  		{name: "CMPWconst", argLength: 1, reg: gp1flags, asm: "CMPW", typ: "Flags", aux: "Int16"},
   465  		{name: "CMPBconst", argLength: 1, reg: gp1flags, asm: "CMPB", typ: "Flags", aux: "Int8"},
   466  
   467  		// CMPxload: compare *(arg0+auxint+aux) to arg1 (in that order). arg2=mem.
   468  		{name: "CMPQload", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   469  		{name: "CMPLload", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   470  		{name: "CMPWload", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   471  		{name: "CMPBload", argLength: 3, reg: gp1flagsLoad, asm: "CMPB", aux: "SymOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   472  
   473  		// CMPxconstload: compare *(arg0+ValAndOff(AuxInt).Off()+aux) to ValAndOff(AuxInt).Val() (in that order). arg1=mem.
   474  		{name: "CMPQconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPQ", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   475  		{name: "CMPLconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPL", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   476  		{name: "CMPWconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPW", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   477  		{name: "CMPBconstload", argLength: 2, reg: gp0flagsLoad, asm: "CMPB", aux: "SymValAndOff", typ: "Flags", symEffect: "Read", faultOnNilArg0: true, addrSinkArg0: true},
   478  
   479  		// CMPxloadidx: compare *(arg0+N*arg1+auxint+aux) to arg2 (in that order). arg3=mem.
   480  		{name: "CMPQloadidx8", argLength: 4, reg: gp2flagsLoad, asm: "CMPQ", scale: 8, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   481  		{name: "CMPQloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPQ", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   482  		{name: "CMPLloadidx4", argLength: 4, reg: gp2flagsLoad, asm: "CMPL", scale: 4, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   483  		{name: "CMPLloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPL", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   484  		{name: "CMPWloadidx2", argLength: 4, reg: gp2flagsLoad, asm: "CMPW", scale: 2, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   485  		{name: "CMPWloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPW", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   486  		{name: "CMPBloadidx1", argLength: 4, reg: gp2flagsLoad, asm: "CMPB", scale: 1, commutative: true, aux: "SymOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   487  
   488  		// CMPxconstloadidx: compare *(arg0+N*arg1+ValAndOff(AuxInt).Off()+aux) to ValAndOff(AuxInt).Val() (in that order). arg2=mem.
   489  		{name: "CMPQconstloadidx8", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", scale: 8, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   490  		{name: "CMPQconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPQ", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   491  		{name: "CMPLconstloadidx4", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", scale: 4, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   492  		{name: "CMPLconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPL", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   493  		{name: "CMPWconstloadidx2", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", scale: 2, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true},
   494  		{name: "CMPWconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPW", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   495  		{name: "CMPBconstloadidx1", argLength: 3, reg: gp1flagsLoad, asm: "CMPB", scale: 1, commutative: true, aux: "SymValAndOff", typ: "Flags", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   496  
   497  		// UCOMISx: floating-point compare arg0 to arg1
   498  		// x==S for float32, x==D for float64
   499  		{name: "UCOMISS", argLength: 2, reg: fp2flags, asm: "UCOMISS", typ: "Flags"},
   500  		{name: "UCOMISD", argLength: 2, reg: fp2flags, asm: "UCOMISD", typ: "Flags"},
   501  
   502  		// bit test/set/clear operations
   503  		{name: "BTL", argLength: 2, reg: gp2flags, asm: "BTL", typ: "Flags"},                                                          // test whether bit arg0%32 in arg1 is set
   504  		{name: "BTQ", argLength: 2, reg: gp2flags, asm: "BTQ", typ: "Flags"},                                                          // test whether bit arg0%64 in arg1 is set
   505  		{name: "BTCL", argLength: 2, reg: gp21, asm: "BTCL", resultInArg0: true, clobberFlags: true, earlyOk: true},                   // complement bit arg1%32 in arg0
   506  		{name: "BTCQ", argLength: 2, reg: gp21, asm: "BTCQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                   // complement bit arg1%64 in arg0
   507  		{name: "BTRL", argLength: 2, reg: gp21, asm: "BTRL", resultInArg0: true, clobberFlags: true, earlyOk: true},                   // reset bit arg1%32 in arg0
   508  		{name: "BTRQ", argLength: 2, reg: gp21, asm: "BTRQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                   // reset bit arg1%64 in arg0
   509  		{name: "BTSL", argLength: 2, reg: gp21, asm: "BTSL", resultInArg0: true, clobberFlags: true, earlyOk: true},                   // set bit arg1%32 in arg0
   510  		{name: "BTSQ", argLength: 2, reg: gp21, asm: "BTSQ", resultInArg0: true, clobberFlags: true, earlyOk: true},                   // set bit arg1%64 in arg0
   511  		{name: "BTLconst", argLength: 1, reg: gp1flags, asm: "BTL", typ: "Flags", aux: "Int8", earlyOk: true},                         // test whether bit auxint in arg0 is set, 0 <= auxint < 32
   512  		{name: "BTQconst", argLength: 1, reg: gp1flags, asm: "BTQ", typ: "Flags", aux: "Int8", earlyOk: true},                         // test whether bit auxint in arg0 is set, 0 <= auxint < 64
   513  		{name: "BTCQconst", argLength: 1, reg: gp11, asm: "BTCQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true}, // complement bit auxint in arg0, 31 <= auxint < 64
   514  		{name: "BTRQconst", argLength: 1, reg: gp11, asm: "BTRQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true}, // reset bit auxint in arg0, 31 <= auxint < 64
   515  		{name: "BTSQconst", argLength: 1, reg: gp11, asm: "BTSQ", resultInArg0: true, clobberFlags: true, aux: "Int8", earlyOk: true}, // set bit auxint in arg0, 31 <= auxint < 64
   516  
   517  		// BT[SRC]Qconstmodify
   518  		//
   519  		//  S: set bit
   520  		//  R: reset (clear) bit
   521  		//  C: complement bit
   522  		//
   523  		// Apply operation to bit ValAndOff(AuxInt).Val() in the 64 bits at
   524  		// memory address arg0+ValAndOff(AuxInt).Off()+aux
   525  		// Bit index must be in range (31-63).
   526  		// (We use OR/AND/XOR for thinner targets and lower bit indexes.)
   527  		// arg1=mem, returns mem
   528  		//
   529  		// Note that there aren't non-const versions of these instructions.
   530  		// Well, there are such instructions, but they are slow and weird so we don't use them.
   531  		{name: "BTSQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTSQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   532  		{name: "BTRQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTRQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   533  		{name: "BTCQconstmodify", argLength: 2, reg: gpstoreconst, asm: "BTCQ", aux: "SymValAndOff", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   534  
   535  		// TESTx: compare (arg0 & arg1) to 0
   536  		{name: "TESTQ", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTQ", typ: "Flags"},
   537  		{name: "TESTL", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTL", typ: "Flags"},
   538  		{name: "TESTW", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTW", typ: "Flags"},
   539  		{name: "TESTB", argLength: 2, reg: gp2flags, commutative: true, asm: "TESTB", typ: "Flags"},
   540  
   541  		// TESTxconst: compare (arg0 & auxint) to 0
   542  		{name: "TESTQconst", argLength: 1, reg: gp1flags, asm: "TESTQ", typ: "Flags", aux: "Int32"},
   543  		{name: "TESTLconst", argLength: 1, reg: gp1flags, asm: "TESTL", typ: "Flags", aux: "Int32"},
   544  		{name: "TESTWconst", argLength: 1, reg: gp1flags, asm: "TESTW", typ: "Flags", aux: "Int16"},
   545  		{name: "TESTBconst", argLength: 1, reg: gp1flags, asm: "TESTB", typ: "Flags", aux: "Int8"},
   546  
   547  		// S{HL, HR, AR}x: shift operations
   548  		// SHL: shift left
   549  		// SHR: shift right logical (0s are shifted in from beyond the word size)
   550  		// SAR: shift right arithmetic (sign bit is shifted in from beyond the word size)
   551  		// arg0 is the value being shifted
   552  		// arg1 is the amount to shift, interpreted mod (Q=64,L=32,W=32,B=32)
   553  		// (Note: x86 is weird, the 16 and 8 byte shifts still use all 5 bits of shift amount!)
   554  		// For *const versions, use auxint instead of arg1 as the shift amount. auxint must be in the range 0 to (Q=63,L=31,W=15,B=7) inclusive.
   555  		{name: "SHLQ", argLength: 2, reg: gp21shift, asm: "SHLQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   556  		{name: "SHLL", argLength: 2, reg: gp21shift, asm: "SHLL", resultInArg0: true, clobberFlags: true, earlyOk: true},
   557  		{name: "SHLQconst", argLength: 1, reg: gp11, asm: "SHLQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   558  		{name: "SHLLconst", argLength: 1, reg: gp11, asm: "SHLL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   559  
   560  		{name: "SHRQ", argLength: 2, reg: gp21shift, asm: "SHRQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   561  		{name: "SHRL", argLength: 2, reg: gp21shift, asm: "SHRL", resultInArg0: true, clobberFlags: true, earlyOk: true},
   562  		{name: "SHRW", argLength: 2, reg: gp21shift, asm: "SHRW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   563  		{name: "SHRB", argLength: 2, reg: gp21shift, asm: "SHRB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   564  		{name: "SHRQconst", argLength: 1, reg: gp11, asm: "SHRQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   565  		{name: "SHRLconst", argLength: 1, reg: gp11, asm: "SHRL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   566  		{name: "SHRWconst", argLength: 1, reg: gp11, asm: "SHRW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   567  		{name: "SHRBconst", argLength: 1, reg: gp11, asm: "SHRB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   568  
   569  		{name: "SARQ", argLength: 2, reg: gp21shift, asm: "SARQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   570  		{name: "SARL", argLength: 2, reg: gp21shift, asm: "SARL", resultInArg0: true, clobberFlags: true, earlyOk: true},
   571  		{name: "SARW", argLength: 2, reg: gp21shift, asm: "SARW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   572  		{name: "SARB", argLength: 2, reg: gp21shift, asm: "SARB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   573  		{name: "SARQconst", argLength: 1, reg: gp11, asm: "SARQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   574  		{name: "SARLconst", argLength: 1, reg: gp11, asm: "SARL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   575  		{name: "SARWconst", argLength: 1, reg: gp11, asm: "SARW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   576  		{name: "SARBconst", argLength: 1, reg: gp11, asm: "SARB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   577  
   578  		// RO{L,R}x: rotate instructions
   579  		// computes arg0 rotate (L=left,R=right) arg1 bits.
   580  		// Bits are rotated within the low (Q=64,L=32,W=16,B=8) bits of the register.
   581  		// For *const versions use auxint instead of arg1 as the rotate amount. auxint must be in the range 0 to (Q=63,L=31,W=15,B=7) inclusive.
   582  		// x==L versions zero the upper 32 bits of the destination register.
   583  		// x==W and x==B versions leave the upper bits unspecified.
   584  		{name: "ROLQ", argLength: 2, reg: gp21shift, asm: "ROLQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   585  		{name: "ROLL", argLength: 2, reg: gp21shift, asm: "ROLL", resultInArg0: true, clobberFlags: true, earlyOk: true},
   586  		{name: "ROLW", argLength: 2, reg: gp21shift, asm: "ROLW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   587  		{name: "ROLB", argLength: 2, reg: gp21shift, asm: "ROLB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   588  		{name: "RORQ", argLength: 2, reg: gp21shift, asm: "RORQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   589  		{name: "RORL", argLength: 2, reg: gp21shift, asm: "RORL", resultInArg0: true, clobberFlags: true, earlyOk: true},
   590  		{name: "RORW", argLength: 2, reg: gp21shift, asm: "RORW", resultInArg0: true, clobberFlags: true, earlyOk: true},
   591  		{name: "RORB", argLength: 2, reg: gp21shift, asm: "RORB", resultInArg0: true, clobberFlags: true, earlyOk: true},
   592  		{name: "ROLQconst", argLength: 1, reg: gp11, asm: "ROLQ", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   593  		{name: "ROLLconst", argLength: 1, reg: gp11, asm: "ROLL", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   594  		{name: "ROLWconst", argLength: 1, reg: gp11, asm: "ROLW", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   595  		{name: "ROLBconst", argLength: 1, reg: gp11, asm: "ROLB", aux: "Int8", resultInArg0: true, clobberFlags: true, earlyOk: true},
   596  
   597  		// [ADD,SUB,AND,OR]xload: integer load/op combo
   598  		// L = int32, Q = int64
   599  		// x==L operations zero the upper 4 bytes of the destination register.
   600  		// computes arg0 op *(arg1+auxint+aux), arg2=mem
   601  		{name: "ADDLload", argLength: 3, reg: gp21load, asm: "ADDL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   602  		{name: "ADDQload", argLength: 3, reg: gp21load, asm: "ADDQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   603  		{name: "SUBQload", argLength: 3, reg: gp21load, asm: "SUBQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   604  		{name: "SUBLload", argLength: 3, reg: gp21load, asm: "SUBL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   605  		{name: "ANDLload", argLength: 3, reg: gp21load, asm: "ANDL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   606  		{name: "ANDQload", argLength: 3, reg: gp21load, asm: "ANDQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   607  		{name: "ORQload", argLength: 3, reg: gp21load, asm: "ORQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   608  		{name: "ORLload", argLength: 3, reg: gp21load, asm: "ORL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   609  		{name: "XORQload", argLength: 3, reg: gp21load, asm: "XORQ", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   610  		{name: "XORLload", argLength: 3, reg: gp21load, asm: "XORL", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, symEffect: "Read", addrSinkArg1: true},
   611  
   612  		// integer indexed load/op combo
   613  		// L = int32, Q = int64
   614  		// L operations zero the upper 4 bytes of the destination register.
   615  		// computes arg0 op *(arg1+scale*arg2+auxint+aux), arg3=mem
   616  		{name: "ADDLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   617  		{name: "ADDLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   618  		{name: "ADDLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ADDL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   619  		{name: "ADDQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ADDQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   620  		{name: "ADDQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ADDQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   621  		{name: "SUBLloadidx1", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   622  		{name: "SUBLloadidx4", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   623  		{name: "SUBLloadidx8", argLength: 4, reg: gp21loadidx, asm: "SUBL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   624  		{name: "SUBQloadidx1", argLength: 4, reg: gp21loadidx, asm: "SUBQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   625  		{name: "SUBQloadidx8", argLength: 4, reg: gp21loadidx, asm: "SUBQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   626  		{name: "ANDLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   627  		{name: "ANDLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   628  		{name: "ANDLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ANDL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   629  		{name: "ANDQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ANDQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   630  		{name: "ANDQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ANDQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   631  		{name: "ORLloadidx1", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   632  		{name: "ORLloadidx4", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   633  		{name: "ORLloadidx8", argLength: 4, reg: gp21loadidx, asm: "ORL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   634  		{name: "ORQloadidx1", argLength: 4, reg: gp21loadidx, asm: "ORQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   635  		{name: "ORQloadidx8", argLength: 4, reg: gp21loadidx, asm: "ORQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   636  		{name: "XORLloadidx1", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   637  		{name: "XORLloadidx4", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 4, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   638  		{name: "XORLloadidx8", argLength: 4, reg: gp21loadidx, asm: "XORL", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   639  		{name: "XORQloadidx1", argLength: 4, reg: gp21loadidx, asm: "XORQ", scale: 1, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   640  		{name: "XORQloadidx8", argLength: 4, reg: gp21loadidx, asm: "XORQ", scale: 8, aux: "SymOff", resultInArg0: true, clobberFlags: true, symEffect: "Read", addrSinkArg1: true},
   641  
   642  		// direct binary op on memory (read-modify-write)
   643  		// L = int32, Q = int64
   644  		// does *(arg0+auxint+aux) op= arg1, arg2=mem
   645  		{name: "ADDQmodify", argLength: 3, reg: gpstore, asm: "ADDQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   646  		{name: "SUBQmodify", argLength: 3, reg: gpstore, asm: "SUBQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   647  		{name: "ANDQmodify", argLength: 3, reg: gpstore, asm: "ANDQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   648  		{name: "ORQmodify", argLength: 3, reg: gpstore, asm: "ORQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   649  		{name: "XORQmodify", argLength: 3, reg: gpstore, asm: "XORQ", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   650  		{name: "ADDLmodify", argLength: 3, reg: gpstore, asm: "ADDL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   651  		{name: "SUBLmodify", argLength: 3, reg: gpstore, asm: "SUBL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   652  		{name: "ANDLmodify", argLength: 3, reg: gpstore, asm: "ANDL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   653  		{name: "ORLmodify", argLength: 3, reg: gpstore, asm: "ORL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   654  		{name: "XORLmodify", argLength: 3, reg: gpstore, asm: "XORL", aux: "SymOff", typ: "Mem", clobberFlags: true, faultOnNilArg0: true, symEffect: "Read,Write", addrSinkArg0: true},
   655  
   656  		// indexed direct binary op on memory.
   657  		// does *(arg0+scale*arg1+auxint+aux) op= arg2, arg3=mem
   658  		{name: "ADDQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ADDQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   659  		{name: "ADDQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ADDQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   660  		{name: "SUBQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "SUBQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   661  		{name: "SUBQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "SUBQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   662  		{name: "ANDQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ANDQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   663  		{name: "ANDQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ANDQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   664  		{name: "ORQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ORQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   665  		{name: "ORQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ORQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   666  		{name: "XORQmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "XORQ", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   667  		{name: "XORQmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "XORQ", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   668  		{name: "ADDLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   669  		{name: "ADDLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   670  		{name: "ADDLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ADDL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   671  		{name: "SUBLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   672  		{name: "SUBLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   673  		{name: "SUBLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "SUBL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   674  		{name: "ANDLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   675  		{name: "ANDLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   676  		{name: "ANDLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ANDL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   677  		{name: "ORLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   678  		{name: "ORLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   679  		{name: "ORLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "ORL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   680  		{name: "XORLmodifyidx1", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 1, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   681  		{name: "XORLmodifyidx4", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 4, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   682  		{name: "XORLmodifyidx8", argLength: 4, reg: gpstoreidx, asm: "XORL", scale: 8, aux: "SymOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   683  
   684  		// indexed direct binary op on memory with constant argument.
   685  		// does *(arg0+scale*arg1+ValAndOff(AuxInt).Off()+aux) op= ValAndOff(AuxInt).Val(), arg2=mem
   686  		{name: "ADDQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   687  		{name: "ADDQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ADDQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   688  		{name: "ANDQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   689  		{name: "ANDQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ANDQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   690  		{name: "ORQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   691  		{name: "ORQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ORQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   692  		{name: "XORQconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORQ", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   693  		{name: "XORQconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "XORQ", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   694  		{name: "ADDLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   695  		{name: "ADDLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   696  		{name: "ADDLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ADDL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   697  		{name: "ANDLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   698  		{name: "ANDLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   699  		{name: "ANDLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ANDL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   700  		{name: "ORLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   701  		{name: "ORLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   702  		{name: "ORLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "ORL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   703  		{name: "XORLconstmodifyidx1", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 1, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   704  		{name: "XORLconstmodifyidx4", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 4, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   705  		{name: "XORLconstmodifyidx8", argLength: 3, reg: gpstoreconstidx, asm: "XORL", scale: 8, aux: "SymValAndOff", typ: "Mem", clobberFlags: true, symEffect: "Read,Write", addrSinkArg0: true},
   706  
   707  		// {NEG,NOT}x: unary ops
   708  		// computes [NEG:-,NOT:^]arg0
   709  		// L = int32, Q = int64
   710  		// L operations zero the upper 4 bytes of the destination register.
   711  		{name: "NEGQ", argLength: 1, reg: gp11, asm: "NEGQ", resultInArg0: true, clobberFlags: true, earlyOk: true},
   712  		{name: "NEGL", argLength: 1, reg: gp11, asm: "NEGL", resultInArg0: true, clobberFlags: true, earlyOk: true},
   713  		{name: "NOTQ", argLength: 1, reg: gp11, asm: "NOTQ", resultInArg0: true, earlyOk: true},
   714  		{name: "NOTL", argLength: 1, reg: gp11, asm: "NOTL", resultInArg0: true, earlyOk: true},
   715  
   716  		// BS{F,R}Q returns a tuple [result, flags]
   717  		// result is undefined if the input is zero.
   718  		// flags are set to "equal" if the input is zero, "not equal" otherwise.
   719  		// BS{F,R}L returns only the result.
   720  		{name: "BSFQ", argLength: 1, reg: gp11flags, asm: "BSFQ", typ: "(UInt64,Flags)", earlyOk: true},        // # of low-order zeroes in 64-bit arg
   721  		{name: "BSFL", argLength: 1, reg: gp11, asm: "BSFL", typ: "UInt32", clobberFlags: true, earlyOk: true}, // # of low-order zeroes in 32-bit arg
   722  		{name: "BSRQ", argLength: 1, reg: gp11flags, asm: "BSRQ", typ: "(UInt64,Flags)", earlyOk: true},        // # of high-order zeroes in 64-bit arg
   723  		{name: "BSRL", argLength: 1, reg: gp11, asm: "BSRL", typ: "UInt32", clobberFlags: true, earlyOk: true}, // # of high-order zeroes in 32-bit arg
   724  
   725  		// CMOV instructions: 64, 32 and 16-bit sizes.
   726  		// if arg2 encodes a true result, return arg1, else arg0
   727  		{name: "CMOVQEQ", argLength: 3, reg: gp21, asm: "CMOVQEQ", resultInArg0: true, earlyOk: true},
   728  		{name: "CMOVQNE", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, earlyOk: true},
   729  		{name: "CMOVQLT", argLength: 3, reg: gp21, asm: "CMOVQLT", resultInArg0: true, earlyOk: true},
   730  		{name: "CMOVQGT", argLength: 3, reg: gp21, asm: "CMOVQGT", resultInArg0: true, earlyOk: true},
   731  		{name: "CMOVQLE", argLength: 3, reg: gp21, asm: "CMOVQLE", resultInArg0: true, earlyOk: true},
   732  		{name: "CMOVQGE", argLength: 3, reg: gp21, asm: "CMOVQGE", resultInArg0: true, earlyOk: true},
   733  		{name: "CMOVQLS", argLength: 3, reg: gp21, asm: "CMOVQLS", resultInArg0: true, earlyOk: true},
   734  		{name: "CMOVQHI", argLength: 3, reg: gp21, asm: "CMOVQHI", resultInArg0: true, earlyOk: true},
   735  		{name: "CMOVQCC", argLength: 3, reg: gp21, asm: "CMOVQCC", resultInArg0: true, earlyOk: true},
   736  		{name: "CMOVQCS", argLength: 3, reg: gp21, asm: "CMOVQCS", resultInArg0: true, earlyOk: true},
   737  
   738  		{name: "CMOVLEQ", argLength: 3, reg: gp21, asm: "CMOVLEQ", resultInArg0: true, earlyOk: true},
   739  		{name: "CMOVLNE", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, earlyOk: true},
   740  		{name: "CMOVLLT", argLength: 3, reg: gp21, asm: "CMOVLLT", resultInArg0: true, earlyOk: true},
   741  		{name: "CMOVLGT", argLength: 3, reg: gp21, asm: "CMOVLGT", resultInArg0: true, earlyOk: true},
   742  		{name: "CMOVLLE", argLength: 3, reg: gp21, asm: "CMOVLLE", resultInArg0: true, earlyOk: true},
   743  		{name: "CMOVLGE", argLength: 3, reg: gp21, asm: "CMOVLGE", resultInArg0: true, earlyOk: true},
   744  		{name: "CMOVLLS", argLength: 3, reg: gp21, asm: "CMOVLLS", resultInArg0: true, earlyOk: true},
   745  		{name: "CMOVLHI", argLength: 3, reg: gp21, asm: "CMOVLHI", resultInArg0: true, earlyOk: true},
   746  		{name: "CMOVLCC", argLength: 3, reg: gp21, asm: "CMOVLCC", resultInArg0: true, earlyOk: true},
   747  		{name: "CMOVLCS", argLength: 3, reg: gp21, asm: "CMOVLCS", resultInArg0: true, earlyOk: true},
   748  
   749  		{name: "CMOVWEQ", argLength: 3, reg: gp21, asm: "CMOVWEQ", resultInArg0: true, earlyOk: true},
   750  		{name: "CMOVWNE", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, earlyOk: true},
   751  		{name: "CMOVWLT", argLength: 3, reg: gp21, asm: "CMOVWLT", resultInArg0: true, earlyOk: true},
   752  		{name: "CMOVWGT", argLength: 3, reg: gp21, asm: "CMOVWGT", resultInArg0: true, earlyOk: true},
   753  		{name: "CMOVWLE", argLength: 3, reg: gp21, asm: "CMOVWLE", resultInArg0: true, earlyOk: true},
   754  		{name: "CMOVWGE", argLength: 3, reg: gp21, asm: "CMOVWGE", resultInArg0: true, earlyOk: true},
   755  		{name: "CMOVWLS", argLength: 3, reg: gp21, asm: "CMOVWLS", resultInArg0: true, earlyOk: true},
   756  		{name: "CMOVWHI", argLength: 3, reg: gp21, asm: "CMOVWHI", resultInArg0: true, earlyOk: true},
   757  		{name: "CMOVWCC", argLength: 3, reg: gp21, asm: "CMOVWCC", resultInArg0: true, earlyOk: true},
   758  		{name: "CMOVWCS", argLength: 3, reg: gp21, asm: "CMOVWCS", resultInArg0: true, earlyOk: true},
   759  
   760  		// CMOV with floating point instructions. We need separate pseudo-op to handle
   761  		// InvertFlags correctly, and to generate special code that handles NaN (unordered flag).
   762  		// NOTE: the fact that CMOV*EQF here is marked to generate CMOV*NE is not a bug. See
   763  		// code generation in amd64/ssa.go.
   764  		{name: "CMOVQEQF", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, needIntTemp: true, earlyOk: true},
   765  		{name: "CMOVQNEF", argLength: 3, reg: gp21, asm: "CMOVQNE", resultInArg0: true, earlyOk: true},
   766  		{name: "CMOVQGTF", argLength: 3, reg: gp21, asm: "CMOVQHI", resultInArg0: true, earlyOk: true},
   767  		{name: "CMOVQGEF", argLength: 3, reg: gp21, asm: "CMOVQCC", resultInArg0: true, earlyOk: true},
   768  		{name: "CMOVLEQF", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, needIntTemp: true, earlyOk: true},
   769  		{name: "CMOVLNEF", argLength: 3, reg: gp21, asm: "CMOVLNE", resultInArg0: true, earlyOk: true},
   770  		{name: "CMOVLGTF", argLength: 3, reg: gp21, asm: "CMOVLHI", resultInArg0: true, earlyOk: true},
   771  		{name: "CMOVLGEF", argLength: 3, reg: gp21, asm: "CMOVLCC", resultInArg0: true, earlyOk: true},
   772  		{name: "CMOVWEQF", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, needIntTemp: true, earlyOk: true},
   773  		{name: "CMOVWNEF", argLength: 3, reg: gp21, asm: "CMOVWNE", resultInArg0: true, earlyOk: true},
   774  		{name: "CMOVWGTF", argLength: 3, reg: gp21, asm: "CMOVWHI", resultInArg0: true, earlyOk: true},
   775  		{name: "CMOVWGEF", argLength: 3, reg: gp21, asm: "CMOVWCC", resultInArg0: true, earlyOk: true},
   776  
   777  		// BSWAPx swaps the low-order (L=4,Q=8) bytes of arg0.
   778  		// Q: abcdefgh -> hgfedcba
   779  		// L: abcdefgh -> 0000hgfe (L zeros the upper 4 bytes)
   780  		{name: "BSWAPQ", argLength: 1, reg: gp11, asm: "BSWAPQ", resultInArg0: true, earlyOk: true},
   781  		{name: "BSWAPL", argLength: 1, reg: gp11, asm: "BSWAPL", resultInArg0: true, earlyOk: true},
   782  
   783  		// POPCNTx counts the number of set bits in the low-order (L=32,Q=64) bits of arg0.
   784  		// POPCNTx instructions are only guaranteed to be available if GOAMD64>=v2.
   785  		// For GOAMD64<v2, any use must be preceded by a successful runtime check of runtime.x86HasPOPCNT.
   786  		{name: "POPCNTQ", argLength: 1, reg: gp11, asm: "POPCNTQ", clobberFlags: true},
   787  		{name: "POPCNTL", argLength: 1, reg: gp11, asm: "POPCNTL", clobberFlags: true},
   788  
   789  		// SQRTSx computes sqrt(arg0)
   790  		// S = float32, D = float64
   791  		{name: "SQRTSD", argLength: 1, reg: fp11, asm: "SQRTSD", earlyOk: true},
   792  		{name: "SQRTSS", argLength: 1, reg: fp11, asm: "SQRTSS", earlyOk: true},
   793  
   794  		// ROUNDSD rounds arg0 to an integer depending on auxint
   795  		// 0 means math.RoundToEven, 1 means math.Floor, 2 math.Ceil, 3 math.Trunc
   796  		// (The result is still a float64.)
   797  		// ROUNDSD instruction is only guaraneteed to be available if GOAMD64>=v2.
   798  		// For GOAMD64<v2, any use must be preceded by a successful check of runtime.x86HasSSE41.
   799  		{name: "ROUNDSD", argLength: 1, reg: fp11, aux: "Int8", asm: "ROUNDSD"},
   800  		{name: "ROUNDSS", argLength: 1, reg: fp11, aux: "Int8", asm: "ROUNDSS"},
   801  		// See why we need those in issue #71204
   802  		{name: "LoweredRound32F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   803  		{name: "LoweredRound64F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   804  
   805  		// VFMADD231Sx only exist on platforms with the FMA3 instruction set.
   806  		// Any use must be preceded by a successful check of runtime.x86HasFMA or a check of GOAMD64>=v3.
   807  		// x==S for float32, x==D for float64
   808  		// arg0 + arg1*arg2, with no intermediate rounding.
   809  		{name: "VFMADD231SS", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMADD231SS"},
   810  		{name: "VFMADD231SD", argLength: 3, reg: fp31, resultInArg0: true, asm: "VFMADD231SD"},
   811  
   812  		// Note that these operations don't exactly match the semantics of Go's
   813  		// builtin min. In particular, these aren't commutative, because on various
   814  		// special cases the 2nd argument is preferred.
   815  		{name: "MINSD", argLength: 2, reg: fp21, resultInArg0: true, asm: "MINSD", earlyOk: true}, // min(arg0,arg1)
   816  		{name: "MINSS", argLength: 2, reg: fp21, resultInArg0: true, asm: "MINSS", earlyOk: true}, // min(arg0,arg1)
   817  
   818  		{name: "SBBQcarrymask", argLength: 1, reg: flagsgp, asm: "SBBQ", earlyOk: true}, // (int64)(-1) if carry is set, 0 if carry is clear.
   819  		{name: "SBBLcarrymask", argLength: 1, reg: flagsgp, asm: "SBBL", earlyOk: true}, // (int32)(-1) if carry is set, 0 if carry is clear.
   820  		// Note: SBBW and SBBB are subsumed by SBBL
   821  
   822  		{name: "SETEQ", argLength: 1, reg: readflags, asm: "SETEQ", earlyOk: true}, // extract == condition from arg0
   823  		{name: "SETNE", argLength: 1, reg: readflags, asm: "SETNE", earlyOk: true}, // extract != condition from arg0
   824  		{name: "SETL", argLength: 1, reg: readflags, asm: "SETLT", earlyOk: true},  // extract signed < condition from arg0
   825  		{name: "SETLE", argLength: 1, reg: readflags, asm: "SETLE", earlyOk: true}, // extract signed <= condition from arg0
   826  		{name: "SETG", argLength: 1, reg: readflags, asm: "SETGT", earlyOk: true},  // extract signed > condition from arg0
   827  		{name: "SETGE", argLength: 1, reg: readflags, asm: "SETGE", earlyOk: true}, // extract signed >= condition from arg0
   828  		{name: "SETB", argLength: 1, reg: readflags, asm: "SETCS", earlyOk: true},  // extract unsigned < condition from arg0
   829  		{name: "SETBE", argLength: 1, reg: readflags, asm: "SETLS", earlyOk: true}, // extract unsigned <= condition from arg0
   830  		{name: "SETA", argLength: 1, reg: readflags, asm: "SETHI", earlyOk: true},  // extract unsigned > condition from arg0
   831  		{name: "SETAE", argLength: 1, reg: readflags, asm: "SETCC", earlyOk: true}, // extract unsigned >= condition from arg0
   832  		{name: "SETO", argLength: 1, reg: readflags, asm: "SETOS", earlyOk: true},  // extract if overflow flag is set from arg0
   833  		// Variants that store result to memory
   834  		{name: "SETEQstore", argLength: 3, reg: gpstoreconst, asm: "SETEQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract == condition from arg1 to arg0+auxint+aux, arg2=mem
   835  		{name: "SETNEstore", argLength: 3, reg: gpstoreconst, asm: "SETNE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract != condition from arg1 to arg0+auxint+aux, arg2=mem
   836  		{name: "SETLstore", argLength: 3, reg: gpstoreconst, asm: "SETLT", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract signed < condition from arg1 to arg0+auxint+aux, arg2=mem
   837  		{name: "SETLEstore", argLength: 3, reg: gpstoreconst, asm: "SETLE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract signed <= condition from arg1 to arg0+auxint+aux, arg2=mem
   838  		{name: "SETGstore", argLength: 3, reg: gpstoreconst, asm: "SETGT", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract signed > condition from arg1 to arg0+auxint+aux, arg2=mem
   839  		{name: "SETGEstore", argLength: 3, reg: gpstoreconst, asm: "SETGE", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract signed >= condition from arg1 to arg0+auxint+aux, arg2=mem
   840  		{name: "SETBstore", argLength: 3, reg: gpstoreconst, asm: "SETCS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract unsigned < condition from arg1 to arg0+auxint+aux, arg2=mem
   841  		{name: "SETBEstore", argLength: 3, reg: gpstoreconst, asm: "SETLS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract unsigned <= condition from arg1 to arg0+auxint+aux, arg2=mem
   842  		{name: "SETAstore", argLength: 3, reg: gpstoreconst, asm: "SETHI", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                    // extract unsigned > condition from arg1 to arg0+auxint+aux, arg2=mem
   843  		{name: "SETAEstore", argLength: 3, reg: gpstoreconst, asm: "SETCC", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},                                   // extract unsigned >= condition from arg1 to arg0+auxint+aux, arg2=mem
   844  		{name: "SETEQstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETEQ", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract == condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   845  		{name: "SETNEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETNE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract != condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   846  		{name: "SETLstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLT", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract signed < condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   847  		{name: "SETLEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract signed <= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   848  		{name: "SETGstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETGT", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract signed > condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   849  		{name: "SETGEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETGE", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract signed >= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   850  		{name: "SETBstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETCS", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract unsigned < condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   851  		{name: "SETBEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETLS", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract unsigned <= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   852  		{name: "SETAstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETHI", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},  // extract unsigned > condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   853  		{name: "SETAEstoreidx1", argLength: 4, reg: gpstoreconstidx, asm: "SETCC", aux: "SymOff", typ: "Mem", scale: 1, commutative: true, symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // extract unsigned >= condition from arg2 to arg0+arg1+auxint+aux, arg3=mem
   854  
   855  		// Need different opcodes for floating point conditions because
   856  		// any comparison involving a NaN is always FALSE and thus
   857  		// the patterns for inverting conditions cannot be used.
   858  		{name: "SETEQF", argLength: 1, reg: flagsgp, asm: "SETEQ", clobberFlags: true, needIntTemp: true, earlyOk: true}, // extract == condition from arg0
   859  		{name: "SETNEF", argLength: 1, reg: flagsgp, asm: "SETNE", clobberFlags: true, needIntTemp: true, earlyOk: true}, // extract != condition from arg0
   860  		{name: "SETORD", argLength: 1, reg: flagsgp, asm: "SETPC", earlyOk: true},                                        // extract "ordered" (No Nan present) condition from arg0
   861  		{name: "SETNAN", argLength: 1, reg: flagsgp, asm: "SETPS", earlyOk: true},                                        // extract "unordered" (Nan present) condition from arg0
   862  
   863  		{name: "SETGF", argLength: 1, reg: flagsgp, asm: "SETHI", earlyOk: true},  // extract floating > condition from arg0
   864  		{name: "SETGEF", argLength: 1, reg: flagsgp, asm: "SETCC", earlyOk: true}, // extract floating >= condition from arg0
   865  
   866  		{name: "MOVBQSX", argLength: 1, reg: gp11, asm: "MOVBQSX", earlyOk: true}, // sign extend arg0 from int8 to int64
   867  		{name: "MOVBQZX", argLength: 1, reg: gp11, asm: "MOVBLZX", earlyOk: true}, // zero extend arg0 from int8 to int64
   868  		{name: "MOVWQSX", argLength: 1, reg: gp11, asm: "MOVWQSX", earlyOk: true}, // sign extend arg0 from int16 to int64
   869  		{name: "MOVWQZX", argLength: 1, reg: gp11, asm: "MOVWLZX", earlyOk: true}, // zero extend arg0 from int16 to int64
   870  		{name: "MOVLQSX", argLength: 1, reg: gp11, asm: "MOVLQSX", earlyOk: true}, // sign extend arg0 from int32 to int64
   871  		{name: "MOVLQZX", argLength: 1, reg: gp11, asm: "MOVL", earlyOk: true},    // zero extend arg0 from int32 to int64
   872  
   873  		{name: "MOVLconst", reg: gp01, asm: "MOVL", typ: "UInt32", aux: "Int32", rematerializeable: true, earlyOk: true}, // 32 low bits of auxint (upper 32 are zeroed)
   874  		{name: "MOVQconst", reg: gp01, asm: "MOVQ", typ: "UInt64", aux: "Int64", rematerializeable: true, earlyOk: true}, // auxint
   875  
   876  		{name: "CVTTSD2SL", argLength: 1, reg: fpgp, asm: "CVTTSD2SL", earlyOk: true}, // convert float64 to int32
   877  		{name: "CVTTSD2SQ", argLength: 1, reg: fpgp, asm: "CVTTSD2SQ", earlyOk: true}, // convert float64 to int64
   878  		{name: "CVTTSS2SL", argLength: 1, reg: fpgp, asm: "CVTTSS2SL", earlyOk: true}, // convert float32 to int32
   879  		{name: "CVTTSS2SQ", argLength: 1, reg: fpgp, asm: "CVTTSS2SQ", earlyOk: true}, // convert float32 to int64
   880  		{name: "CVTSL2SS", argLength: 1, reg: gpfp, asm: "CVTSL2SS", earlyOk: true},   // convert int32 to float32
   881  		{name: "CVTSL2SD", argLength: 1, reg: gpfp, asm: "CVTSL2SD", earlyOk: true},   // convert int32 to float64
   882  		{name: "CVTSQ2SS", argLength: 1, reg: gpfp, asm: "CVTSQ2SS", earlyOk: true},   // convert int64 to float32
   883  		{name: "CVTSQ2SD", argLength: 1, reg: gpfp, asm: "CVTSQ2SD", earlyOk: true},   // convert int64 to float64
   884  		{name: "CVTSD2SS", argLength: 1, reg: fp11, asm: "CVTSD2SS", earlyOk: true},   // convert float64 to float32
   885  		{name: "CVTSS2SD", argLength: 1, reg: fp11, asm: "CVTSS2SD", earlyOk: true},   // convert float32 to float64
   886  
   887  		// Move values between int and float registers, with no conversion.
   888  		// TODO: should we have generic versions of these?
   889  		{name: "MOVQi2f", argLength: 1, reg: gpfp, typ: "Float64", earlyOk: true}, // move 64 bits from int to float reg
   890  		{name: "MOVQf2i", argLength: 1, reg: fpgp, typ: "UInt64", earlyOk: true},  // move 64 bits from float to int reg
   891  		{name: "MOVLi2f", argLength: 1, reg: gpfp, typ: "Float32", earlyOk: true}, // move 32 bits from int to float reg
   892  		{name: "MOVLf2i", argLength: 1, reg: fpgp, typ: "UInt32", earlyOk: true},  // move 32 bits from float to int reg, zero extend
   893  
   894  		{name: "PXOR", argLength: 2, reg: fp21, asm: "PXOR", commutative: true, resultInArg0: true, earlyOk: true}, // exclusive or, applied to X regs (for float negation).
   895  		{name: "POR", argLength: 2, reg: fp21, asm: "POR", commutative: true, resultInArg0: true, earlyOk: true},   // inclusive or, applied to X regs (for float min/max).
   896  
   897  		{name: "LEAQ", argLength: 1, reg: gp11sb, asm: "LEAQ", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true}, // arg0 + auxint + offset encoded in aux
   898  		{name: "LEAL", argLength: 1, reg: gp11sb, asm: "LEAL", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true}, // arg0 + auxint + offset encoded in aux
   899  		{name: "LEAW", argLength: 1, reg: gp11sb, asm: "LEAW", aux: "SymOff", rematerializeable: true, symEffect: "Addr", earlyOk: true}, // arg0 + auxint + offset encoded in aux
   900  
   901  		// LEAxn computes arg0 + n*arg1 + auxint + aux
   902  		// x==L zeroes the upper 4 bytes.
   903  		{name: "LEAQ1", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true}, // arg0 + arg1 + auxint + aux
   904  		{name: "LEAL1", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true}, // arg0 + arg1 + auxint + aux
   905  		{name: "LEAW1", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 1, commutative: true, aux: "SymOff", symEffect: "Addr", earlyOk: true}, // arg0 + arg1 + auxint + aux
   906  		{name: "LEAQ2", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 2*arg1 + auxint + aux
   907  		{name: "LEAL2", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 2*arg1 + auxint + aux
   908  		{name: "LEAW2", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 2, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 2*arg1 + auxint + aux
   909  		{name: "LEAQ4", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 4*arg1 + auxint + aux
   910  		{name: "LEAL4", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 4*arg1 + auxint + aux
   911  		{name: "LEAW4", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 4, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 4*arg1 + auxint + aux
   912  		{name: "LEAQ8", argLength: 2, reg: gp21sb, asm: "LEAQ", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 8*arg1 + auxint + aux
   913  		{name: "LEAL8", argLength: 2, reg: gp21sb, asm: "LEAL", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 8*arg1 + auxint + aux
   914  		{name: "LEAW8", argLength: 2, reg: gp21sb, asm: "LEAW", scale: 8, aux: "SymOff", symEffect: "Addr", earlyOk: true},                    // arg0 + 8*arg1 + auxint + aux
   915  		// Note: LEAx{1,2,4,8} must not have OpSB as either argument.
   916  
   917  		// MOVxload: loads
   918  		// Load (Q=8,L=4,W=2,B=1) bytes from (arg0+auxint+aux), arg1=mem.
   919  		// "+auxint+aux" == add auxint and the offset of the symbol in aux (if any) to the effective address
   920  		// Standard versions zero extend the result. SX versions sign extend the result.
   921  		{name: "MOVBload", argLength: 2, reg: gpload, asm: "MOVBLZX", aux: "SymOff", typ: "UInt8", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   922  		{name: "MOVBQSXload", argLength: 2, reg: gpload, asm: "MOVBQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   923  		{name: "MOVWload", argLength: 2, reg: gpload, asm: "MOVWLZX", aux: "SymOff", typ: "UInt16", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   924  		{name: "MOVWQSXload", argLength: 2, reg: gpload, asm: "MOVWQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   925  		{name: "MOVLload", argLength: 2, reg: gpload, asm: "MOVL", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   926  		{name: "MOVLQSXload", argLength: 2, reg: gpload, asm: "MOVLQSX", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   927  		{name: "MOVQload", argLength: 2, reg: gpload, asm: "MOVQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},
   928  
   929  		// MOVxstore: stores
   930  		// Store (Q=8,L=4,W=2,B=1) low bytes of arg1.
   931  		// Does *(arg0+auxint+aux) = arg1, arg2=mem.
   932  		{name: "MOVBstore", argLength: 3, reg: gpstore, asm: "MOVB", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   933  		{name: "MOVWstore", argLength: 3, reg: gpstore, asm: "MOVW", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   934  		{name: "MOVLstore", argLength: 3, reg: gpstore, asm: "MOVL", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   935  		{name: "MOVQstore", argLength: 3, reg: gpstore, asm: "MOVQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   936  
   937  		// MOVOload/store: 16 byte load/store
   938  		// These operations are only used to move data around: there is no *O arithmetic, for example.
   939  		{name: "MOVOload", argLength: 2, reg: fpload, asm: "MOVUPS", aux: "SymOff", typ: "Int128", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // load 16 bytes from arg0+auxint+aux. arg1=mem
   940  		{name: "MOVOstore", argLength: 3, reg: fpstore, asm: "MOVUPS", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 16 bytes in arg1 to arg0+auxint+aux. arg2=mem
   941  
   942  		// MOVxloadidx: indexed loads
   943  		// load (Q=8,L=4,W=2,B=1) bytes from (arg0+scale*arg1+auxint+aux), arg2=mem.
   944  		// Results are zero-extended. (TODO: sign-extending indexed loads)
   945  		{name: "MOVBloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBLZX", scale: 1, aux: "SymOff", typ: "UInt8", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   946  		{name: "MOVWloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVWLZX", scale: 1, aux: "SymOff", typ: "UInt16", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   947  		{name: "MOVWloadidx2", argLength: 3, reg: gploadidx, asm: "MOVWLZX", scale: 2, aux: "SymOff", typ: "UInt16", symEffect: "Read", addrSinkArg0: true},
   948  		{name: "MOVLloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   949  		{name: "MOVLloadidx4", argLength: 3, reg: gploadidx, asm: "MOVL", scale: 4, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true},
   950  		{name: "MOVLloadidx8", argLength: 3, reg: gploadidx, asm: "MOVL", scale: 8, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true},
   951  		{name: "MOVQloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true},
   952  		{name: "MOVQloadidx8", argLength: 3, reg: gploadidx, asm: "MOVQ", scale: 8, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true},
   953  
   954  		// MOVxstoreidx: indexed stores
   955  		// Store (Q=8,L=4,W=2,B=1) low bytes of arg2.
   956  		// Does *(arg0+scale*arg1+auxint+aux) = arg2, arg3=mem.
   957  		{name: "MOVBstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVB", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   958  		{name: "MOVWstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVW", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   959  		{name: "MOVWstoreidx2", argLength: 4, reg: gpstoreidx, asm: "MOVW", scale: 2, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   960  		{name: "MOVLstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   961  		{name: "MOVLstoreidx4", argLength: 4, reg: gpstoreidx, asm: "MOVL", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   962  		{name: "MOVLstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVL", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   963  		{name: "MOVQstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   964  		{name: "MOVQstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVQ", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},
   965  
   966  		// TODO: add size-mismatched indexed loads/stores, like MOVBstoreidx4?
   967  
   968  		// MOVxstoreconst: constant stores
   969  		// Store (O=16,Q=8,L=4,W=2,B=1) constant bytes.
   970  		// Does *(arg0+ValAndOff(AuxInt).Off()+aux) = ValAndOff(AuxInt).Val(), arg1=mem.
   971  		// O version can only store the constant 0.
   972  		{name: "MOVBstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVB", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   973  		{name: "MOVWstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVW", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   974  		{name: "MOVLstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVL", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   975  		{name: "MOVQstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVQ", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   976  		{name: "MOVOstoreconst", argLength: 2, reg: gpstoreconst, asm: "MOVUPS", aux: "SymValAndOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},
   977  
   978  		// MOVxstoreconstidx: constant indexed stores
   979  		// Store (Q=8,L=4,W=2,B=1) constant bytes.
   980  		// Does *(arg0+scale*arg1+ValAndOff(AuxInt).Off()+aux) = ValAndOff(AuxInt).Val(), arg2=mem.
   981  		{name: "MOVBstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVB", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   982  		{name: "MOVWstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVW", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   983  		{name: "MOVWstoreconstidx2", argLength: 3, reg: gpstoreconstidx, asm: "MOVW", scale: 2, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
   984  		{name: "MOVLstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVL", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   985  		{name: "MOVLstoreconstidx4", argLength: 3, reg: gpstoreconstidx, asm: "MOVL", scale: 4, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
   986  		{name: "MOVQstoreconstidx1", argLength: 3, reg: gpstoreconstidx, commutative: true, asm: "MOVQ", scale: 1, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true},
   987  		{name: "MOVQstoreconstidx8", argLength: 3, reg: gpstoreconstidx, asm: "MOVQ", scale: 8, aux: "SymValAndOff", typ: "Mem", symEffect: "Write", addrSinkArg0: true},
   988  
   989  		// arg0 = pointer to start of memory to zero
   990  		// arg1 = mem
   991  		// auxint = # of bytes to zero
   992  		// returns mem
   993  		{
   994  			name:      "LoweredZero",
   995  			aux:       "Int64",
   996  			argLength: 2,
   997  			reg: regInfo{
   998  				inputs: []regMask{gp},
   999  			},
  1000  			faultOnNilArg0: true,
  1001  			addrSinkArg0:   true,
  1002  		},
  1003  
  1004  		// arg0 = pointer to start of memory to zero
  1005  		// arg1 = mem
  1006  		// auxint = # of bytes to zero
  1007  		// returns mem
  1008  		{
  1009  			name:      "LoweredZeroLoop",
  1010  			aux:       "Int64",
  1011  			argLength: 2,
  1012  			reg: regInfo{
  1013  				inputs:       []regMask{gp},
  1014  				clobbersArg0: true,
  1015  			},
  1016  			clobberFlags:   true,
  1017  			faultOnNilArg0: true,
  1018  			addrSinkArg0:   true,
  1019  			needIntTemp:    true,
  1020  		},
  1021  
  1022  		// arg0 = address of memory to zero
  1023  		// arg1 = # of 8-byte words to zero
  1024  		// arg2 = value to store (will always be zero)
  1025  		// arg3 = mem
  1026  		// returns mem
  1027  		{
  1028  			name:      "REPSTOSQ",
  1029  			argLength: 4,
  1030  			reg: regInfo{
  1031  				inputs:   []regMask{buildReg("DI"), buildReg("CX"), buildReg("AX")},
  1032  				clobbers: buildReg("DI CX"),
  1033  			},
  1034  			faultOnNilArg0: true,
  1035  			addrSinkArg0:   true,
  1036  		},
  1037  
  1038  		// With a register ABI, the actual register info for these instructions (i.e., what is used in regalloc) is augmented with per-call-site bindings of additional arguments to specific in and out registers.
  1039  		{name: "CALLstatic", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                                      // call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
  1040  		{name: "CALLtail", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},                                        // tail call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
  1041  		{name: "CALLtailinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},            // tail call fn by pointer. arg0=codeptr, last arg=mem, auxint=argsize, returns mem
  1042  		{name: "CALLclosure", argLength: -1, reg: regInfo{inputs: []regMask{gpsp, buildReg("DX"), regMask{}}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true}, // call function via closure.  arg0=codeptr, arg1=closure, last arg=mem, auxint=argsize, returns mem
  1043  		{name: "CALLinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                // call fn by pointer.  arg0=codeptr, last arg=mem, auxint=argsize, returns mem
  1044  
  1045  		// arg0 = destination pointer
  1046  		// arg1 = source pointer
  1047  		// arg2 = mem
  1048  		// auxint = # of bytes to copy
  1049  		// returns memory
  1050  		{
  1051  			name:      "LoweredMove",
  1052  			aux:       "Int64",
  1053  			argLength: 3,
  1054  			reg: regInfo{
  1055  				inputs:   []regMask{gp, gp},
  1056  				clobbers: buildReg("X14"), // uses X14 as a temporary
  1057  			},
  1058  			faultOnNilArg0: true,
  1059  			faultOnNilArg1: true,
  1060  			addrSinkArg0:   true,
  1061  			addrSinkArg1:   true,
  1062  		},
  1063  		// arg0 = destination pointer
  1064  		// arg1 = source pointer
  1065  		// arg2 = mem
  1066  		// auxint = # of bytes to copy
  1067  		// returns memory
  1068  		{
  1069  			name:      "LoweredMoveLoop",
  1070  			aux:       "Int64",
  1071  			argLength: 3,
  1072  			reg: regInfo{
  1073  				inputs:       []regMask{gp, gp},
  1074  				clobbers:     buildReg("X14"), // uses X14 as a temporary
  1075  				clobbersArg0: true,
  1076  				clobbersArg1: true,
  1077  			},
  1078  			clobberFlags:   true,
  1079  			faultOnNilArg0: true,
  1080  			faultOnNilArg1: true,
  1081  			addrSinkArg0:   true,
  1082  			addrSinkArg1:   true,
  1083  			needIntTemp:    true,
  1084  		},
  1085  
  1086  		// arg0 = destination pointer
  1087  		// arg1 = source pointer
  1088  		// arg2 = # of 8-byte words to copy
  1089  		// arg3 = mem
  1090  		// returns memory
  1091  		{
  1092  			name:      "REPMOVSQ",
  1093  			argLength: 4,
  1094  			reg: regInfo{
  1095  				inputs:   []regMask{buildReg("DI"), buildReg("SI"), buildReg("CX")},
  1096  				clobbers: buildReg("DI SI CX"),
  1097  			},
  1098  			faultOnNilArg0: true,
  1099  			faultOnNilArg1: true,
  1100  			addrSinkArg0:   true,
  1101  			addrSinkArg1:   true,
  1102  		},
  1103  
  1104  		// (InvertFlags (CMPQ a b)) == (CMPQ b a)
  1105  		// So if we want (SETL (CMPQ a b)) but we can't do that because a is a constant,
  1106  		// then we do (SETL (InvertFlags (CMPQ b a))) instead.
  1107  		// Rewrites will convert this to (SETG (CMPQ b a)).
  1108  		// InvertFlags is a pseudo-op which can't appear in assembly output.
  1109  		{name: "InvertFlags", argLength: 1}, // reverse direction of arg0
  1110  
  1111  		// Pseudo-ops
  1112  		{name: "LoweredGetG", argLength: 1, reg: gp01}, // arg0=mem
  1113  		// Scheduler ensures LoweredGetClosurePtr occurs only in entry block,
  1114  		// and sorts it to the very beginning of the block to prevent other
  1115  		// use of DX (the closure pointer)
  1116  		{name: "LoweredGetClosurePtr", reg: regInfo{outputs: []regMask{buildReg("DX")}}, zeroWidth: true},
  1117  		// LoweredGetCallerPC evaluates to the PC to which its "caller" will return.
  1118  		// I.e., if f calls g "calls" sys.GetCallerPC,
  1119  		// the result should be the PC within f that g will return to.
  1120  		// See runtime/stubs.go for a more detailed discussion.
  1121  		{name: "LoweredGetCallerPC", reg: gp01, rematerializeable: true},
  1122  		// LoweredGetCallerSP returns the SP of the caller of the current function. arg0=mem
  1123  		{name: "LoweredGetCallerSP", argLength: 1, reg: gp01, rematerializeable: true},
  1124  		//arg0=ptr,arg1=mem, returns void.  Faults if ptr is nil.
  1125  		{name: "LoweredNilCheck", argLength: 2, reg: regInfo{inputs: []regMask{gpsp}}, clobberFlags: true, nilCheck: true, faultOnNilArg0: true},
  1126  		// LoweredWB invokes runtime.gcWriteBarrier{auxint}. arg0=mem, auxint=# of buffer entries needed.
  1127  		// It saves all GP registers if necessary, but may clobber others.
  1128  		// Returns a pointer to a write barrier buffer in R11.
  1129  		{name: "LoweredWB", argLength: 1, reg: regInfo{clobbers: callerSave.minus(gp.union(g)), outputs: []regMask{buildReg("R11")}}, clobberFlags: true, aux: "Int64"},
  1130  
  1131  		{name: "LoweredHasCPUFeature", argLength: 0, reg: gp01, rematerializeable: true, typ: "UInt64", aux: "Sym", symEffect: "None"},
  1132  
  1133  		// LoweredPanicBoundsRR takes x and y, two values that caused a bounds check to fail.
  1134  		// the RC and CR versions are used when one of the arguments is a constant. CC is used
  1135  		// when both are constant (normally both 0, as prove derives the fact that a [0] bounds
  1136  		// failure means the length must have also been 0).
  1137  		// AuxInt contains a report code (see PanicBounds in genericOps.go).
  1138  		{name: "LoweredPanicBoundsRR", argLength: 3, aux: "Int64", reg: regInfo{inputs: []regMask{gp, gp}}, typ: "Mem", call: true},    // arg0=x, arg1=y, arg2=mem, returns memory.
  1139  		{name: "LoweredPanicBoundsRC", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{gp}}, typ: "Mem", call: true}, // arg0=x, arg1=mem, returns memory.
  1140  		{name: "LoweredPanicBoundsCR", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{gp}}, typ: "Mem", call: true}, // arg0=y, arg1=mem, returns memory.
  1141  		{name: "LoweredPanicBoundsCC", argLength: 1, aux: "PanicBoundsCC", reg: regInfo{}, typ: "Mem", call: true},                     // arg0=mem, returns memory.
  1142  
  1143  		// Constant flag values. For any comparison, there are 5 possible
  1144  		// outcomes: the three from the signed total order (<,==,>) and the
  1145  		// three from the unsigned total order. The == cases overlap.
  1146  		// Note: there's a sixth "unordered" outcome for floating-point
  1147  		// comparisons, but we don't use such a beast yet.
  1148  		// These ops are for temporary use by rewrite rules. They
  1149  		// cannot appear in the generated assembly.
  1150  		{name: "FlagEQ"},     // equal
  1151  		{name: "FlagLT_ULT"}, // signed < and unsigned <
  1152  		{name: "FlagLT_UGT"}, // signed < and unsigned >
  1153  		{name: "FlagGT_UGT"}, // signed > and unsigned >
  1154  		{name: "FlagGT_ULT"}, // signed > and unsigned <
  1155  
  1156  		// Atomic loads.  These are just normal loads but return <value,memory> tuples
  1157  		// so they can be properly ordered with other loads.
  1158  		// load from arg0+auxint+aux.  arg1=mem.
  1159  		{name: "MOVBatomicload", argLength: 2, reg: gpload, asm: "MOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1160  		{name: "MOVLatomicload", argLength: 2, reg: gpload, asm: "MOVL", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1161  		{name: "MOVQatomicload", argLength: 2, reg: gpload, asm: "MOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1162  
  1163  		// Atomic stores and exchanges.  Stores use XCHG to get the right memory ordering semantics.
  1164  		// store arg0 to arg1+auxint+aux, arg2=mem.
  1165  		// These ops return a tuple of <old contents of *(arg1+auxint+aux), memory>.
  1166  		// Note: arg0 and arg1 are backwards compared to MOVLstore (to facilitate resultInArg0)!
  1167  		{name: "XCHGB", argLength: 3, reg: gpstorexchg, asm: "XCHGB", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1168  		{name: "XCHGL", argLength: 3, reg: gpstorexchg, asm: "XCHGL", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1169  		{name: "XCHGQ", argLength: 3, reg: gpstorexchg, asm: "XCHGQ", aux: "SymOff", resultInArg0: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1170  
  1171  		// Atomic adds.
  1172  		// *(arg1+auxint+aux) += arg0.  arg2=mem.
  1173  		// Returns a tuple of <old contents of *(arg1+auxint+aux), memory>.
  1174  		// Note: arg0 and arg1 are backwards compared to MOVLstore (to facilitate resultInArg0)!
  1175  		{name: "XADDLlock", argLength: 3, reg: gpstorexchg, asm: "XADDL", typ: "(UInt32,Mem)", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1176  		{name: "XADDQlock", argLength: 3, reg: gpstorexchg, asm: "XADDQ", typ: "(UInt64,Mem)", aux: "SymOff", resultInArg0: true, clobberFlags: true, faultOnNilArg1: true, hasSideEffects: true, symEffect: "RdWr"},
  1177  		{name: "AddTupleFirst32", argLength: 2}, // arg1=tuple <x,y>.  Returns <x+arg0,y>.
  1178  		{name: "AddTupleFirst64", argLength: 2}, // arg1=tuple <x,y>.  Returns <x+arg0,y>.
  1179  
  1180  		// Compare and swap.
  1181  		// arg0 = pointer, arg1 = old value, arg2 = new value, arg3 = memory.
  1182  		// if *(arg0+auxint+aux) == arg1 {
  1183  		//   *(arg0+auxint+aux) = arg2
  1184  		//   return (true, memory)
  1185  		// } else {
  1186  		//   return (false, memory)
  1187  		// }
  1188  		// Note that these instructions also return the old value in AX, but we ignore it.
  1189  		// TODO: have these return flags instead of bool.  The current system generates:
  1190  		//    CMPXCHGQ ...
  1191  		//    SETEQ AX
  1192  		//    CMPB  AX, $0
  1193  		//    JNE ...
  1194  		// instead of just
  1195  		//    CMPXCHGQ ...
  1196  		//    JEQ ...
  1197  		// but we can't do that because memory-using ops can't generate flags yet
  1198  		// (flagalloc wants to move flag-generating instructions around).
  1199  		{name: "CMPXCHGLlock", argLength: 4, reg: cmpxchg, asm: "CMPXCHGL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1200  		{name: "CMPXCHGQlock", argLength: 4, reg: cmpxchg, asm: "CMPXCHGQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},
  1201  
  1202  		// Atomic memory updates using logical operations.
  1203  		// Old style that just returns the memory state.
  1204  		{name: "ANDBlock", argLength: 3, reg: gpstore, asm: "ANDB", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1205  		{name: "ANDLlock", argLength: 3, reg: gpstore, asm: "ANDL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1206  		{name: "ANDQlock", argLength: 3, reg: gpstore, asm: "ANDQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"}, // *(arg0+auxint+aux) &= arg1
  1207  		{name: "ORBlock", argLength: 3, reg: gpstore, asm: "ORB", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1208  		{name: "ORLlock", argLength: 3, reg: gpstore, asm: "ORL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1209  		{name: "ORQlock", argLength: 3, reg: gpstore, asm: "ORQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr"},   // *(arg0+auxint+aux) |= arg1
  1210  
  1211  		// Atomic memory updates using logical operations.
  1212  		// *(arg0+auxint+aux) op= arg1. arg2=mem.
  1213  		// New style that returns a tuple of <old contents of *(arg0+auxint+aux), memory>.
  1214  		{name: "LoweredAtomicAnd64", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ANDQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1215  		{name: "LoweredAtomicAnd32", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ANDL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1216  		{name: "LoweredAtomicOr64", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ORQ", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1217  		{name: "LoweredAtomicOr32", argLength: 3, reg: atomicLogic, resultNotInArgs: true, asm: "ORL", aux: "SymOff", clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, symEffect: "RdWr", unsafePoint: true, needIntTemp: true},
  1218  
  1219  		// Prefetch instructions
  1220  		// Do prefetch arg0 address. arg0=addr, arg1=memory. Instruction variant selects locality hint
  1221  		{name: "PrefetchT0", argLength: 2, reg: prefreg, asm: "PREFETCHT0", hasSideEffects: true},
  1222  		{name: "PrefetchNTA", argLength: 2, reg: prefreg, asm: "PREFETCHNTA", hasSideEffects: true},
  1223  
  1224  		// CPUID feature: BMI1.
  1225  		{name: "ANDNQ", argLength: 2, reg: gp21, asm: "ANDNQ", clobberFlags: true},         // arg0 &^ arg1
  1226  		{name: "ANDNL", argLength: 2, reg: gp21, asm: "ANDNL", clobberFlags: true},         // arg0 &^ arg1
  1227  		{name: "BLSIQ", argLength: 1, reg: gp11, asm: "BLSIQ", clobberFlags: true},         // arg0 & -arg0
  1228  		{name: "BLSIL", argLength: 1, reg: gp11, asm: "BLSIL", clobberFlags: true},         // arg0 & -arg0
  1229  		{name: "BLSMSKQ", argLength: 1, reg: gp11, asm: "BLSMSKQ", clobberFlags: true},     // arg0 ^ (arg0 - 1)
  1230  		{name: "BLSMSKL", argLength: 1, reg: gp11, asm: "BLSMSKL", clobberFlags: true},     // arg0 ^ (arg0 - 1)
  1231  		{name: "BLSRQ", argLength: 1, reg: gp11flags, asm: "BLSRQ", typ: "(UInt64,Flags)"}, // arg0 & (arg0 - 1)
  1232  		{name: "BLSRL", argLength: 1, reg: gp11flags, asm: "BLSRL", typ: "(UInt32,Flags)"}, // arg0 & (arg0 - 1)
  1233  		// count the number of trailing zero bits, prefer TZCNTQ over BSFQ, as TZCNTQ(0)==64
  1234  		// and BSFQ(0) is undefined. Same for TZCNTL(0)==32
  1235  		{name: "TZCNTQ", argLength: 1, reg: gp11, asm: "TZCNTQ", clobberFlags: true},
  1236  		{name: "TZCNTL", argLength: 1, reg: gp11, asm: "TZCNTL", clobberFlags: true},
  1237  
  1238  		// CPUID feature: LZCNT.
  1239  		// count the number of leading zero bits.
  1240  		{name: "LZCNTQ", argLength: 1, reg: gp11, asm: "LZCNTQ", typ: "UInt64", clobberFlags: true},
  1241  		{name: "LZCNTL", argLength: 1, reg: gp11, asm: "LZCNTL", typ: "UInt32", clobberFlags: true},
  1242  
  1243  		// CPUID feature: MOVBE
  1244  		// MOVBEWload does not satisfy zero extended, so only use MOVBEWstore
  1245  		{name: "MOVBEWstore", argLength: 3, reg: gpstore, asm: "MOVBEW", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // swap and store 2 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1246  		{name: "MOVBELload", argLength: 2, reg: gpload, asm: "MOVBEL", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // load and swap 4 bytes from arg0+auxint+aux. arg1=mem.  Zero extend.
  1247  		{name: "MOVBELstore", argLength: 3, reg: gpstore, asm: "MOVBEL", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // swap and store 4 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1248  		{name: "MOVBEQload", argLength: 2, reg: gpload, asm: "MOVBEQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // load and swap 8 bytes from arg0+auxint+aux. arg1=mem
  1249  		{name: "MOVBEQstore", argLength: 3, reg: gpstore, asm: "MOVBEQ", aux: "SymOff", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // swap and store 8 bytes in arg1 to arg0+auxint+aux. arg2=mem
  1250  		// indexed MOVBE loads
  1251  		{name: "MOVBELloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBEL", scale: 1, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true}, // load and swap 4 bytes from arg0+arg1+auxint+aux. arg2=mem. Zero extend.
  1252  		{name: "MOVBELloadidx4", argLength: 3, reg: gploadidx, asm: "MOVBEL", scale: 4, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true},                                        // load and swap 4 bytes from arg0+4*arg1+auxint+aux. arg2=mem. Zero extend.
  1253  		{name: "MOVBELloadidx8", argLength: 3, reg: gploadidx, asm: "MOVBEL", scale: 8, aux: "SymOff", typ: "UInt32", symEffect: "Read", addrSinkArg0: true},                                        // load and swap 4 bytes from arg0+8*arg1+auxint+aux. arg2=mem. Zero extend.
  1254  		{name: "MOVBEQloadidx1", argLength: 3, reg: gploadidx, commutative: true, asm: "MOVBEQ", scale: 1, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true, addrSinkArg1: true}, // load and swap 8 bytes from arg0+arg1+auxint+aux. arg2=mem
  1255  		{name: "MOVBEQloadidx8", argLength: 3, reg: gploadidx, asm: "MOVBEQ", scale: 8, aux: "SymOff", typ: "UInt64", symEffect: "Read", addrSinkArg0: true},                                        // load and swap 8 bytes from arg0+8*arg1+auxint+aux. arg2=mem
  1256  		// indexed MOVBE stores
  1257  		{name: "MOVBEWstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEW", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 2 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1258  		{name: "MOVBEWstoreidx2", argLength: 4, reg: gpstoreidx, asm: "MOVBEW", scale: 2, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 2 bytes in arg2 to arg0+2*arg1+auxint+aux. arg3=mem
  1259  		{name: "MOVBELstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEL", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 4 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1260  		{name: "MOVBELstoreidx4", argLength: 4, reg: gpstoreidx, asm: "MOVBEL", scale: 4, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 4 bytes in arg2 to arg0+4*arg1+auxint+aux. arg3=mem
  1261  		{name: "MOVBELstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVBEL", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 4 bytes in arg2 to arg0+8*arg1+auxint+aux. arg3=mem
  1262  		{name: "MOVBEQstoreidx1", argLength: 4, reg: gpstoreidx, commutative: true, asm: "MOVBEQ", scale: 1, aux: "SymOff", symEffect: "Write", addrSinkArg0: true, addrSinkArg1: true}, // swap and store 8 bytes in arg2 to arg0+arg1+auxint+aux. arg3=mem
  1263  		{name: "MOVBEQstoreidx8", argLength: 4, reg: gpstoreidx, asm: "MOVBEQ", scale: 8, aux: "SymOff", symEffect: "Write", addrSinkArg0: true},                                        // swap and store 8 bytes in arg2 to arg0+8*arg1+auxint+aux. arg3=mem
  1264  
  1265  		// CPUID feature: BMI2.
  1266  		{name: "SARXQ", argLength: 2, reg: gp21, asm: "SARXQ"}, // signed arg0 >> arg1, shift amount is mod 64
  1267  		{name: "SARXL", argLength: 2, reg: gp21, asm: "SARXL"}, // signed int32(arg0) >> arg1, shift amount is mod 32
  1268  		{name: "SHLXQ", argLength: 2, reg: gp21, asm: "SHLXQ"}, // arg0 << arg1, shift amount is mod 64
  1269  		{name: "SHLXL", argLength: 2, reg: gp21, asm: "SHLXL"}, // arg0 << arg1, shift amount is mod 32
  1270  		{name: "SHRXQ", argLength: 2, reg: gp21, asm: "SHRXQ"}, // unsigned arg0 >> arg1, shift amount is mod 64
  1271  		{name: "SHRXL", argLength: 2, reg: gp21, asm: "SHRXL"}, // unsigned uint32(arg0) >> arg1, shift amount is mod 32
  1272  
  1273  		{name: "SARXLload", argLength: 3, reg: gp21shxload, asm: "SARXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // signed *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 32
  1274  		{name: "SARXQload", argLength: 3, reg: gp21shxload, asm: "SARXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // signed *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 64
  1275  		{name: "SHLXLload", argLength: 3, reg: gp21shxload, asm: "SHLXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // *(arg0+auxint+aux) << arg1, arg2=mem, shift amount is mod 32
  1276  		{name: "SHLXQload", argLength: 3, reg: gp21shxload, asm: "SHLXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // *(arg0+auxint+aux) << arg1, arg2=mem, shift amount is mod 64
  1277  		{name: "SHRXLload", argLength: 3, reg: gp21shxload, asm: "SHRXL", aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // unsigned *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 32
  1278  		{name: "SHRXQload", argLength: 3, reg: gp21shxload, asm: "SHRXQ", aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // unsigned *(arg0+auxint+aux) >> arg1, arg2=mem, shift amount is mod 64
  1279  
  1280  		{name: "SARXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // signed *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1281  		{name: "SARXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // signed *(arg0+4*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1282  		{name: "SARXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SARXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // signed *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1283  		{name: "SARXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SARXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // signed *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1284  		{name: "SARXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SARXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // signed *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1285  		{name: "SHLXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // *(arg0+1*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1286  		{name: "SHLXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // *(arg0+4*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1287  		{name: "SHLXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHLXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // *(arg0+8*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 32
  1288  		{name: "SHLXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHLXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // *(arg0+1*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 64
  1289  		{name: "SHLXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHLXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // *(arg0+8*arg1+auxint+aux) << arg2, arg3=mem, shift amount is mod 64
  1290  		{name: "SHRXLloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 1, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // unsigned *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1291  		{name: "SHRXLloadidx4", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 4, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // unsigned *(arg0+4*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1292  		{name: "SHRXLloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHRXL", scale: 8, aux: "SymOff", typ: "Uint32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // unsigned *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 32
  1293  		{name: "SHRXQloadidx1", argLength: 4, reg: gp21shxloadidx, asm: "SHRXQ", scale: 1, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // unsigned *(arg0+1*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1294  		{name: "SHRXQloadidx8", argLength: 4, reg: gp21shxloadidx, asm: "SHRXQ", scale: 8, aux: "SymOff", typ: "Uint64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // unsigned *(arg0+8*arg1+auxint+aux) >> arg2, arg3=mem, shift amount is mod 64
  1295  
  1296  		// Unpack bytes, low 64-bits.
  1297  		//
  1298  		// Input/output registers treated as [8]uint8.
  1299  		//
  1300  		// output = {in1[0], in2[0], in1[1], in2[1], in1[2], in2[2], in1[3], in2[3]}
  1301  		{name: "PUNPCKLBW", argLength: 2, reg: fp21, resultInArg0: true, asm: "PUNPCKLBW"},
  1302  
  1303  		// Shuffle 16-bit words, low 64-bits.
  1304  		//
  1305  		// Input/output registers treated as [4]uint16.
  1306  		// aux=source word index for each destination word, 2 bits per index.
  1307  		//
  1308  		// output[i] = input[(aux>>2*i)&3].
  1309  		{name: "PSHUFLW", argLength: 1, reg: fp11, aux: "Int8", asm: "PSHUFLW"},
  1310  
  1311  		// Broadcast input byte.
  1312  		//
  1313  		// Input treated as uint8, output treated as [16]uint8.
  1314  		//
  1315  		// output[i] = input.
  1316  		{name: "PSHUFBbroadcast", argLength: 1, reg: fp11, resultInArg0: true, asm: "PSHUFB"}, // PSHUFB with mask zero, (GOAMD64=v1)
  1317  		{name: "VPBROADCASTB", argLength: 1, reg: gpfp, asm: "VPBROADCASTB"},                  // Broadcast input byte from gp (GOAMD64=v3)
  1318  
  1319  		// Byte negate/zero/preserve (GOAMD64=v2).
  1320  		//
  1321  		// Input/output registers treated as [16]uint8.
  1322  		//
  1323  		// if in2[i] > 0 {
  1324  		//   output[i] = in1[i]
  1325  		// } else if in2[i] == 0 {
  1326  		//   output[i] = 0
  1327  		// } else {
  1328  		//   output[i] = -1 * in1[i]
  1329  		// }
  1330  		{name: "PSIGNB", argLength: 2, reg: fp21, resultInArg0: true, asm: "PSIGNB"},
  1331  
  1332  		// Byte compare.
  1333  		//
  1334  		// Input/output registers treated as [16]uint8.
  1335  		//
  1336  		// if in1[i] == in2[i] {
  1337  		//   output[i] = 0xff
  1338  		// } else {
  1339  		//   output[i] = 0
  1340  		// }
  1341  		{name: "PCMPEQB", argLength: 2, reg: fp21, resultInArg0: true, asm: "PCMPEQB", commutative: true},
  1342  
  1343  		// Byte sign mask. Output is a bitmap of sign bits from each input byte.
  1344  		//
  1345  		// Input treated as [16]uint8. Output is [16]bit (uint16 bitmap).
  1346  		//
  1347  		// output[i] = (input[i] >> 7) & 1
  1348  		{name: "PMOVMSKB", argLength: 1, reg: fpgp, asm: "PMOVMSKB"},
  1349  
  1350  		// SIMD ops
  1351  		{name: "VMOVDQUload128", argLength: 2, reg: vload, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1352  		{name: "VMOVDQUstore128", argLength: 3, reg: vstore, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1353  
  1354  		{name: "VMOVDQUload256", argLength: 2, reg: vload, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1355  		{name: "VMOVDQUstore256", argLength: 3, reg: vstore, asm: "VMOVDQU", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1356  
  1357  		{name: "VMOVDQUload512", argLength: 2, reg: wload, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1 = mem
  1358  		{name: "VMOVDQUstore512", argLength: 3, reg: wstore, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg1, arg2 = mem
  1359  
  1360  		// AVX2 32 and 64-bit element int-vector masked moves.
  1361  		{name: "VPMASK32load128", argLength: 3, reg: vloadv, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1362  		{name: "VPMASK32store128", argLength: 4, reg: vstorev, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1363  		{name: "VPMASK64load128", argLength: 3, reg: vloadv, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1364  		{name: "VPMASK64store128", argLength: 4, reg: vstorev, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1365  
  1366  		{name: "VPMASK32load256", argLength: 3, reg: vloadv, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1367  		{name: "VPMASK32store256", argLength: 4, reg: vstorev, asm: "VPMASKMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1368  		{name: "VPMASK64load256", argLength: 3, reg: vloadv, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=integer mask, arg2 = mem
  1369  		{name: "VPMASK64store256", argLength: 4, reg: vstorev, asm: "VPMASKMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=integer mask, arg3 = mem
  1370  
  1371  		// AVX512 8-64-bit element mask-register masked moves
  1372  		{name: "VPMASK8load512", argLength: 3, reg: wloadk, asm: "VMOVDQU8", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},      // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1373  		{name: "VPMASK8store512", argLength: 4, reg: wstorek, asm: "VMOVDQU8", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},   // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1374  		{name: "VPMASK16load512", argLength: 3, reg: wloadk, asm: "VMOVDQU16", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1375  		{name: "VPMASK16store512", argLength: 4, reg: wstorek, asm: "VMOVDQU16", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1376  		{name: "VPMASK32load512", argLength: 3, reg: wloadk, asm: "VMOVDQU32", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1377  		{name: "VPMASK32store512", argLength: 4, reg: wstorek, asm: "VMOVDQU32", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1378  		{name: "VPMASK64load512", argLength: 3, reg: wloadk, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},    // load from arg0+auxint+aux, arg1=k mask, arg2 = mem
  1379  		{name: "VPMASK64store512", argLength: 4, reg: wstorek, asm: "VMOVDQU64", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"}, // store, *(arg0+auxint+aux) = arg2, arg1=k mask, arg3 = mem
  1380  
  1381  		// AVX512 moves between int-vector and mask registers
  1382  		{name: "VPMOVMToVec8x16", argLength: 1, reg: kv, asm: "VPMOVM2B"},
  1383  		{name: "VPMOVMToVec8x32", argLength: 1, reg: kv, asm: "VPMOVM2B"},
  1384  		{name: "VPMOVMToVec8x64", argLength: 1, reg: kw, asm: "VPMOVM2B"},
  1385  
  1386  		{name: "VPMOVMToVec16x8", argLength: 1, reg: kv, asm: "VPMOVM2W"},
  1387  		{name: "VPMOVMToVec16x16", argLength: 1, reg: kv, asm: "VPMOVM2W"},
  1388  		{name: "VPMOVMToVec16x32", argLength: 1, reg: kw, asm: "VPMOVM2W"},
  1389  
  1390  		{name: "VPMOVMToVec32x4", argLength: 1, reg: kv, asm: "VPMOVM2D"},
  1391  		{name: "VPMOVMToVec32x8", argLength: 1, reg: kv, asm: "VPMOVM2D"},
  1392  		{name: "VPMOVMToVec32x16", argLength: 1, reg: kw, asm: "VPMOVM2D"},
  1393  
  1394  		{name: "VPMOVMToVec64x2", argLength: 1, reg: kv, asm: "VPMOVM2Q"},
  1395  		{name: "VPMOVMToVec64x4", argLength: 1, reg: kv, asm: "VPMOVM2Q"},
  1396  		{name: "VPMOVMToVec64x8", argLength: 1, reg: kw, asm: "VPMOVM2Q"},
  1397  
  1398  		{name: "VPMOVVec8x16ToM", argLength: 1, reg: vk, asm: "VPMOVB2M"},
  1399  		{name: "VPMOVVec8x32ToM", argLength: 1, reg: vk, asm: "VPMOVB2M"},
  1400  		{name: "VPMOVVec8x64ToM", argLength: 1, reg: wk, asm: "VPMOVB2M"},
  1401  
  1402  		{name: "VPMOVVec16x8ToM", argLength: 1, reg: vk, asm: "VPMOVW2M"},
  1403  		{name: "VPMOVVec16x16ToM", argLength: 1, reg: vk, asm: "VPMOVW2M"},
  1404  		{name: "VPMOVVec16x32ToM", argLength: 1, reg: wk, asm: "VPMOVW2M"},
  1405  
  1406  		{name: "VPMOVVec32x4ToM", argLength: 1, reg: vk, asm: "VPMOVD2M"},
  1407  		{name: "VPMOVVec32x8ToM", argLength: 1, reg: vk, asm: "VPMOVD2M"},
  1408  		{name: "VPMOVVec32x16ToM", argLength: 1, reg: wk, asm: "VPMOVD2M"},
  1409  
  1410  		{name: "VPMOVVec64x2ToM", argLength: 1, reg: vk, asm: "VPMOVQ2M"},
  1411  		{name: "VPMOVVec64x4ToM", argLength: 1, reg: vk, asm: "VPMOVQ2M"},
  1412  		{name: "VPMOVVec64x8ToM", argLength: 1, reg: wk, asm: "VPMOVQ2M"},
  1413  
  1414  		// AVX1/2 moves from int-vector to bitmask (extracting sign bits)
  1415  		{name: "VPMOVMSKB128", argLength: 1, reg: vgp, asm: "VPMOVMSKB"},
  1416  		{name: "VPMOVMSKB256", argLength: 1, reg: vgp, asm: "VPMOVMSKB"},
  1417  		{name: "VMOVMSKPS128", argLength: 1, reg: vgp, asm: "VMOVMSKPS"},
  1418  		{name: "VMOVMSKPS256", argLength: 1, reg: vgp, asm: "VMOVMSKPS"},
  1419  		{name: "VMOVMSKPD128", argLength: 1, reg: vgp, asm: "VMOVMSKPD"},
  1420  		{name: "VMOVMSKPD256", argLength: 1, reg: vgp, asm: "VMOVMSKPD"},
  1421  
  1422  		// X15 is the zero register up to 128-bit. For larger values, we zero it on the fly.
  1423  		{name: "Zero128", argLength: 0, reg: x15only, zeroWidth: true, fixedReg: true},
  1424  		{name: "Zero256", argLength: 0, reg: v01, asm: "VPXOR"},
  1425  		{name: "Zero512", argLength: 0, reg: w01, asm: "VPXORQ"},
  1426  
  1427  		// Move a 32/64 bit float to a 128-bit SIMD register.
  1428  		{name: "VMOVSDf2v", argLength: 1, reg: fpv, asm: "VMOVSD"},
  1429  		{name: "VMOVSSf2v", argLength: 1, reg: fpv, asm: "VMOVSS"},
  1430  
  1431  		{name: "VMOVQ", argLength: 1, reg: gpv, asm: "VMOVQ"},
  1432  		{name: "VMOVD", argLength: 1, reg: gpv, asm: "VMOVD"},
  1433  
  1434  		{name: "VMOVQload", argLength: 2, reg: fpload, asm: "VMOVQ", aux: "SymOff", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read"},
  1435  		{name: "VMOVDload", argLength: 2, reg: fpload, asm: "VMOVD", aux: "SymOff", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read"},
  1436  		{name: "VMOVSSload", argLength: 2, reg: fpload, asm: "VMOVSS", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1437  		{name: "VMOVSDload", argLength: 2, reg: fpload, asm: "VMOVSD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1438  
  1439  		{name: "VMOVSSconst", reg: fp01, asm: "VMOVSS", aux: "Float32", rematerializeable: true},
  1440  		{name: "VMOVSDconst", reg: fp01, asm: "VMOVSD", aux: "Float64", rematerializeable: true},
  1441  
  1442  		{name: "VZEROUPPER", argLength: 1, reg: regInfo{clobbers: v}, asm: "VZEROUPPER"}, // arg=mem, returns mem
  1443  		{name: "VZEROALL", argLength: 1, reg: regInfo{clobbers: v}, asm: "VZEROALL"},     // arg=mem, returns mem
  1444  
  1445  		// KMOVxload: loads masks
  1446  		// Load (Q=8,D=4,W=2,B=1) bytes from (arg0+auxint+aux), arg1=mem.
  1447  		// "+auxint+aux" == add auxint and the offset of the symbol in aux (if any) to the effective address
  1448  		{name: "KMOVBload", argLength: 2, reg: kload, asm: "KMOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1449  		{name: "KMOVWload", argLength: 2, reg: kload, asm: "KMOVW", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1450  		{name: "KMOVDload", argLength: 2, reg: kload, asm: "KMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1451  		{name: "KMOVQload", argLength: 2, reg: kload, asm: "KMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Read"},
  1452  
  1453  		// KMOVxstore: stores masks
  1454  		// Store (Q=8,D=4,W=2,B=1) low bytes of arg1.
  1455  		// Does *(arg0+auxint+aux) = arg1, arg2=mem.
  1456  		{name: "KMOVBstore", argLength: 3, reg: kstore, asm: "KMOVB", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1457  		{name: "KMOVWstore", argLength: 3, reg: kstore, asm: "KMOVW", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1458  		{name: "KMOVDstore", argLength: 3, reg: kstore, asm: "KMOVD", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1459  		{name: "KMOVQstore", argLength: 3, reg: kstore, asm: "KMOVQ", aux: "SymOff", faultOnNilArg0: true, symEffect: "Write"},
  1460  
  1461  		// Move GP directly to mask register
  1462  		{name: "KMOVQk", argLength: 1, reg: gpk, asm: "KMOVQ"},
  1463  		{name: "KMOVDk", argLength: 1, reg: gpk, asm: "KMOVD"},
  1464  		{name: "KMOVWk", argLength: 1, reg: gpk, asm: "KMOVW"},
  1465  		{name: "KMOVBk", argLength: 1, reg: gpk, asm: "KMOVB"},
  1466  		{name: "KMOVQi", argLength: 1, reg: kgp, asm: "KMOVQ"},
  1467  		{name: "KMOVDi", argLength: 1, reg: kgp, asm: "KMOVD"},
  1468  		{name: "KMOVWi", argLength: 1, reg: kgp, asm: "KMOVW"},
  1469  		{name: "KMOVBi", argLength: 1, reg: kgp, asm: "KMOVB"},
  1470  
  1471  		// Mask logical operations
  1472  		{name: "KANDB", argLength: 2, reg: k2k, asm: "KANDB", typ: "Mask"},
  1473  		{name: "KANDW", argLength: 2, reg: k2k, asm: "KANDW", typ: "Mask"},
  1474  		{name: "KANDD", argLength: 2, reg: k2k, asm: "KANDD", typ: "Mask"},
  1475  		{name: "KANDQ", argLength: 2, reg: k2k, asm: "KANDQ", typ: "Mask"},
  1476  
  1477  		{name: "KORB", argLength: 2, reg: k2k, asm: "KORB", typ: "Mask"},
  1478  		{name: "KORW", argLength: 2, reg: k2k, asm: "KORW", typ: "Mask"},
  1479  		{name: "KORD", argLength: 2, reg: k2k, asm: "KORD", typ: "Mask"},
  1480  		{name: "KORQ", argLength: 2, reg: k2k, asm: "KORQ", typ: "Mask"},
  1481  
  1482  		{name: "KXORB", argLength: 2, reg: k2k, asm: "KXORB", typ: "Mask"},
  1483  		{name: "KXORW", argLength: 2, reg: k2k, asm: "KXORW", typ: "Mask"},
  1484  		{name: "KXORD", argLength: 2, reg: k2k, asm: "KXORD", typ: "Mask"},
  1485  		{name: "KXORQ", argLength: 2, reg: k2k, asm: "KXORQ", typ: "Mask"},
  1486  
  1487  		// Following Intel convention, we call it XNOR instead of EQ.
  1488  		{name: "KXNORB", argLength: 2, reg: k2k, asm: "KXNORB", typ: "Mask"},
  1489  		{name: "KXNORW", argLength: 2, reg: k2k, asm: "KXNORW", typ: "Mask"},
  1490  		{name: "KXNORD", argLength: 2, reg: k2k, asm: "KXNORD", typ: "Mask"},
  1491  		{name: "KXNORQ", argLength: 2, reg: k2k, asm: "KXNORQ", typ: "Mask"},
  1492  
  1493  		// VPTEST
  1494  		{name: "VPTEST", asm: "VPTEST", argLength: 2, reg: v2flags, clobberFlags: true, typ: "Flags"},
  1495  	}
  1496  
  1497  	var AMD64blocks = []blockData{
  1498  		{name: "EQ", controls: 1},
  1499  		{name: "NE", controls: 1},
  1500  		{name: "LT", controls: 1},
  1501  		{name: "LE", controls: 1},
  1502  		{name: "GT", controls: 1},
  1503  		{name: "GE", controls: 1},
  1504  		{name: "OS", controls: 1},
  1505  		{name: "OC", controls: 1},
  1506  		{name: "ULT", controls: 1},
  1507  		{name: "ULE", controls: 1},
  1508  		{name: "UGT", controls: 1},
  1509  		{name: "UGE", controls: 1},
  1510  		{name: "EQF", controls: 1},
  1511  		{name: "NEF", controls: 1},
  1512  		{name: "ORD", controls: 1}, // FP, ordered comparison (parity zero)
  1513  		{name: "NAN", controls: 1}, // FP, unordered comparison (parity one)
  1514  
  1515  		// JUMPTABLE implements jump tables.
  1516  		// Aux is the symbol (an *obj.LSym) for the jump table.
  1517  		// control[0] is the index into the jump table.
  1518  		// control[1] is the address of the jump table (the address of the symbol stored in Aux).
  1519  		{name: "JUMPTABLE", controls: 2, aux: "Sym"},
  1520  	}
  1521  
  1522  	archs = append(archs, arch{
  1523  		name:        "AMD64",
  1524  		pkg:         "cmd/internal/obj/x86",
  1525  		genfile:     "../../amd64/ssa.go",
  1526  		genSIMDfile: "../../amd64/simdssa.go",
  1527  		ops: append(AMD64ops, simdAMD64Ops(v11, v21, v2k, vkv, v2kv, v2kk, v31, v3kv, vgpv, vgp, vfpv, vfpkv,
  1528  			w11, w21, w2k, wkw, w2kw, w2kk, w31, w3kw, wgpw, wgp, wfpw, wfpkw, wkwload, v21load, v31load, v11load,
  1529  			w21load, w31load, w2kload, w2kwload, w11load, w3kwload, w2kkload, v31x0AtIn2)...), // AMD64ops,
  1530  		blocks:             AMD64blocks,
  1531  		regnames:           regNamesAMD64,
  1532  		ParamIntRegNames:   "AX BX CX DI SI R8 R9 R10 R11",
  1533  		ParamFloatRegNames: "X0 X1 X2 X3 X4 X5 X6 X7 X8 X9 X10 X11 X12 X13 X14",
  1534  		gpregmask:          gp,
  1535  		fpregmask:          fp,
  1536  		specialregmask:     mask.union(w.minus(v)),
  1537  		simdregmask:        v,
  1538  		framepointerreg:    int8(num["BP"]),
  1539  		linkreg:            -1, // not used
  1540  	})
  1541  }
  1542  

View as plain text