Source file src/cmd/compile/internal/ssa/_gen/ARM64Ops.go

     1  // Copyright 2016 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package main
     6  
     7  import "strings"
     8  
     9  // Notes:
    10  //  - Integer types live in the low portion of registers. Upper portions are junk.
    11  //  - Boolean types use the low-order byte of a register. 0=false, 1=true.
    12  //    Upper bytes are junk.
    13  //  - *const instructions may use a constant larger than the instruction can encode.
    14  //    In this case the assembler expands to multiple instructions and uses tmp
    15  //    register (R27).
    16  //  - All 32-bit Ops will zero the upper 32 bits of the destination register.
    17  
    18  // Suffixes encode the bit width of various instructions.
    19  // D (double word) = 64 bit
    20  // W (word)        = 32 bit
    21  // H (half word)   = 16 bit
    22  // HU              = 16 bit unsigned
    23  // B (byte)        = 8 bit
    24  // BU              = 8 bit unsigned
    25  // S (single)      = 32 bit float
    26  // D (double)      = 64 bit float
    27  
    28  // Note: registers not used in regalloc are not included in this list,
    29  // so that regmask stays within int64
    30  // Be careful when hand coding regmasks.
    31  var regNamesARM64 = []string{
    32  	"R0",
    33  	"R1",
    34  	"R2",
    35  	"R3",
    36  	"R4",
    37  	"R5",
    38  	"R6",
    39  	"R7",
    40  	"R8",
    41  	"R9",
    42  	"R10",
    43  	"R11",
    44  	"R12",
    45  	"R13",
    46  	"R14",
    47  	"R15",
    48  	"R16",
    49  	"R17",
    50  	// R18 = platform register, not used
    51  	"R19",
    52  	"R20",
    53  	"R21",
    54  	"R22",
    55  	"R23",
    56  	"R24",
    57  	"R25",
    58  	"R26",
    59  	// R27 = REGTMP not used in regalloc
    60  	"g",    // aka R28
    61  	"R29",  // frame pointer, not used
    62  	"R30",  // aka REGLINK
    63  	"ZERO", // zero register (aka R31)
    64  	"SP",   // stack pointer (aka R31)
    65  
    66  	// Note: both ZERO and SP are register number 31!
    67  	// What r31 means in a particular instruction depends on
    68  	// the instruction.  Generally, for arguments of instructions
    69  	// which are addresses to load or store from, r31 means SP.
    70  	// In other instructions, r31 means ZERO. But there are
    71  	// exceptions.
    72  	// See https://stackoverflow.com/questions/61532867
    73  	// This does not have much of an effect here, as the
    74  	// cmd/internal/obj/arm64 interface treats them as two
    75  	// different registers and picks the right instruction
    76  	// that encodes what r31 means. But see issue 71651.
    77  
    78  	"F0",
    79  	"F1",
    80  	"F2",
    81  	"F3",
    82  	"F4",
    83  	"F5",
    84  	"F6",
    85  	"F7",
    86  	"F8",
    87  	"F9",
    88  	"F10",
    89  	"F11",
    90  	"F12",
    91  	"F13",
    92  	"F14",
    93  	"F15",
    94  	"F16",
    95  	"F17",
    96  	"F18",
    97  	"F19",
    98  	"F20",
    99  	"F21",
   100  	"F22",
   101  	"F23",
   102  	"F24",
   103  	"F25",
   104  	"F26",
   105  	"F27",
   106  	"F28",
   107  	"F29",
   108  	"F30",
   109  	"F31",
   110  
   111  	// If you add registers, update asyncPreempt in runtime.
   112  
   113  	// pseudo-registers
   114  	"SB",
   115  }
   116  
   117  func init() {
   118  	// Make map from reg names to reg integers.
   119  	if len(regNamesARM64) > 64 {
   120  		panic("too many registers")
   121  	}
   122  	num := map[string]int{}
   123  	for i, name := range regNamesARM64 {
   124  		num[name] = i
   125  	}
   126  	buildReg := func(s string) regMask {
   127  		m := regMask{}
   128  		for _, r := range strings.Split(s, " ") {
   129  			if n, ok := num[r]; ok {
   130  				m = m.addReg(uint(n))
   131  				continue
   132  			}
   133  			panic("register " + r + " not found")
   134  		}
   135  		return m
   136  	}
   137  
   138  	// Common individual register masks
   139  	var (
   140  		gp         = buildReg("R0 R1 R2 R3 R4 R5 R6 R7 R8 R9 R10 R11 R12 R13 R14 R15 R16 R17 R19 R20 R21 R22 R23 R24 R25 R26 R30")
   141  		gpg        = gp.union(buildReg("g"))
   142  		gpsp       = gp.union(buildReg("SP"))
   143  		gpspg      = gpg.union(buildReg("SP"))
   144  		gpspsbg    = gpspg.union(buildReg("SB"))
   145  		fp         = buildReg("F0 F1 F2 F3 F4 F5 F6 F7 F8 F9 F10 F11 F12 F13 F14 F15 F16 F17 F18 F19 F20 F21 F22 F23 F24 F25 F26 F27 F28 F29 F30 F31")
   146  		callerSave = gp.union(fp).union(buildReg("g")) // runtime.setg (and anything calling it) may clobber g
   147  		r25        = buildReg("R25")
   148  		r24to25    = buildReg("R24 R25")
   149  		f16to17    = buildReg("F16 F17")
   150  		rz         = buildReg("ZERO")
   151  		first16    = buildReg("R0 R1 R2 R3 R4 R5 R6 R7 R8 R9 R10 R11 R12 R13 R14 R15")
   152  	)
   153  	// Common regInfo
   154  	var (
   155  		gp01           = regInfo{inputs: nil, outputs: []regMask{gp}}
   156  		gp0flags1      = regInfo{inputs: []regMask{regMask{}}, outputs: []regMask{gp}}
   157  		gp11           = regInfo{inputs: []regMask{gpg}, outputs: []regMask{gp}}
   158  		gp11sp         = regInfo{inputs: []regMask{gpspg}, outputs: []regMask{gp}}
   159  		gp1flags       = regInfo{inputs: []regMask{gpg}}
   160  		gp1flagsflags  = regInfo{inputs: []regMask{gpg}}
   161  		gp1flags1      = regInfo{inputs: []regMask{gpg}, outputs: []regMask{gp}}
   162  		gp11flags      = regInfo{inputs: []regMask{gpg}, outputs: []regMask{gp, regMask{}}}
   163  		gp21           = regInfo{inputs: []regMask{gpg, gpg}, outputs: []regMask{gp}}
   164  		gp21nog        = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp}}
   165  		gp21flags      = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp, regMask{}}}
   166  		gp2flags       = regInfo{inputs: []regMask{gpg, gpg}}
   167  		gp2flagsflags  = regInfo{inputs: []regMask{gpg, gpg}}
   168  		gp2flags1      = regInfo{inputs: []regMask{gp, gp}, outputs: []regMask{gp}}
   169  		gp2flags1flags = regInfo{inputs: []regMask{gp, gp, regMask{}}, outputs: []regMask{gp, regMask{}}}
   170  		gp2load        = regInfo{inputs: []regMask{gpspsbg, gpg}, outputs: []regMask{gp}}
   171  		gp31           = regInfo{inputs: []regMask{gpg, gpg, gpg}, outputs: []regMask{gp}}
   172  		gpload         = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{gp}}
   173  		gpload2        = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{gpg, gpg}}
   174  		gpstore        = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz)}}
   175  		gpstore2       = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz), gpg.union(rz)}}
   176  		gpxchg         = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz)}, outputs: []regMask{gp}}
   177  		gpcas          = regInfo{inputs: []regMask{gpspsbg, gpg.union(rz), gpg.union(rz)}, outputs: []regMask{gp}}
   178  		fp01           = regInfo{inputs: nil, outputs: []regMask{fp}}
   179  		fp11           = regInfo{inputs: []regMask{fp}, outputs: []regMask{fp}}
   180  		fpgp           = regInfo{inputs: []regMask{fp}, outputs: []regMask{gp}}
   181  		fpgpfp         = regInfo{inputs: []regMask{fp, gp}, outputs: []regMask{fp}}
   182  		gpfp           = regInfo{inputs: []regMask{gp}, outputs: []regMask{fp}}
   183  		fp21           = regInfo{inputs: []regMask{fp, fp}, outputs: []regMask{fp}}
   184  		fp31           = regInfo{inputs: []regMask{fp, fp, fp}, outputs: []regMask{fp}}
   185  		fp2flags       = regInfo{inputs: []regMask{fp, fp}}
   186  		fp1flags       = regInfo{inputs: []regMask{fp}}
   187  		fpload         = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{fp}}
   188  		fpload2        = regInfo{inputs: []regMask{gpspsbg}, outputs: []regMask{fp, fp}}
   189  		fp2load        = regInfo{inputs: []regMask{gpspsbg, gpg}, outputs: []regMask{fp}}
   190  		fpstore        = regInfo{inputs: []regMask{gpspsbg, fp}}
   191  		fpstoreidx     = regInfo{inputs: []regMask{gpspsbg, gpg, fp}}
   192  		fpstore2       = regInfo{inputs: []regMask{gpspsbg, fp, fp}}
   193  		readflags      = regInfo{inputs: nil, outputs: []regMask{gp}}
   194  		prefreg        = regInfo{inputs: []regMask{gpspsbg}}
   195  	)
   196  	ops := []opData{
   197  		// binary ops
   198  		{name: "ADCSflags", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "ADCS", commutative: true}, // arg0+arg1+carry, set flags.
   199  		{name: "ADCzerocarry", argLength: 1, reg: gp0flags1, typ: "UInt64", asm: "ADC", earlyOk: true},                // ZR+ZR+carry
   200  		{name: "ADD", argLength: 2, reg: gp21, asm: "ADD", commutative: true, earlyOk: true},                          // arg0 + arg1
   201  		{name: "ADDconst", argLength: 1, reg: gp11sp, asm: "ADD", aux: "Int64", earlyOk: true},                        // arg0 + auxInt
   202  		{name: "ADDSconstflags", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "ADDS", aux: "Int64"},      // arg0+auxint, set flags.
   203  		{name: "ADDSflags", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "ADDS", commutative: true},      // arg0+arg1, set flags.
   204  		{name: "SUB", argLength: 2, reg: gp21, asm: "SUB", earlyOk: true},                                             // arg0 - arg1
   205  		{name: "SUBconst", argLength: 1, reg: gp11, asm: "SUB", aux: "Int64", earlyOk: true},                          // arg0 - auxInt
   206  		{name: "SBCSflags", argLength: 3, reg: gp2flags1flags, typ: "(UInt64,Flags)", asm: "SBCS"},                    // arg0-(arg1+borrowing), set flags.
   207  		{name: "SUBSflags", argLength: 2, reg: gp21flags, typ: "(UInt64,Flags)", asm: "SUBS"},                         // arg0 - arg1, set flags.
   208  		{name: "MUL", argLength: 2, reg: gp21, asm: "MUL", commutative: true, earlyOk: true},                          // arg0 * arg1
   209  		{name: "MULW", argLength: 2, reg: gp21, asm: "MULW", commutative: true, earlyOk: true},                        // arg0 * arg1, 32-bit
   210  		{name: "MNEG", argLength: 2, reg: gp21, asm: "MNEG", commutative: true, earlyOk: true},                        // -arg0 * arg1
   211  		{name: "MNEGW", argLength: 2, reg: gp21, asm: "MNEGW", commutative: true, earlyOk: true},                      // -arg0 * arg1, 32-bit
   212  		{name: "MULH", argLength: 2, reg: gp21, asm: "SMULH", commutative: true, earlyOk: true},                       // (arg0 * arg1) >> 64, signed
   213  		{name: "UMULH", argLength: 2, reg: gp21, asm: "UMULH", commutative: true, earlyOk: true},                      // (arg0 * arg1) >> 64, unsigned
   214  		{name: "MULL", argLength: 2, reg: gp21, asm: "SMULL", commutative: true, earlyOk: true},                       // arg0 * arg1, signed, 32-bit mult results in 64-bit
   215  		{name: "UMULL", argLength: 2, reg: gp21, asm: "UMULL", commutative: true, earlyOk: true},                      // arg0 * arg1, unsigned, 32-bit mult results in 64-bit
   216  		{name: "DIV", argLength: 2, reg: gp21, asm: "SDIV", earlyOk: true},                                            // arg0 / arg1, signed
   217  		{name: "UDIV", argLength: 2, reg: gp21, asm: "UDIV", earlyOk: true},                                           // arg0 / arg1, unsigned
   218  		{name: "DIVW", argLength: 2, reg: gp21, asm: "SDIVW", earlyOk: true},                                          // arg0 / arg1, signed, 32 bit
   219  		{name: "UDIVW", argLength: 2, reg: gp21, asm: "UDIVW", earlyOk: true},                                         // arg0 / arg1, unsigned, 32 bit
   220  		{name: "MOD", argLength: 2, reg: gp21, asm: "REM", earlyOk: true},                                             // arg0 % arg1, signed
   221  		{name: "UMOD", argLength: 2, reg: gp21, asm: "UREM", earlyOk: true},                                           // arg0 % arg1, unsigned
   222  		{name: "MODW", argLength: 2, reg: gp21, asm: "REMW", earlyOk: true},                                           // arg0 % arg1, signed, 32 bit
   223  		{name: "UMODW", argLength: 2, reg: gp21, asm: "UREMW", earlyOk: true},                                         // arg0 % arg1, unsigned, 32 bit
   224  
   225  		{name: "FADDS", argLength: 2, reg: fp21, asm: "FADDS", commutative: true, earlyOk: true},   // arg0 + arg1
   226  		{name: "FADDD", argLength: 2, reg: fp21, asm: "FADDD", commutative: true, earlyOk: true},   // arg0 + arg1
   227  		{name: "FSUBS", argLength: 2, reg: fp21, asm: "FSUBS", earlyOk: true},                      // arg0 - arg1
   228  		{name: "FSUBD", argLength: 2, reg: fp21, asm: "FSUBD", earlyOk: true},                      // arg0 - arg1
   229  		{name: "FMULS", argLength: 2, reg: fp21, asm: "FMULS", commutative: true, earlyOk: true},   // arg0 * arg1
   230  		{name: "FMULD", argLength: 2, reg: fp21, asm: "FMULD", commutative: true, earlyOk: true},   // arg0 * arg1
   231  		{name: "FNMULS", argLength: 2, reg: fp21, asm: "FNMULS", commutative: true, earlyOk: true}, // -(arg0 * arg1)
   232  		{name: "FNMULD", argLength: 2, reg: fp21, asm: "FNMULD", commutative: true, earlyOk: true}, // -(arg0 * arg1)
   233  		{name: "FDIVS", argLength: 2, reg: fp21, asm: "FDIVS", earlyOk: true},                      // arg0 / arg1
   234  		{name: "FDIVD", argLength: 2, reg: fp21, asm: "FDIVD", earlyOk: true},                      // arg0 / arg1
   235  
   236  		{name: "AND", argLength: 2, reg: gp21, asm: "AND", commutative: true, earlyOk: true}, // arg0 & arg1
   237  		{name: "ANDconst", argLength: 1, reg: gp11, asm: "AND", aux: "Int64", earlyOk: true}, // arg0 & auxInt
   238  		{name: "OR", argLength: 2, reg: gp21, asm: "ORR", commutative: true, earlyOk: true},  // arg0 | arg1
   239  		{name: "ORconst", argLength: 1, reg: gp11, asm: "ORR", aux: "Int64", earlyOk: true},  // arg0 | auxInt
   240  		{name: "XOR", argLength: 2, reg: gp21, asm: "EOR", commutative: true, earlyOk: true}, // arg0 ^ arg1
   241  		{name: "XORconst", argLength: 1, reg: gp11, asm: "EOR", aux: "Int64", earlyOk: true}, // arg0 ^ auxInt
   242  		{name: "BIC", argLength: 2, reg: gp21, asm: "BIC", earlyOk: true},                    // arg0 &^ arg1
   243  		{name: "EON", argLength: 2, reg: gp21, asm: "EON", earlyOk: true},                    // arg0 ^ ^arg1
   244  		{name: "ORN", argLength: 2, reg: gp21, asm: "ORN", earlyOk: true},                    // arg0 | ^arg1
   245  
   246  		// unary ops
   247  		{name: "MVN", argLength: 1, reg: gp11, asm: "MVN", earlyOk: true},                              // ^arg0
   248  		{name: "NEG", argLength: 1, reg: gp11, asm: "NEG", earlyOk: true},                              // -arg0
   249  		{name: "NEGSflags", argLength: 1, reg: gp11flags, typ: "(UInt64,Flags)", asm: "NEGS"},          // -arg0, set flags.
   250  		{name: "NGCzerocarry", argLength: 1, reg: gp0flags1, typ: "UInt64", asm: "NGC", earlyOk: true}, // -1 if borrowing, 0 otherwise.
   251  		{name: "FABSD", argLength: 1, reg: fp11, asm: "FABSD", earlyOk: true},                          // abs(arg0), float64
   252  		{name: "FABSS", argLength: 1, reg: fp11, asm: "FABSS", earlyOk: true},                          // abs(arg0), float32
   253  		{name: "FNEGS", argLength: 1, reg: fp11, asm: "FNEGS", earlyOk: true},                          // -arg0, float32
   254  		{name: "FNEGD", argLength: 1, reg: fp11, asm: "FNEGD", earlyOk: true},                          // -arg0, float64
   255  		{name: "FSQRTD", argLength: 1, reg: fp11, asm: "FSQRTD", earlyOk: true},                        // sqrt(arg0), float64
   256  		{name: "FSQRTS", argLength: 1, reg: fp11, asm: "FSQRTS", earlyOk: true},                        // sqrt(arg0), float32
   257  		{name: "FMIND", argLength: 2, reg: fp21, asm: "FMIND", earlyOk: true},                          // min(arg0, arg1)
   258  		{name: "FMINS", argLength: 2, reg: fp21, asm: "FMINS", earlyOk: true},                          // min(arg0, arg1)
   259  		{name: "FMAXD", argLength: 2, reg: fp21, asm: "FMAXD", earlyOk: true},                          // max(arg0, arg1)
   260  		{name: "FMAXS", argLength: 2, reg: fp21, asm: "FMAXS", earlyOk: true},                          // max(arg0, arg1)
   261  		{name: "REV", argLength: 1, reg: gp11, asm: "REV", earlyOk: true},                              // byte reverse, 64-bit
   262  		{name: "REVW", argLength: 1, reg: gp11, asm: "REVW", earlyOk: true},                            // byte reverse, 32-bit
   263  		{name: "REV16", argLength: 1, reg: gp11, asm: "REV16", earlyOk: true},                          // byte reverse in each 16-bit halfword, 64-bit
   264  		{name: "REV16W", argLength: 1, reg: gp11, asm: "REV16W", earlyOk: true},                        // byte reverse in each 16-bit halfword, 32-bit
   265  		{name: "RBIT", argLength: 1, reg: gp11, asm: "RBIT", earlyOk: true},                            // bit reverse, 64-bit
   266  		{name: "RBITW", argLength: 1, reg: gp11, asm: "RBITW", earlyOk: true},                          // bit reverse, 32-bit
   267  		{name: "CLZ", argLength: 1, reg: gp11, asm: "CLZ", earlyOk: true},                              // count leading zero, 64-bit
   268  		{name: "CLZW", argLength: 1, reg: gp11, asm: "CLZW", earlyOk: true},                            // count leading zero, 32-bit
   269  		{name: "VCNT", argLength: 1, reg: fp11, asm: "VCNT", earlyOk: true},                            // count set bits for each 8-bit unit and store the result in each 8-bit unit
   270  		{name: "VUADDLV", argLength: 1, reg: fp11, asm: "VUADDLV", earlyOk: true},                      // unsigned sum of eight bytes in a 64-bit value, zero extended to 64-bit.
   271  		{name: "LoweredRound32F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   272  		{name: "LoweredRound64F", argLength: 1, reg: fp11, resultInArg0: true, zeroWidth: true, earlyOk: true},
   273  
   274  		// 3-operand, the addend comes first
   275  		{name: "FMADDS", argLength: 3, reg: fp31, asm: "FMADDS", earlyOk: true},   // +arg0 + (arg1 * arg2)
   276  		{name: "FMADDD", argLength: 3, reg: fp31, asm: "FMADDD", earlyOk: true},   // +arg0 + (arg1 * arg2)
   277  		{name: "FNMADDS", argLength: 3, reg: fp31, asm: "FNMADDS", earlyOk: true}, // -arg0 - (arg1 * arg2)
   278  		{name: "FNMADDD", argLength: 3, reg: fp31, asm: "FNMADDD", earlyOk: true}, // -arg0 - (arg1 * arg2)
   279  		{name: "FMSUBS", argLength: 3, reg: fp31, asm: "FMSUBS", earlyOk: true},   // +arg0 - (arg1 * arg2)
   280  		{name: "FMSUBD", argLength: 3, reg: fp31, asm: "FMSUBD", earlyOk: true},   // +arg0 - (arg1 * arg2)
   281  		{name: "FNMSUBS", argLength: 3, reg: fp31, asm: "FNMSUBS", earlyOk: true}, // -arg0 + (arg1 * arg2)
   282  		{name: "FNMSUBD", argLength: 3, reg: fp31, asm: "FNMSUBD", earlyOk: true}, // -arg0 + (arg1 * arg2)
   283  		{name: "MADD", argLength: 3, reg: gp31, asm: "MADD", earlyOk: true},       // +arg0 + (arg1 * arg2)
   284  		{name: "MADDW", argLength: 3, reg: gp31, asm: "MADDW", earlyOk: true},     // +arg0 + (arg1 * arg2), 32-bit
   285  		{name: "MSUB", argLength: 3, reg: gp31, asm: "MSUB", earlyOk: true},       // +arg0 - (arg1 * arg2)
   286  		{name: "MSUBW", argLength: 3, reg: gp31, asm: "MSUBW", earlyOk: true},     // +arg0 - (arg1 * arg2), 32-bit
   287  
   288  		// shifts
   289  		{name: "SLL", argLength: 2, reg: gp21, asm: "LSL", earlyOk: true},                        // arg0 << arg1, shift amount is mod 64
   290  		{name: "SLLconst", argLength: 1, reg: gp11, asm: "LSL", aux: "Int64", earlyOk: true},     // arg0 << auxInt, auxInt should be in the range 0 to 63.
   291  		{name: "SRL", argLength: 2, reg: gp21, asm: "LSR", earlyOk: true},                        // arg0 >> arg1, unsigned, shift amount is mod 64
   292  		{name: "SRLconst", argLength: 1, reg: gp11, asm: "LSR", aux: "Int64", earlyOk: true},     // arg0 >> auxInt, unsigned, auxInt should be in the range 0 to 63.
   293  		{name: "SRA", argLength: 2, reg: gp21, asm: "ASR", earlyOk: true},                        // arg0 >> arg1, signed, shift amount is mod 64
   294  		{name: "SRAconst", argLength: 1, reg: gp11, asm: "ASR", aux: "Int64", earlyOk: true},     // arg0 >> auxInt, signed, auxInt should be in the range 0 to 63.
   295  		{name: "ROR", argLength: 2, reg: gp21, asm: "ROR", earlyOk: true},                        // arg0 right rotate by (arg1 mod 64) bits
   296  		{name: "RORW", argLength: 2, reg: gp21, asm: "RORW", earlyOk: true},                      // arg0 right rotate by (arg1 mod 32) bits
   297  		{name: "RORconst", argLength: 1, reg: gp11, asm: "ROR", aux: "Int64", earlyOk: true},     // arg0 right rotate by auxInt bits, auxInt should be in the range 0 to 63.
   298  		{name: "RORWconst", argLength: 1, reg: gp11, asm: "RORW", aux: "Int64", earlyOk: true},   // uint32(arg0) right rotate by auxInt bits, auxInt should be in the range 0 to 31.
   299  		{name: "EXTRconst", argLength: 2, reg: gp21, asm: "EXTR", aux: "Int64", earlyOk: true},   // extract 64 bits from arg0:arg1 starting at lsb auxInt, auxInt should be in the range 0 to 63.
   300  		{name: "EXTRWconst", argLength: 2, reg: gp21, asm: "EXTRW", aux: "Int64", earlyOk: true}, // extract 32 bits from arg0[31:0]:arg1[31:0] starting at lsb auxInt and zero top 32 bits, auxInt should be in the range 0 to 31.
   301  
   302  		// comparisons
   303  		{name: "CMP", argLength: 2, reg: gp2flags, asm: "CMP", typ: "Flags"},                      // arg0 compare to arg1
   304  		{name: "CMPconst", argLength: 1, reg: gp1flags, asm: "CMP", aux: "Int64", typ: "Flags"},   // arg0 compare to auxInt
   305  		{name: "CMPW", argLength: 2, reg: gp2flags, asm: "CMPW", typ: "Flags"},                    // arg0 compare to arg1, 32 bit
   306  		{name: "CMPWconst", argLength: 1, reg: gp1flags, asm: "CMPW", aux: "Int32", typ: "Flags"}, // arg0 compare to auxInt, 32 bit
   307  		{name: "CMN", argLength: 2, reg: gp2flags, asm: "CMN", typ: "Flags", commutative: true},   // arg0 compare to -arg1, provided arg1 is not 1<<63
   308  		{name: "CMNconst", argLength: 1, reg: gp1flags, asm: "CMN", aux: "Int64", typ: "Flags"},   // arg0 compare to -auxInt
   309  		{name: "CMNW", argLength: 2, reg: gp2flags, asm: "CMNW", typ: "Flags", commutative: true}, // arg0 compare to -arg1, 32 bit, provided arg1 is not 1<<31
   310  		{name: "CMNWconst", argLength: 1, reg: gp1flags, asm: "CMNW", aux: "Int32", typ: "Flags"}, // arg0 compare to -auxInt, 32 bit
   311  		{name: "TST", argLength: 2, reg: gp2flags, asm: "TST", typ: "Flags", commutative: true},   // arg0 & arg1 compare to 0
   312  		{name: "TSTconst", argLength: 1, reg: gp1flags, asm: "TST", aux: "Int64", typ: "Flags"},   // arg0 & auxInt compare to 0
   313  		{name: "TSTW", argLength: 2, reg: gp2flags, asm: "TSTW", typ: "Flags", commutative: true}, // arg0 & arg1 compare to 0, 32 bit
   314  		{name: "TSTWconst", argLength: 1, reg: gp1flags, asm: "TSTW", aux: "Int32", typ: "Flags"}, // arg0 & auxInt compare to 0, 32 bit
   315  		{name: "FCMPS", argLength: 2, reg: fp2flags, asm: "FCMPS", typ: "Flags"},                  // arg0 compare to arg1, float32
   316  		{name: "FCMPD", argLength: 2, reg: fp2flags, asm: "FCMPD", typ: "Flags"},                  // arg0 compare to arg1, float64
   317  		{name: "FCMPS0", argLength: 1, reg: fp1flags, asm: "FCMPS", typ: "Flags"},                 // arg0 compare to 0, float32
   318  		{name: "FCMPD0", argLength: 1, reg: fp1flags, asm: "FCMPD", typ: "Flags"},                 // arg0 compare to 0, float64
   319  
   320  		// shifted ops
   321  		{name: "MVNshiftLL", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0<<auxInt), auxInt should be in the range 0 to 63.
   322  		{name: "MVNshiftRL", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   323  		{name: "MVNshiftRA", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   324  		{name: "MVNshiftRO", argLength: 1, reg: gp11, asm: "MVN", aux: "Int64", earlyOk: true},    // ^(arg0 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   325  		{name: "NEGshiftLL", argLength: 1, reg: gp11, asm: "NEG", aux: "Int64", earlyOk: true},    // -(arg0<<auxInt), auxInt should be in the range 0 to 63.
   326  		{name: "NEGshiftRL", argLength: 1, reg: gp11, asm: "NEG", aux: "Int64", earlyOk: true},    // -(arg0>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   327  		{name: "NEGshiftRA", argLength: 1, reg: gp11, asm: "NEG", aux: "Int64", earlyOk: true},    // -(arg0>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   328  		{name: "ADDshiftLL", argLength: 2, reg: gp21, asm: "ADD", aux: "Int64", earlyOk: true},    // arg0 + arg1<<auxInt, auxInt should be in the range 0 to 63.
   329  		{name: "ADDshiftRL", argLength: 2, reg: gp21, asm: "ADD", aux: "Int64", earlyOk: true},    // arg0 + arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   330  		{name: "ADDshiftRA", argLength: 2, reg: gp21, asm: "ADD", aux: "Int64", earlyOk: true},    // arg0 + arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   331  		{name: "SUBshiftLL", argLength: 2, reg: gp21, asm: "SUB", aux: "Int64", earlyOk: true},    // arg0 - arg1<<auxInt, auxInt should be in the range 0 to 63.
   332  		{name: "SUBshiftRL", argLength: 2, reg: gp21, asm: "SUB", aux: "Int64", earlyOk: true},    // arg0 - arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   333  		{name: "SUBshiftRA", argLength: 2, reg: gp21, asm: "SUB", aux: "Int64", earlyOk: true},    // arg0 - arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   334  		{name: "ANDshiftLL", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1<<auxInt), auxInt should be in the range 0 to 63.
   335  		{name: "ANDshiftRL", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   336  		{name: "ANDshiftRA", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   337  		{name: "ANDshiftRO", argLength: 2, reg: gp21, asm: "AND", aux: "Int64", earlyOk: true},    // arg0 & (arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   338  		{name: "ORshiftLL", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1<<auxInt, auxInt should be in the range 0 to 63.
   339  		{name: "ORshiftRL", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   340  		{name: "ORshiftRA", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   341  		{name: "ORshiftRO", argLength: 2, reg: gp21, asm: "ORR", aux: "Int64", earlyOk: true},     // arg0 | arg1 ROR auxInt, signed shift, auxInt should be in the range 0 to 63.
   342  		{name: "XORshiftLL", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1<<auxInt, auxInt should be in the range 0 to 63.
   343  		{name: "XORshiftRL", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   344  		{name: "XORshiftRA", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   345  		{name: "XORshiftRO", argLength: 2, reg: gp21, asm: "EOR", aux: "Int64", earlyOk: true},    // arg0 ^ arg1 ROR auxInt, signed shift, auxInt should be in the range 0 to 63.
   346  		{name: "BICshiftLL", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1<<auxInt), auxInt should be in the range 0 to 63.
   347  		{name: "BICshiftRL", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   348  		{name: "BICshiftRA", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   349  		{name: "BICshiftRO", argLength: 2, reg: gp21, asm: "BIC", aux: "Int64", earlyOk: true},    // arg0 &^ (arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   350  		{name: "EONshiftLL", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1<<auxInt), auxInt should be in the range 0 to 63.
   351  		{name: "EONshiftRL", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   352  		{name: "EONshiftRA", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   353  		{name: "EONshiftRO", argLength: 2, reg: gp21, asm: "EON", aux: "Int64", earlyOk: true},    // arg0 ^ ^(arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   354  		{name: "ORNshiftLL", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1<<auxInt), auxInt should be in the range 0 to 63.
   355  		{name: "ORNshiftRL", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1>>auxInt), unsigned shift, auxInt should be in the range 0 to 63.
   356  		{name: "ORNshiftRA", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1>>auxInt), signed shift, auxInt should be in the range 0 to 63.
   357  		{name: "ORNshiftRO", argLength: 2, reg: gp21, asm: "ORN", aux: "Int64", earlyOk: true},    // arg0 | ^(arg1 ROR auxInt), signed shift, auxInt should be in the range 0 to 63.
   358  		{name: "CMPshiftLL", argLength: 2, reg: gp2flags, asm: "CMP", aux: "Int64", typ: "Flags"}, // arg0 compare to arg1<<auxInt, auxInt should be in the range 0 to 63.
   359  		{name: "CMPshiftRL", argLength: 2, reg: gp2flags, asm: "CMP", aux: "Int64", typ: "Flags"}, // arg0 compare to arg1>>auxInt, unsigned shift, auxInt should be in the range 0 to 63.
   360  		{name: "CMPshiftRA", argLength: 2, reg: gp2flags, asm: "CMP", aux: "Int64", typ: "Flags"}, // arg0 compare to arg1>>auxInt, signed shift, auxInt should be in the range 0 to 63.
   361  		{name: "CMNshiftLL", argLength: 2, reg: gp2flags, asm: "CMN", aux: "Int64", typ: "Flags"}, // (arg0 + arg1<<auxInt) compare to 0, auxInt should be in the range 0 to 63.
   362  		{name: "CMNshiftRL", argLength: 2, reg: gp2flags, asm: "CMN", aux: "Int64", typ: "Flags"}, // (arg0 + arg1>>auxInt) compare to 0, unsigned shift, auxInt should be in the range 0 to 63.
   363  		{name: "CMNshiftRA", argLength: 2, reg: gp2flags, asm: "CMN", aux: "Int64", typ: "Flags"}, // (arg0 + arg1>>auxInt) compare to 0, signed shift, auxInt should be in the range 0 to 63.
   364  		{name: "TSTshiftLL", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1<<auxInt) compare to 0, auxInt should be in the range 0 to 63.
   365  		{name: "TSTshiftRL", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1>>auxInt) compare to 0, unsigned shift, auxInt should be in the range 0 to 63.
   366  		{name: "TSTshiftRA", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1>>auxInt) compare to 0, signed shift, auxInt should be in the range 0 to 63.
   367  		{name: "TSTshiftRO", argLength: 2, reg: gp2flags, asm: "TST", aux: "Int64", typ: "Flags"}, // (arg0 & arg1 ROR auxInt) compare to 0, signed shift, auxInt should be in the range 0 to 63.
   368  
   369  		// bitfield ops
   370  		// for all bitfield ops lsb is auxInt>>8, width is auxInt&0xff
   371  		// insert low width bits of arg1 into the result starting at bit lsb, copy other bits from arg0
   372  		{name: "BFI", argLength: 2, reg: gp21nog, asm: "BFI", aux: "ARM64BitField", resultInArg0: true, earlyOk: true},
   373  		// extract width bits of arg1 starting at bit lsb and insert at low end of result, copy other bits from arg0
   374  		{name: "BFXIL", argLength: 2, reg: gp21nog, asm: "BFXIL", aux: "ARM64BitField", resultInArg0: true, earlyOk: true},
   375  		// insert low width bits of arg0 into the result starting at bit lsb, bits to the left of the inserted bit field are set to the high/sign bit of the inserted bit field, bits to the right are zeroed
   376  		{name: "SBFIZ", argLength: 1, reg: gp11, asm: "SBFIZ", aux: "ARM64BitField", earlyOk: true},
   377  		// extract width bits of arg0 starting at bit lsb and insert at low end of result, remaining high bits are set to the high/sign bit of the extracted bitfield
   378  		{name: "SBFX", argLength: 1, reg: gp11, asm: "SBFX", aux: "ARM64BitField", earlyOk: true},
   379  		// insert low width bits of arg0 into the result starting at bit lsb, bits to the left and right of the inserted bit field are zeroed
   380  		{name: "UBFIZ", argLength: 1, reg: gp11, asm: "UBFIZ", aux: "ARM64BitField", earlyOk: true},
   381  		// extract width bits of arg0 starting at bit lsb and insert at low end of result, remaining high bits are zeroed
   382  		{name: "UBFX", argLength: 1, reg: gp11, asm: "UBFX", aux: "ARM64BitField", earlyOk: true},
   383  
   384  		// moves
   385  		{name: "MOVDconst", argLength: 0, reg: gp01, aux: "Int64", asm: "MOVD", typ: "UInt64", rematerializeable: true, earlyOk: true},      // 64 bits from auxint
   386  		{name: "FMOVSconst", argLength: 0, reg: fp01, aux: "Float64", asm: "FMOVS", typ: "Float32", rematerializeable: true, earlyOk: true}, // auxint as 64-bit float, convert to 32-bit float
   387  		{name: "FMOVDconst", argLength: 0, reg: fp01, aux: "Float64", asm: "FMOVD", typ: "Float64", rematerializeable: true, earlyOk: true}, // auxint as 64-bit float
   388  
   389  		{name: "MOVDaddr", argLength: 1, reg: regInfo{inputs: []regMask{buildReg("SP").union(buildReg("SB"))}, outputs: []regMask{gp}}, aux: "SymOff", asm: "MOVD", rematerializeable: true, symEffect: "Addr", earlyOk: true}, // arg0 + auxInt + aux.(*gc.Sym), arg0=SP/SB
   390  
   391  		{name: "MOVBload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVB", typ: "Int8", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},      // load from arg0 + auxInt + aux.  arg1=mem.
   392  		{name: "MOVBUload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVBU", typ: "UInt8", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},   // load from arg0 + auxInt + aux.  arg1=mem.
   393  		{name: "MOVHload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVH", typ: "Int16", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},     // load from arg0 + auxInt + aux.  arg1=mem.
   394  		{name: "MOVHUload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVHU", typ: "UInt16", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},  // load from arg0 + auxInt + aux.  arg1=mem.
   395  		{name: "MOVWload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVW", typ: "Int32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},     // load from arg0 + auxInt + aux.  arg1=mem.
   396  		{name: "MOVWUload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVWU", typ: "UInt32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},  // load from arg0 + auxInt + aux.  arg1=mem.
   397  		{name: "MOVDload", argLength: 2, reg: gpload, aux: "SymOff", asm: "MOVD", typ: "UInt64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},    // load from arg0 + auxInt + aux.  arg1=mem.
   398  		{name: "FMOVSload", argLength: 2, reg: fpload, aux: "SymOff", asm: "FMOVS", typ: "Float32", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // load from arg0 + auxInt + aux.  arg1=mem.
   399  		{name: "FMOVDload", argLength: 2, reg: fpload, aux: "SymOff", asm: "FMOVD", typ: "Float64", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // load from arg0 + auxInt + aux.  arg1=mem.
   400  		{name: "FMOVQload", argLength: 2, reg: fpload, aux: "SymOff", asm: "FMOVQ", typ: "Vec128", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},  // load from arg0 + auxInt + aux.  arg1=mem.
   401  
   402  		// LDP instructions load the contents of two adjacent locations in memory into registers.
   403  		// Address to start loading is addr = arg0 + auxInt + aux.
   404  		// x := *(*T)(addr)
   405  		// y := *(*T)(addr+sizeof(T))
   406  		// arg1=mem
   407  		// Returns the tuple <x,y>.
   408  		{name: "LDP", argLength: 2, reg: gpload2, aux: "SymOff", asm: "LDP", typ: "(UInt64,UInt64)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},       // T=int64 (gp reg destination)
   409  		{name: "LDPW", argLength: 2, reg: gpload2, aux: "SymOff", asm: "LDPW", typ: "(UInt32,UInt32)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},     // T=int32 (gp reg destination) unsigned extension
   410  		{name: "LDPSW", argLength: 2, reg: gpload2, aux: "SymOff", asm: "LDPSW", typ: "(Int32,Int32)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},     // T=int32 (gp reg destination) signed extension
   411  		{name: "FLDPD", argLength: 2, reg: fpload2, aux: "SymOff", asm: "FLDPD", typ: "(Float64,Float64)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // T=float64 (fp reg destination)
   412  		{name: "FLDPS", argLength: 2, reg: fpload2, aux: "SymOff", asm: "FLDPS", typ: "(Float32,Float32)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true}, // T=float32 (fp reg destination)
   413  		{name: "FLDPQ", argLength: 2, reg: fpload2, aux: "SymOff", asm: "FLDPQ", typ: "(Vec128,Vec128)", faultOnNilArg0: true, symEffect: "Read", addrSinkArg0: true},   // T=vec128 (fp reg destination)
   414  
   415  		// register indexed load
   416  		{name: "MOVDloadidx", argLength: 3, reg: gp2load, asm: "MOVD", typ: "UInt64", addrSinkArg0: true, addrSinkArg1: true},    // load 64-bit dword from arg0 + arg1, arg2 = mem.
   417  		{name: "MOVWloadidx", argLength: 3, reg: gp2load, asm: "MOVW", typ: "Int32", addrSinkArg0: true, addrSinkArg1: true},     // load 32-bit word from arg0 + arg1, sign-extended to 64-bit, arg2=mem.
   418  		{name: "MOVWUloadidx", argLength: 3, reg: gp2load, asm: "MOVWU", typ: "UInt32", addrSinkArg0: true, addrSinkArg1: true},  // load 32-bit word from arg0 + arg1, zero-extended to 64-bit, arg2=mem.
   419  		{name: "MOVHloadidx", argLength: 3, reg: gp2load, asm: "MOVH", typ: "Int16", addrSinkArg0: true, addrSinkArg1: true},     // load 16-bit word from arg0 + arg1, sign-extended to 64-bit, arg2=mem.
   420  		{name: "MOVHUloadidx", argLength: 3, reg: gp2load, asm: "MOVHU", typ: "UInt16", addrSinkArg0: true, addrSinkArg1: true},  // load 16-bit word from arg0 + arg1, zero-extended to 64-bit, arg2=mem.
   421  		{name: "MOVBloadidx", argLength: 3, reg: gp2load, asm: "MOVB", typ: "Int8", addrSinkArg0: true, addrSinkArg1: true},      // load 8-bit word from arg0 + arg1, sign-extended to 64-bit, arg2=mem.
   422  		{name: "MOVBUloadidx", argLength: 3, reg: gp2load, asm: "MOVBU", typ: "UInt8", addrSinkArg0: true, addrSinkArg1: true},   // load 8-bit word from arg0 + arg1, zero-extended to 64-bit, arg2=mem.
   423  		{name: "FMOVSloadidx", argLength: 3, reg: fp2load, asm: "FMOVS", typ: "Float32", addrSinkArg0: true, addrSinkArg1: true}, // load 32-bit float from arg0 + arg1, arg2=mem.
   424  		{name: "FMOVDloadidx", argLength: 3, reg: fp2load, asm: "FMOVD", typ: "Float64", addrSinkArg0: true, addrSinkArg1: true}, // load 64-bit float from arg0 + arg1, arg2=mem.
   425  
   426  		// shifted register indexed load
   427  		{name: "MOVHloadidx2", argLength: 3, reg: gp2load, asm: "MOVH", typ: "Int16", addrSinkArg0: true},     // load 16-bit half-word from arg0 + arg1*2, sign-extended to 64-bit, arg2=mem.
   428  		{name: "MOVHUloadidx2", argLength: 3, reg: gp2load, asm: "MOVHU", typ: "UInt16", addrSinkArg0: true},  // load 16-bit half-word from arg0 + arg1*2, zero-extended to 64-bit, arg2=mem.
   429  		{name: "MOVWloadidx4", argLength: 3, reg: gp2load, asm: "MOVW", typ: "Int32", addrSinkArg0: true},     // load 32-bit word from arg0 + arg1*4, sign-extended to 64-bit, arg2=mem.
   430  		{name: "MOVWUloadidx4", argLength: 3, reg: gp2load, asm: "MOVWU", typ: "UInt32", addrSinkArg0: true},  // load 32-bit word from arg0 + arg1*4, zero-extended to 64-bit, arg2=mem.
   431  		{name: "MOVDloadidx8", argLength: 3, reg: gp2load, asm: "MOVD", typ: "UInt64", addrSinkArg0: true},    // load 64-bit double-word from arg0 + arg1*8, arg2 = mem.
   432  		{name: "FMOVSloadidx4", argLength: 3, reg: fp2load, asm: "FMOVS", typ: "Float32", addrSinkArg0: true}, // load 32-bit float from arg0 + arg1*4, arg2 = mem.
   433  		{name: "FMOVDloadidx8", argLength: 3, reg: fp2load, asm: "FMOVD", typ: "Float64", addrSinkArg0: true}, // load 64-bit float from arg0 + arg1*8, arg2 = mem.
   434  
   435  		{name: "MOVBstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVB", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 1 byte of arg1 to arg0 + auxInt + aux.  arg2=mem.
   436  		{name: "MOVHstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVH", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 2 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   437  		{name: "MOVWstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVW", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 4 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   438  		{name: "MOVDstore", argLength: 3, reg: gpstore, aux: "SymOff", asm: "MOVD", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // store 8 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   439  		{name: "FMOVSstore", argLength: 3, reg: fpstore, aux: "SymOff", asm: "FMOVS", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 4 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   440  		{name: "FMOVDstore", argLength: 3, reg: fpstore, aux: "SymOff", asm: "FMOVD", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 8 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   441  		{name: "FMOVQstore", argLength: 3, reg: fpstore, aux: "SymOff", asm: "FMOVQ", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // store 16 bytes of arg1 to arg0 + auxInt + aux.  arg2=mem.
   442  
   443  		// STP instructions store the contents of two registers to adjacent locations in memory.
   444  		// Address to start storing is addr = arg0 + auxInt + aux.
   445  		// *(*T)(addr) = arg1
   446  		// *(*T)(addr+sizeof(T)) = arg2
   447  		// arg3=mem. Returns mem.
   448  		{name: "STP", argLength: 4, reg: gpstore2, aux: "SymOff", asm: "STP", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},     // T=int64 (gp reg source)
   449  		{name: "STPW", argLength: 4, reg: gpstore2, aux: "SymOff", asm: "STPW", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true},   // T=int32 (gp reg source)
   450  		{name: "FSTPD", argLength: 4, reg: fpstore2, aux: "SymOff", asm: "FSTPD", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // T=float64 (fp reg source)
   451  		{name: "FSTPS", argLength: 4, reg: fpstore2, aux: "SymOff", asm: "FSTPS", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // T=float32 (fp reg source)
   452  		{name: "FSTPQ", argLength: 4, reg: fpstore2, aux: "SymOff", asm: "FSTPQ", typ: "Mem", faultOnNilArg0: true, symEffect: "Write", addrSinkArg0: true}, // T=vec128 (fp reg source)
   453  
   454  		// register indexed store
   455  		{name: "MOVBstoreidx", argLength: 4, reg: gpstore2, asm: "MOVB", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 1 byte of arg2 to arg0 + arg1, arg3 = mem.
   456  		{name: "MOVHstoreidx", argLength: 4, reg: gpstore2, asm: "MOVH", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 2 bytes of arg2 to arg0 + arg1, arg3 = mem.
   457  		{name: "MOVWstoreidx", argLength: 4, reg: gpstore2, asm: "MOVW", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 4 bytes of arg2 to arg0 + arg1, arg3 = mem.
   458  		{name: "MOVDstoreidx", argLength: 4, reg: gpstore2, asm: "MOVD", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true},     // store 8 bytes of arg2 to arg0 + arg1, arg3 = mem.
   459  		{name: "FMOVSstoreidx", argLength: 4, reg: fpstoreidx, asm: "FMOVS", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true}, // store 32-bit float of arg2 to arg0 + arg1, arg3=mem.
   460  		{name: "FMOVDstoreidx", argLength: 4, reg: fpstoreidx, asm: "FMOVD", typ: "Mem", addrSinkArg0: true, addrSinkArg1: true}, // store 64-bit float of arg2 to arg0 + arg1, arg3=mem.
   461  
   462  		// shifted register indexed store
   463  		{name: "MOVHstoreidx2", argLength: 4, reg: gpstore2, asm: "MOVH", typ: "Mem", addrSinkArg0: true},     // store 2 bytes of arg2 to arg0 + arg1*2, arg3 = mem.
   464  		{name: "MOVWstoreidx4", argLength: 4, reg: gpstore2, asm: "MOVW", typ: "Mem", addrSinkArg0: true},     // store 4 bytes of arg2 to arg0 + arg1*4, arg3 = mem.
   465  		{name: "MOVDstoreidx8", argLength: 4, reg: gpstore2, asm: "MOVD", typ: "Mem", addrSinkArg0: true},     // store 8 bytes of arg2 to arg0 + arg1*8, arg3 = mem.
   466  		{name: "FMOVSstoreidx4", argLength: 4, reg: fpstoreidx, asm: "FMOVS", typ: "Mem", addrSinkArg0: true}, // store 32-bit float of arg2 to arg0 + arg1*4, arg3=mem.
   467  		{name: "FMOVDstoreidx8", argLength: 4, reg: fpstoreidx, asm: "FMOVD", typ: "Mem", addrSinkArg0: true}, // store 64-bit float of arg2 to arg0 + arg1*8, arg3=mem.
   468  
   469  		{name: "FMOVDgpfp", argLength: 1, reg: gpfp, asm: "FMOVD", earlyOk: true}, // move int64 to float64 (no conversion)
   470  		{name: "FMOVDfpgp", argLength: 1, reg: fpgp, asm: "FMOVD", earlyOk: true}, // move float64 to int64 (no conversion)
   471  		{name: "FMOVSgpfp", argLength: 1, reg: gpfp, asm: "FMOVS", earlyOk: true}, // move 32bits from int to float reg (no conversion)
   472  		{name: "FMOVSfpgp", argLength: 1, reg: fpgp, asm: "FMOVS", earlyOk: true}, // move 32bits from float to int reg, zero extend (no conversion)
   473  
   474  		// conversions
   475  		{name: "MOVBreg", argLength: 1, reg: gp11, asm: "MOVB", earlyOk: true},   // move from arg0, sign-extended from byte
   476  		{name: "MOVBUreg", argLength: 1, reg: gp11, asm: "MOVBU", earlyOk: true}, // move from arg0, unsign-extended from byte
   477  		{name: "MOVHreg", argLength: 1, reg: gp11, asm: "MOVH", earlyOk: true},   // move from arg0, sign-extended from half
   478  		{name: "MOVHUreg", argLength: 1, reg: gp11, asm: "MOVHU", earlyOk: true}, // move from arg0, unsign-extended from half
   479  		{name: "MOVWreg", argLength: 1, reg: gp11, asm: "MOVW", earlyOk: true},   // move from arg0, sign-extended from word
   480  		{name: "MOVWUreg", argLength: 1, reg: gp11, asm: "MOVWU", earlyOk: true}, // move from arg0, unsign-extended from word
   481  		{name: "MOVDreg", argLength: 1, reg: gp11, asm: "MOVD", earlyOk: true},   // move from arg0
   482  
   483  		{name: "MOVDnop", argLength: 1, reg: regInfo{inputs: []regMask{gp}, outputs: []regMask{gp}}, resultInArg0: true, earlyOk: true}, // nop, return arg0 in same register
   484  
   485  		{name: "SCVTFWS", argLength: 1, reg: gpfp, asm: "SCVTFWS", earlyOk: true},   // int32 -> float32
   486  		{name: "SCVTFWD", argLength: 1, reg: gpfp, asm: "SCVTFWD", earlyOk: true},   // int32 -> float64
   487  		{name: "UCVTFWS", argLength: 1, reg: gpfp, asm: "UCVTFWS", earlyOk: true},   // uint32 -> float32
   488  		{name: "UCVTFWD", argLength: 1, reg: gpfp, asm: "UCVTFWD", earlyOk: true},   // uint32 -> float64
   489  		{name: "SCVTFS", argLength: 1, reg: gpfp, asm: "SCVTFS", earlyOk: true},     // int64 -> float32
   490  		{name: "SCVTFD", argLength: 1, reg: gpfp, asm: "SCVTFD", earlyOk: true},     // int64 -> float64
   491  		{name: "UCVTFS", argLength: 1, reg: gpfp, asm: "UCVTFS", earlyOk: true},     // uint64 -> float32
   492  		{name: "UCVTFD", argLength: 1, reg: gpfp, asm: "UCVTFD", earlyOk: true},     // uint64 -> float64
   493  		{name: "FCVTZSSW", argLength: 1, reg: fpgp, asm: "FCVTZSSW", earlyOk: true}, // float32 -> int32
   494  		{name: "FCVTZSDW", argLength: 1, reg: fpgp, asm: "FCVTZSDW", earlyOk: true}, // float64 -> int32
   495  		{name: "FCVTZUSW", argLength: 1, reg: fpgp, asm: "FCVTZUSW", earlyOk: true}, // float32 -> uint32
   496  		{name: "FCVTZUDW", argLength: 1, reg: fpgp, asm: "FCVTZUDW", earlyOk: true}, // float64 -> uint32
   497  		{name: "FCVTZSS", argLength: 1, reg: fpgp, asm: "FCVTZSS", earlyOk: true},   // float32 -> int64
   498  		{name: "FCVTZSD", argLength: 1, reg: fpgp, asm: "FCVTZSD", earlyOk: true},   // float64 -> int64
   499  		{name: "FCVTZUS", argLength: 1, reg: fpgp, asm: "FCVTZUS", earlyOk: true},   // float32 -> uint64
   500  		{name: "FCVTZUD", argLength: 1, reg: fpgp, asm: "FCVTZUD", earlyOk: true},   // float64 -> uint64
   501  		{name: "FCVTSD", argLength: 1, reg: fp11, asm: "FCVTSD", earlyOk: true},     // float32 -> float64
   502  		{name: "FCVTDS", argLength: 1, reg: fp11, asm: "FCVTDS", earlyOk: true},     // float64 -> float32
   503  
   504  		// 64-bit floating-point round to integers in 64-bit FP format
   505  		{name: "FRINTAD", argLength: 1, reg: fp11, asm: "FRINTAD", earlyOk: true}, // Round (ties Away from zero; 0.5 -> 1, -0.5 -> -1)
   506  		{name: "FRINTMD", argLength: 1, reg: fp11, asm: "FRINTMD", earlyOk: true}, // Floor (towards Minus; 0.5 -> 0, -0.5 -> -1)
   507  		{name: "FRINTND", argLength: 1, reg: fp11, asm: "FRINTND", earlyOk: true}, // Round (ties to even; ; 0.5 -> 0, 1.5 -> 2)
   508  		{name: "FRINTPD", argLength: 1, reg: fp11, asm: "FRINTPD", earlyOk: true}, // Ceil (towards Positive; 0.5 -> 1, -0.5 -> 0)
   509  		{name: "FRINTZD", argLength: 1, reg: fp11, asm: "FRINTZD", earlyOk: true}, // Trunc (towards Zero; 0.5 -> 0, -0.5 -> 0))
   510  		// 32-bit floating-point round to integers in 32-bit FP format
   511  		{name: "FRINTAS", argLength: 1, reg: fp11, asm: "FRINTAS", earlyOk: true}, // Round (ties Away from zero; 0.5 -> 1, -0.5 -> -1)
   512  		{name: "FRINTMS", argLength: 1, reg: fp11, asm: "FRINTMS", earlyOk: true}, // Floor (towards Minus; 0.5 -> 0, -0.5 -> -1)
   513  		{name: "FRINTNS", argLength: 1, reg: fp11, asm: "FRINTNS", earlyOk: true}, // Round (ties to even; ; 0.5 -> 0, 1.5 -> 2)
   514  		{name: "FRINTPS", argLength: 1, reg: fp11, asm: "FRINTPS", earlyOk: true}, // Ceil (towards Positive; 0.5 -> 1, -0.5 -> 0)
   515  		{name: "FRINTZS", argLength: 1, reg: fp11, asm: "FRINTZS", earlyOk: true}, // Trunc (towards Zero; 0.5 -> 0, -0.5 -> 0))
   516  
   517  		// conditional instructions; auxint is
   518  		// one of the arm64 comparison pseudo-ops (LessThan, LessThanU, etc.)
   519  		{name: "CSEL", argLength: 3, reg: gp2flags1, asm: "CSEL", aux: "CCop", earlyOk: true},   // auxint(flags) ? arg0 : arg1
   520  		{name: "CSEL0", argLength: 2, reg: gp1flags1, asm: "CSEL", aux: "CCop", earlyOk: true},  // auxint(flags) ? arg0 : 0
   521  		{name: "CSINC", argLength: 3, reg: gp2flags1, asm: "CSINC", aux: "CCop", earlyOk: true}, // auxint(flags) ? arg0 : arg1 + 1
   522  		{name: "CSINV", argLength: 3, reg: gp2flags1, asm: "CSINV", aux: "CCop", earlyOk: true}, // auxint(flags) ? arg0 : ^arg1
   523  		{name: "CSNEG", argLength: 3, reg: gp2flags1, asm: "CSNEG", aux: "CCop", earlyOk: true}, // auxint(flags) ? arg0 : -arg1
   524  		{name: "CSETM", argLength: 1, reg: readflags, asm: "CSETM", aux: "CCop", earlyOk: true}, // auxint(flags) ? -1 : 0
   525  
   526  		// conditional comparison instructions; auxint is
   527  		// combination of Cond, Nzcv and optional ConstValue
   528  		// Behavior:
   529  		//   If the condition 'Cond' evaluates to true against current flags,
   530  		//   flags are set to the result of the comparison operation.
   531  		//   Otherwise, flags are set to the fallback value 'Nzcv'.
   532  		{name: "CCMP", argLength: 3, reg: gp2flagsflags, asm: "CCMP", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMP arg0 arg1 else flags = Nzcv
   533  		{name: "CCMN", argLength: 3, reg: gp2flagsflags, asm: "CCMN", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMN arg0 arg1 else flags = Nzcv
   534  		{name: "CCMPconst", argLength: 2, reg: gp1flagsflags, asm: "CCMP", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CMPconst [ConstValue] arg0 else flags = Nzcv
   535  		{name: "CCMNconst", argLength: 2, reg: gp1flagsflags, asm: "CCMN", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CMNconst [ConstValue] arg0 else flags = Nzcv
   536  
   537  		{name: "CCMPW", argLength: 3, reg: gp2flagsflags, asm: "CCMPW", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMPW arg0 arg1 else flags = Nzcv
   538  		{name: "CCMNW", argLength: 3, reg: gp2flagsflags, asm: "CCMNW", aux: "ARM64ConditionalParams", typ: "Flags"},      // If Cond then flags = CMNW arg0 arg1 else flags = Nzcv
   539  		{name: "CCMPWconst", argLength: 2, reg: gp1flagsflags, asm: "CCMPW", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CCMPWconst [ConstValue] arg0 else flags = Nzcv
   540  		{name: "CCMNWconst", argLength: 2, reg: gp1flagsflags, asm: "CCMNW", aux: "ARM64ConditionalParams", typ: "Flags"}, // If Cond then flags = CCMNWconst [ConstValue] arg0 else flags = Nzcv
   541  
   542  		// function calls
   543  		{name: "CALLstatic", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                                       // call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
   544  		{name: "CALLtail", argLength: -1, reg: regInfo{clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},                                         // tail call static function aux.(*obj.LSym).  last arg=mem, auxint=argsize, returns mem
   545  		{name: "CALLtailinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true, tailCall: true},             // tail call fn by pointer. arg0=codeptr, last arg=mem, auxint=argsize, returns mem
   546  		{name: "CALLclosure", argLength: -1, reg: regInfo{inputs: []regMask{gpsp, buildReg("R26"), regMask{}}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true}, // call function via closure.  arg0=codeptr, arg1=closure, last arg=mem, auxint=argsize, returns mem
   547  		{name: "CALLinter", argLength: -1, reg: regInfo{inputs: []regMask{gp}, clobbers: callerSave}, aux: "CallOff", clobberFlags: true, call: true},                                 // call fn by pointer.  arg0=codeptr, last arg=mem, auxint=argsize, returns mem
   548  
   549  		// pseudo-ops
   550  		{name: "LoweredNilCheck", argLength: 2, reg: regInfo{inputs: []regMask{gpg}}, nilCheck: true, faultOnNilArg0: true},                                                                                                                                                      // panic if arg0 is nil.  arg1=mem.
   551  		{name: "LoweredMemEq", argLength: 4, reg: regInfo{inputs: []regMask{buildReg("R0"), buildReg("R1"), buildReg("R2")}, outputs: []regMask{buildReg("R0")}, clobbers: callerSave}, typ: "Bool", faultOnNilArg0: true, faultOnNilArg1: true, clobberFlags: true, call: true}, // arg0, arg1 - pointers to memory, arg2=size, arg3=mem.
   552  
   553  		{name: "Equal", argLength: 1, reg: readflags, earlyOk: true},            // bool, true flags encode x==y false otherwise.
   554  		{name: "NotEqual", argLength: 1, reg: readflags, earlyOk: true},         // bool, true flags encode x!=y false otherwise.
   555  		{name: "LessThan", argLength: 1, reg: readflags, earlyOk: true},         // bool, true flags encode signed x<y false otherwise.
   556  		{name: "LessEqual", argLength: 1, reg: readflags, earlyOk: true},        // bool, true flags encode signed x<=y false otherwise.
   557  		{name: "GreaterThan", argLength: 1, reg: readflags, earlyOk: true},      // bool, true flags encode signed x>y false otherwise.
   558  		{name: "GreaterEqual", argLength: 1, reg: readflags, earlyOk: true},     // bool, true flags encode signed x>=y false otherwise.
   559  		{name: "LessThanU", argLength: 1, reg: readflags, earlyOk: true},        // bool, true flags encode unsigned x<y false otherwise.
   560  		{name: "LessEqualU", argLength: 1, reg: readflags, earlyOk: true},       // bool, true flags encode unsigned x<=y false otherwise.
   561  		{name: "GreaterThanU", argLength: 1, reg: readflags, earlyOk: true},     // bool, true flags encode unsigned x>y false otherwise.
   562  		{name: "GreaterEqualU", argLength: 1, reg: readflags, earlyOk: true},    // bool, true flags encode unsigned x>=y false otherwise.
   563  		{name: "LessThanF", argLength: 1, reg: readflags, earlyOk: true},        // bool, true flags encode floating-point x<y false otherwise.
   564  		{name: "LessEqualF", argLength: 1, reg: readflags, earlyOk: true},       // bool, true flags encode floating-point x<=y false otherwise.
   565  		{name: "GreaterThanF", argLength: 1, reg: readflags, earlyOk: true},     // bool, true flags encode floating-point x>y false otherwise.
   566  		{name: "GreaterEqualF", argLength: 1, reg: readflags, earlyOk: true},    // bool, true flags encode floating-point x>=y false otherwise.
   567  		{name: "NotLessThanF", argLength: 1, reg: readflags, earlyOk: true},     // bool, true flags encode floating-point x>=y || x is unordered with y, false otherwise.
   568  		{name: "NotLessEqualF", argLength: 1, reg: readflags, earlyOk: true},    // bool, true flags encode floating-point x>y || x is unordered with y, false otherwise.
   569  		{name: "NotGreaterThanF", argLength: 1, reg: readflags, earlyOk: true},  // bool, true flags encode floating-point x<=y || x is unordered with y, false otherwise.
   570  		{name: "NotGreaterEqualF", argLength: 1, reg: readflags, earlyOk: true}, // bool, true flags encode floating-point x<y || x is unordered with y, false otherwise.
   571  		{name: "LessThanNoov", argLength: 1, reg: readflags, earlyOk: true},     // bool, true flags encode signed x<y but without honoring overflow, false otherwise.
   572  		{name: "GreaterEqualNoov", argLength: 1, reg: readflags, earlyOk: true}, // bool, true flags encode signed x>=y but without honoring overflow, false otherwise.
   573  
   574  		// medium zeroing
   575  		// arg0 = address of memory to zero
   576  		// arg1 = mem
   577  		// auxint = # of bytes to zero
   578  		// returns mem
   579  		{
   580  			name:      "LoweredZero",
   581  			aux:       "Int64",
   582  			argLength: 2,
   583  			reg: regInfo{
   584  				inputs: []regMask{gp},
   585  			},
   586  			faultOnNilArg0: true,
   587  			addrSinkArg0:   true,
   588  		},
   589  
   590  		// large zeroing
   591  		// arg0 = address of memory to zero
   592  		// arg1 = mem
   593  		// auxint = # of bytes to zero
   594  		// returns mem
   595  		{
   596  			name:      "LoweredZeroLoop",
   597  			aux:       "Int64",
   598  			argLength: 2,
   599  			reg: regInfo{
   600  				inputs:       []regMask{gp},
   601  				clobbersArg0: true,
   602  			},
   603  			faultOnNilArg0: true,
   604  			addrSinkArg0:   true,
   605  			needIntTemp:    true,
   606  		},
   607  
   608  		// medium copying
   609  		// arg0 = address of dst memory
   610  		// arg1 = address of src memory
   611  		// arg2 = mem
   612  		// auxint = # of bytes to copy
   613  		// returns mem
   614  		{
   615  			name:      "LoweredMove",
   616  			aux:       "Int64",
   617  			argLength: 3,
   618  			reg: regInfo{
   619  				inputs:   []regMask{gp.minus(r25), gp.minus(r25)},
   620  				clobbers: r25.union(f16to17), // TODO: figure out needIntTemp + x2 for floats
   621  			},
   622  			faultOnNilArg0: true,
   623  			faultOnNilArg1: true,
   624  			addrSinkArg0:   true,
   625  			addrSinkArg1:   true,
   626  		},
   627  
   628  		// large copying
   629  		// arg0 = address of dst memory
   630  		// arg1 = address of src memory
   631  		// arg2 = mem
   632  		// auxint = # of bytes to copy
   633  		// returns mem
   634  		{
   635  			name:      "LoweredMoveLoop",
   636  			aux:       "Int64",
   637  			argLength: 3,
   638  			reg: regInfo{
   639  				inputs:       []regMask{gp.minus(r24to25), gp.minus(r24to25)},
   640  				clobbers:     r24to25.union(f16to17), // TODO: figure out needIntTemp x2 + x2 for floats
   641  				clobbersArg0: true,
   642  				clobbersArg1: true,
   643  			},
   644  			faultOnNilArg0: true,
   645  			faultOnNilArg1: true,
   646  			addrSinkArg0:   true,
   647  			addrSinkArg1:   true,
   648  		},
   649  
   650  		// Scheduler ensures LoweredGetClosurePtr occurs only in entry block,
   651  		// and sorts it to the very beginning of the block to prevent other
   652  		// use of R26 (arm64.REGCTXT, the closure pointer)
   653  		{name: "LoweredGetClosurePtr", reg: regInfo{outputs: []regMask{buildReg("R26")}}, zeroWidth: true},
   654  
   655  		// LoweredGetCallerSP returns the SP of the caller of the current function. arg0=mem
   656  		{name: "LoweredGetCallerSP", argLength: 1, reg: gp01, rematerializeable: true},
   657  
   658  		// LoweredGetCallerPC evaluates to the PC to which its "caller" will return.
   659  		// I.e., if f calls g "calls" sys.GetCallerPC,
   660  		// the result should be the PC within f that g will return to.
   661  		// See runtime/stubs.go for a more detailed discussion.
   662  		{name: "LoweredGetCallerPC", reg: gp01, rematerializeable: true},
   663  
   664  		// Constant flag value.
   665  		// Note: there's an "unordered" outcome for floating-point
   666  		// comparisons, but we don't use such a beast yet.
   667  		// This op is for temporary use by rewrite rules. It
   668  		// cannot appear in the generated assembly.
   669  		{name: "FlagConstant", aux: "FlagConstant"},
   670  
   671  		// (InvertFlags (CMP a b)) == (CMP b a)
   672  		// InvertFlags is a pseudo-op which can't appear in assembly output.
   673  		{name: "InvertFlags", argLength: 1}, // reverse direction of arg0
   674  
   675  		// atomic loads.
   676  		// load from arg0. arg1=mem. auxint must be zero.
   677  		// returns <value,memory> so they can be properly ordered with other loads.
   678  		{name: "LDAR", argLength: 2, reg: gpload, asm: "LDAR", faultOnNilArg0: true},
   679  		{name: "LDARB", argLength: 2, reg: gpload, asm: "LDARB", faultOnNilArg0: true},
   680  		{name: "LDARW", argLength: 2, reg: gpload, asm: "LDARW", faultOnNilArg0: true},
   681  
   682  		// atomic stores.
   683  		// store arg1 to arg0. arg2=mem. returns memory. auxint must be zero.
   684  		{name: "STLRB", argLength: 3, reg: gpstore, asm: "STLRB", faultOnNilArg0: true, hasSideEffects: true},
   685  		{name: "STLR", argLength: 3, reg: gpstore, asm: "STLR", faultOnNilArg0: true, hasSideEffects: true},
   686  		{name: "STLRW", argLength: 3, reg: gpstore, asm: "STLRW", faultOnNilArg0: true, hasSideEffects: true},
   687  
   688  		// atomic exchange.
   689  		// store arg1 to arg0. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   690  		// LDAXR	(Rarg0), Rout
   691  		// STLXR	Rarg1, (Rarg0), Rtmp
   692  		// CBNZ		Rtmp, -2(PC)
   693  		{name: "LoweredAtomicExchange64", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   694  		{name: "LoweredAtomicExchange32", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   695  		{name: "LoweredAtomicExchange8", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   696  
   697  		// atomic exchange variant.
   698  		// store arg1 to arg0. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   699  		// SWPALD	Rarg1, (Rarg0), Rout
   700  		{name: "LoweredAtomicExchange64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   701  		{name: "LoweredAtomicExchange32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   702  		{name: "LoweredAtomicExchange8Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   703  
   704  		// atomic add.
   705  		// *arg0 += arg1. arg2=mem. returns <new content of *arg0, memory>. auxint must be zero.
   706  		// LDAXR	(Rarg0), Rout
   707  		// ADD		Rarg1, Rout
   708  		// STLXR	Rout, (Rarg0), Rtmp
   709  		// CBNZ		Rtmp, -3(PC)
   710  		{name: "LoweredAtomicAdd64", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   711  		{name: "LoweredAtomicAdd32", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   712  
   713  		// atomic add variant.
   714  		// *arg0 += arg1. arg2=mem. returns <new content of *arg0, memory>. auxint must be zero.
   715  		// LDADDAL	(Rarg0), Rarg1, Rout
   716  		// ADD		Rarg1, Rout
   717  		{name: "LoweredAtomicAdd64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   718  		{name: "LoweredAtomicAdd32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   719  
   720  		// atomic compare and swap.
   721  		// arg0 = pointer, arg1 = old value, arg2 = new value, arg3 = memory. auxint must be zero.
   722  		// if *arg0 == arg1 {
   723  		//   *arg0 = arg2
   724  		//   return (true, memory)
   725  		// } else {
   726  		//   return (false, memory)
   727  		// }
   728  		// LDAXR	(Rarg0), Rtmp
   729  		// CMP		Rarg1, Rtmp
   730  		// BNE		3(PC)
   731  		// STLXR	Rarg2, (Rarg0), Rtmp
   732  		// CBNZ		Rtmp, -4(PC)
   733  		// CSET		EQ, Rout
   734  		{name: "LoweredAtomicCas64", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   735  		{name: "LoweredAtomicCas32", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   736  
   737  		// atomic compare and swap variant.
   738  		// arg0 = pointer, arg1 = old value, arg2 = new value, arg3 = memory. auxint must be zero.
   739  		// if *arg0 == arg1 {
   740  		//   *arg0 = arg2
   741  		//   return (true, memory)
   742  		// } else {
   743  		//   return (false, memory)
   744  		// }
   745  		// MOV  	Rarg1, Rtmp
   746  		// CASAL	Rtmp, (Rarg0), Rarg2
   747  		// CMP  	Rarg1, Rtmp
   748  		// CSET 	EQ, Rout
   749  		{name: "LoweredAtomicCas64Variant", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   750  		{name: "LoweredAtomicCas32Variant", argLength: 4, reg: gpcas, resultNotInArgs: true, clobberFlags: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   751  
   752  		// atomic and/or.
   753  		// *arg0 &= (|=) arg1. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   754  		// LDAXR	(Rarg0), Rout
   755  		// AND/OR	Rarg1, Rout, tempReg
   756  		// STLXR	tempReg, (Rarg0), Rtmp
   757  		// CBNZ		Rtmp, -3(PC)
   758  		{name: "LoweredAtomicAnd8", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "AND", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   759  		{name: "LoweredAtomicOr8", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "ORR", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   760  		{name: "LoweredAtomicAnd64", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "AND", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   761  		{name: "LoweredAtomicOr64", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "ORR", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   762  		{name: "LoweredAtomicAnd32", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "AND", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   763  		{name: "LoweredAtomicOr32", argLength: 3, reg: gpxchg, resultNotInArgs: true, asm: "ORR", faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true, needIntTemp: true},
   764  
   765  		// atomic and/or variant.
   766  		// *arg0 &= (|=) arg1. arg2=mem. returns <old content of *arg0, memory>. auxint must be zero.
   767  		//   AND:
   768  		// MNV       Rarg1, Rtemp
   769  		// LDANDALB  Rtemp, (Rarg0), Rout
   770  		//   OR:
   771  		// LDORALB  Rarg1, (Rarg0), Rout
   772  		{name: "LoweredAtomicAnd8Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   773  		{name: "LoweredAtomicOr8Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   774  		{name: "LoweredAtomicAnd64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   775  		{name: "LoweredAtomicOr64Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   776  		{name: "LoweredAtomicAnd32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true, unsafePoint: true},
   777  		{name: "LoweredAtomicOr32Variant", argLength: 3, reg: gpxchg, resultNotInArgs: true, faultOnNilArg0: true, hasSideEffects: true},
   778  
   779  		// LoweredWB invokes runtime.gcWriteBarrier. arg0=mem, auxint=# of buffer entries needed
   780  		// It saves all GP registers if necessary,
   781  		// but clobbers R30 (LR) because it's a call.
   782  		// R16 and R17 may be clobbered by linker trampoline.
   783  		// Returns a pointer to a write barrier buffer in R25.
   784  		{name: "LoweredWB", argLength: 1, reg: regInfo{clobbers: callerSave.minus(gpg).union(buildReg("R16 R17 R30")), outputs: []regMask{buildReg("R25")}}, clobberFlags: true, aux: "Int64"},
   785  
   786  		// LoweredPanicBoundsRR takes x and y, two values that caused a bounds check to fail.
   787  		// the RC and CR versions are used when one of the arguments is a constant. CC is used
   788  		// when both are constant (normally both 0, as prove derives the fact that a [0] bounds
   789  		// failure means the length must have also been 0).
   790  		// AuxInt contains a report code (see PanicBounds in genericOps.go).
   791  		{name: "LoweredPanicBoundsRR", argLength: 3, aux: "Int64", reg: regInfo{inputs: []regMask{first16, first16}}, typ: "Mem", call: true}, // arg0=x, arg1=y, arg2=mem, returns memory.
   792  		{name: "LoweredPanicBoundsRC", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{first16}}, typ: "Mem", call: true},   // arg0=x, arg1=mem, returns memory.
   793  		{name: "LoweredPanicBoundsCR", argLength: 2, aux: "PanicBoundsC", reg: regInfo{inputs: []regMask{first16}}, typ: "Mem", call: true},   // arg0=y, arg1=mem, returns memory.
   794  		{name: "LoweredPanicBoundsCC", argLength: 1, aux: "PanicBoundsCC", reg: regInfo{}, typ: "Mem", call: true},                            // arg0=mem, returns memory.
   795  
   796  		// Prefetch instruction
   797  		// Do prefetch arg0 address with option aux. arg0=addr, arg1=memory, aux=option.
   798  		{name: "PRFM", argLength: 2, aux: "Int64", reg: prefreg, asm: "PRFM", hasSideEffects: true},
   799  
   800  		// Publication barrier
   801  		{name: "DMB", argLength: 1, aux: "Int64", asm: "DMB", hasSideEffects: true}, // Do data barrier. arg0=memory, aux=option.
   802  		{name: "ZERO", zeroWidth: true, fixedReg: true, earlyOk: true},              // reads-as-zero register
   803  
   804  		// Broadcast constant to each lane of a SIMD register. aux=constant.
   805  		// TODO: add the other arrangements after assembler supports them, to be used in simdgen-generated opt rules.
   806  		{name: "VMOVI16B", argLength: 0, reg: fp01, asm: "VMOVI", aux: "UInt8", commutative: false, typ: "Vec128", resultInArg0: false},
   807  	}
   808  
   809  	blocks := []blockData{
   810  		{name: "EQ", controls: 1},
   811  		{name: "NE", controls: 1},
   812  		{name: "LT", controls: 1},
   813  		{name: "LE", controls: 1},
   814  		{name: "GT", controls: 1},
   815  		{name: "GE", controls: 1},
   816  		{name: "ULT", controls: 1},
   817  		{name: "ULE", controls: 1},
   818  		{name: "UGT", controls: 1},
   819  		{name: "UGE", controls: 1},
   820  		{name: "Z", controls: 1},                  // Control == 0 (take a register instead of flags)
   821  		{name: "NZ", controls: 1},                 // Control != 0
   822  		{name: "ZW", controls: 1},                 // Control == 0, 32-bit
   823  		{name: "NZW", controls: 1},                // Control != 0, 32-bit
   824  		{name: "TBZ", controls: 1, aux: "Int64"},  // Control & (1 << AuxInt) == 0
   825  		{name: "TBNZ", controls: 1, aux: "Int64"}, // Control & (1 << AuxInt) != 0
   826  		{name: "FLT", controls: 1},
   827  		{name: "FLE", controls: 1},
   828  		{name: "FGT", controls: 1},
   829  		{name: "FGE", controls: 1},
   830  		{name: "LTnoov", controls: 1}, // 'LT' but without honoring overflow
   831  		{name: "LEnoov", controls: 1}, // 'LE' but without honoring overflow
   832  		{name: "GTnoov", controls: 1}, // 'GT' but without honoring overflow
   833  		{name: "GEnoov", controls: 1}, // 'GE' but without honoring overflow
   834  
   835  		// JUMPTABLE implements jump tables.
   836  		// Aux is the symbol (an *obj.LSym) for the jump table.
   837  		// control[0] is the index into the jump table.
   838  		// control[1] is the address of the jump table (the address of the symbol stored in Aux).
   839  		{name: "JUMPTABLE", controls: 2, aux: "Sym"},
   840  	}
   841  
   842  	archs = append(archs, arch{
   843  		name:               "ARM64",
   844  		pkg:                "cmd/internal/obj/arm64",
   845  		genfile:            "../../arm64/ssa.go",
   846  		genSIMDfile:        "../../arm64/simdssa.go",
   847  		ops:                append(ops, simdARM64Ops(fp11, fp21, fp31, fpgp, fpgpfp, fp21)...),
   848  		blocks:             blocks,
   849  		regnames:           regNamesARM64,
   850  		ParamIntRegNames:   "R0 R1 R2 R3 R4 R5 R6 R7 R8 R9 R10 R11 R12 R13 R14 R15",
   851  		ParamFloatRegNames: "F0 F1 F2 F3 F4 F5 F6 F7 F8 F9 F10 F11 F12 F13 F14 F15",
   852  		gpregmask:          gp,
   853  		fpregmask:          fp,
   854  		simdregmask:        fp,
   855  		framepointerreg:    -1, // not used
   856  		linkreg:            int8(num["R30"]),
   857  	})
   858  }
   859  

View as plain text