Source file src/cmd/internal/obj/x86/asm6.go

     1  // Inferno utils/6l/span.c
     2  // https://bitbucket.org/inferno-os/inferno-os/src/master/utils/6l/span.c
     3  //
     4  //	Copyright © 1994-1999 Lucent Technologies Inc.  All rights reserved.
     5  //	Portions Copyright © 1995-1997 C H Forsyth (forsyth@terzarima.net)
     6  //	Portions Copyright © 1997-1999 Vita Nuova Limited
     7  //	Portions Copyright © 2000-2007 Vita Nuova Holdings Limited (www.vitanuova.com)
     8  //	Portions Copyright © 2004,2006 Bruce Ellis
     9  //	Portions Copyright © 2005-2007 C H Forsyth (forsyth@terzarima.net)
    10  //	Revisions Copyright © 2000-2007 Lucent Technologies Inc. and others
    11  //	Portions Copyright © 2009 The Go Authors. All rights reserved.
    12  //
    13  // Permission is hereby granted, free of charge, to any person obtaining a copy
    14  // of this software and associated documentation files (the "Software"), to deal
    15  // in the Software without restriction, including without limitation the rights
    16  // to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
    17  // copies of the Software, and to permit persons to whom the Software is
    18  // furnished to do so, subject to the following conditions:
    19  //
    20  // The above copyright notice and this permission notice shall be included in
    21  // all copies or substantial portions of the Software.
    22  //
    23  // THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
    24  // IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
    25  // FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.  IN NO EVENT SHALL THE
    26  // AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
    27  // LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
    28  // OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
    29  // THE SOFTWARE.
    30  
    31  package x86
    32  
    33  import (
    34  	"cmd/internal/obj"
    35  	"cmd/internal/objabi"
    36  	"cmd/internal/sys"
    37  	"encoding/binary"
    38  	"fmt"
    39  	"internal/buildcfg"
    40  	"log"
    41  	"strings"
    42  )
    43  
    44  var (
    45  	plan9privates *obj.LSym
    46  )
    47  
    48  // Instruction layout.
    49  
    50  // Loop alignment constants:
    51  // want to align loop entry to loopAlign-byte boundary,
    52  // and willing to insert at most maxLoopPad bytes of NOP to do so.
    53  // We define a loop entry as the target of a backward jump.
    54  //
    55  // gcc uses maxLoopPad = 10 for its 'generic x86-64' config,
    56  // and it aligns all jump targets, not just backward jump targets.
    57  //
    58  // As of 6/1/2012, the effect of setting maxLoopPad = 10 here
    59  // is very slight but negative, so the alignment is disabled by
    60  // setting MaxLoopPad = 0. The code is here for reference and
    61  // for future experiments.
    62  const (
    63  	loopAlign  = 16
    64  	maxLoopPad = 0
    65  )
    66  
    67  // Bit flags that are used to express jump target properties.
    68  const (
    69  	// branchBackwards marks targets that are located behind.
    70  	// Used to express jumps to loop headers.
    71  	branchBackwards = (1 << iota)
    72  	// branchShort marks branches those target is close,
    73  	// with offset is in -128..127 range.
    74  	branchShort
    75  	// branchLoopHead marks loop entry.
    76  	// Used to insert padding for misaligned loops.
    77  	branchLoopHead
    78  )
    79  
    80  // opBytes holds optab encoding bytes.
    81  // Each ytab reserves fixed amount of bytes in this array.
    82  //
    83  // The size should be the minimal number of bytes that
    84  // are enough to hold biggest optab op lines.
    85  type opBytes [31]uint8
    86  
    87  type Optab struct {
    88  	as     obj.As
    89  	ytab   []ytab
    90  	prefix uint8
    91  	op     opBytes
    92  }
    93  
    94  type movtab struct {
    95  	as   obj.As
    96  	ft   uint8
    97  	f3t  uint8
    98  	tt   uint8
    99  	code uint8
   100  	op   [4]uint8
   101  }
   102  
   103  const (
   104  	Yxxx = iota
   105  	Ynone
   106  	Yi0 // $0
   107  	Yi1 // $1
   108  	Yu2 // $x, x fits in uint2
   109  	Yi8 // $x, x fits in int8
   110  	Yu8 // $x, x fits in uint8
   111  	Yu7 // $x, x in 0..127 (fits in both int8 and uint8)
   112  	Ys32
   113  	Yi32
   114  	Yi64
   115  	Yiauto
   116  	Yal
   117  	Ycl
   118  	Yax
   119  	Ycx
   120  	Yrb
   121  	Yrl
   122  	Yrl32 // Yrl on 32-bit system
   123  	Yrf
   124  	Yf0
   125  	Yrx
   126  	Ymb
   127  	Yml
   128  	Ym
   129  	Ybr
   130  	Ycs
   131  	Yss
   132  	Yds
   133  	Yes
   134  	Yfs
   135  	Ygs
   136  	Ygdtr
   137  	Yidtr
   138  	Yldtr
   139  	Ymsw
   140  	Ytask
   141  	Ycr0
   142  	Ycr1
   143  	Ycr2
   144  	Ycr3
   145  	Ycr4
   146  	Ycr5
   147  	Ycr6
   148  	Ycr7
   149  	Ycr8
   150  	Ydr0
   151  	Ydr1
   152  	Ydr2
   153  	Ydr3
   154  	Ydr4
   155  	Ydr5
   156  	Ydr6
   157  	Ydr7
   158  	Ytr0
   159  	Ytr1
   160  	Ytr2
   161  	Ytr3
   162  	Ytr4
   163  	Ytr5
   164  	Ytr6
   165  	Ytr7
   166  	Ymr
   167  	Ymm
   168  	Yxr0          // X0 only. "<XMM0>" notation in Intel manual.
   169  	YxrEvexMulti4 // [ X<n> - X<n+3> ]; multisource YxrEvex
   170  	Yxr           // X0..X15
   171  	YxrEvex       // X0..X31
   172  	Yxm
   173  	YxmEvex       // YxrEvex+Ym
   174  	Yxvm          // VSIB vector array; vm32x/vm64x
   175  	YxvmEvex      // Yxvm which permits High-16 X register as index.
   176  	YyrEvexMulti4 // [ Y<n> - Y<n+3> ]; multisource YyrEvex
   177  	Yyr           // Y0..Y15
   178  	YyrEvex       // Y0..Y31
   179  	Yym
   180  	YymEvex   // YyrEvex+Ym
   181  	Yyvm      // VSIB vector array; vm32y/vm64y
   182  	YyvmEvex  // Yyvm which permits High-16 Y register as index.
   183  	YzrMulti4 // [ Z<n> - Z<n+3> ]; multisource YzrEvex
   184  	Yzr       // Z0..Z31
   185  	Yzm       // Yzr+Ym
   186  	Yzvm      // VSIB vector array; vm32z/vm64z
   187  	Yk0       // K0
   188  	Yknot0    // K1..K7; write mask
   189  	Yk        // K0..K7; used for KOP
   190  	Ykm       // Yk+Ym; used for KOP
   191  	Ytls
   192  	Ytextsize
   193  	Yindir
   194  	Ymax
   195  )
   196  
   197  const (
   198  	Zxxx = iota
   199  	Zlit
   200  	Zlitm_r
   201  	Zlitr_m
   202  	Zlit_m_r
   203  	Z_rp
   204  	Zbr
   205  	Zcall
   206  	Zcallcon
   207  	Zcallduff
   208  	Zcallind
   209  	Zcallindreg
   210  	Zib_
   211  	Zib_rp
   212  	Zibo_m
   213  	Zibo_m_xm
   214  	Zil_
   215  	Zil_rp
   216  	Ziq_rp
   217  	Zilo_m
   218  	Zjmp
   219  	Zjmpcon
   220  	Zloop
   221  	Zo_iw
   222  	Zm_o
   223  	Zm_r
   224  	Z_m_r
   225  	Zm2_r
   226  	Zm_r_xm
   227  	Zm_r_i_xm
   228  	Zm_r_xm_nr
   229  	Zr_m_xm_nr
   230  	Zibm_r // mmx1,mmx2/mem64,imm8
   231  	Zibr_m
   232  	Zmb_r
   233  	Zaut_r
   234  	Zo_m
   235  	Zo_m64
   236  	Zpseudo
   237  	Zr_m
   238  	Zr_m_xm
   239  	Zrp_
   240  	Z_ib
   241  	Z_il
   242  	Zm_ibo
   243  	Zm_ilo
   244  	Zib_rr
   245  	Zil_rr
   246  	Zbyte
   247  
   248  	Zvex_rm_v_r
   249  	Zvex_rm_v_ro
   250  	Zvex_r_v_rm
   251  	Zvex_i_rm_vo
   252  	Zvex_v_rm_r
   253  	Zvex_i_rm_r
   254  	Zvex_i_r_v
   255  	Zvex_i_rm_v_r
   256  	Zvex
   257  	Zvex_rm_r_vo
   258  	Zvex_i_r_rm
   259  	Zvex_hr_rm_v_r
   260  
   261  	Zevex_first
   262  	Zevex_i_r_k_rm
   263  	Zevex_i_r_rm
   264  	Zevex_i_rm_k_r
   265  	Zevex_i_rm_k_vo
   266  	Zevex_i_rm_r
   267  	Zevex_i_rm_v_k_r
   268  	Zevex_i_rm_v_r
   269  	Zevex_i_rm_vo
   270  	Zevex_k_rmo
   271  	Zevex_r_k_rm
   272  	Zevex_r_v_k_rm
   273  	Zevex_r_v_rm
   274  	Zevex_rm_k_r
   275  	Zevex_rm_v_k_r
   276  	Zevex_rm_v_r
   277  	Zevex_last
   278  
   279  	Zmax
   280  )
   281  
   282  const (
   283  	Px   = 0
   284  	Px1  = 1    // symbolic; exact value doesn't matter
   285  	P32  = 0x32 // 32-bit only
   286  	Pe   = 0x66 // operand escape
   287  	Pm   = 0x0f // 2byte opcode escape
   288  	Pq   = 0xff // both escapes: 66 0f
   289  	Pb   = 0xfe // byte operands
   290  	Pf2  = 0xf2 // xmm escape 1: f2 0f
   291  	Pf3  = 0xf3 // xmm escape 2: f3 0f
   292  	Pef3 = 0xf5 // xmm escape 2 with 16-bit prefix: 66 f3 0f
   293  	Pq3  = 0x67 // xmm escape 3: 66 48 0f
   294  	Pq4  = 0x68 // xmm escape 4: 66 0F 38
   295  	Pq4w = 0x69 // Pq4 with Rex.w 66 0F 38
   296  	Pq5  = 0x6a // xmm escape 5: F3 0F 38
   297  	Pq5w = 0x6b // Pq5 with Rex.w F3 0F 38
   298  	Pfw  = 0xf4 // Pf3 with Rex.w: f3 48 0f
   299  	Pw   = 0x48 // Rex.w
   300  	Pw8  = 0x90 // symbolic; exact value doesn't matter
   301  	Py   = 0x80 // defaults to 64-bit mode
   302  	Py1  = 0x81 // symbolic; exact value doesn't matter
   303  	Py3  = 0x83 // symbolic; exact value doesn't matter
   304  	Pavx = 0x84 // symbolic; exact value doesn't matter
   305  
   306  	RxrEvex = 1 << 4 // AVX512 extension to REX.R/VEX.R
   307  	Rxw     = 1 << 3 // =1, 64-bit operand size
   308  	Rxr     = 1 << 2 // extend modrm reg
   309  	Rxx     = 1 << 1 // extend sib index
   310  	Rxb     = 1 << 0 // extend modrm r/m, sib base, or opcode reg
   311  )
   312  
   313  const (
   314  	// Encoding for VEX prefix in tables.
   315  	// The P, L, and W fields are chosen to match
   316  	// their eventual locations in the VEX prefix bytes.
   317  
   318  	// Encoding for VEX prefix in tables.
   319  	// The P, L, and W fields are chosen to match
   320  	// their eventual locations in the VEX prefix bytes.
   321  
   322  	// Using spare bit to make leading [E]VEX encoding byte different from
   323  	// 0x0f even if all other VEX fields are 0.
   324  	avxEscape = 1 << 6
   325  
   326  	// P field - 2 bits
   327  	vex66 = 1 << 0
   328  	vexF3 = 2 << 0
   329  	vexF2 = 3 << 0
   330  	// L field - 1 bit
   331  	vexLZ  = 0 << 2
   332  	vexLIG = 0 << 2
   333  	vex128 = 0 << 2
   334  	vex256 = 1 << 2
   335  	// W field - 1 bit
   336  	vexWIG = 0 << 7
   337  	vexW0  = 0 << 7
   338  	vexW1  = 1 << 7
   339  	// M field - 5 bits, but mostly reserved; we can store up to 3
   340  	vex0F   = 1 << 3
   341  	vex0F38 = 2 << 3
   342  	vex0F3A = 3 << 3
   343  )
   344  
   345  var ycover [Ymax * Ymax]uint8
   346  
   347  var reg [MAXREG]int
   348  
   349  var regrex [MAXREG + 1]int
   350  
   351  var ynone = []ytab{
   352  	{Zlit, 1, argList{}},
   353  }
   354  
   355  var ytext = []ytab{
   356  	{Zpseudo, 0, argList{Ymb, Ytextsize}},
   357  	{Zpseudo, 1, argList{Ymb, Yi32, Ytextsize}},
   358  }
   359  
   360  var ynop = []ytab{
   361  	{Zpseudo, 0, argList{}},
   362  	{Zpseudo, 0, argList{Yiauto}},
   363  	{Zpseudo, 0, argList{Yml}},
   364  	{Zpseudo, 0, argList{Yrf}},
   365  	{Zpseudo, 0, argList{Yxr}},
   366  	{Zpseudo, 0, argList{Yiauto}},
   367  	{Zpseudo, 0, argList{Yml}},
   368  	{Zpseudo, 0, argList{Yrf}},
   369  	{Zpseudo, 1, argList{Yxr}},
   370  }
   371  
   372  var yfuncdata = []ytab{
   373  	{Zpseudo, 0, argList{Yi32, Ym}},
   374  }
   375  
   376  var ypcdata = []ytab{
   377  	{Zpseudo, 0, argList{Yi32, Yi32}},
   378  }
   379  
   380  var yxorb = []ytab{
   381  	{Zib_, 1, argList{Yi32, Yal}},
   382  	{Zibo_m, 2, argList{Yi32, Ymb}},
   383  	{Zr_m, 1, argList{Yrb, Ymb}},
   384  	{Zm_r, 1, argList{Ymb, Yrb}},
   385  }
   386  
   387  var yaddl = []ytab{
   388  	{Zibo_m, 2, argList{Yi8, Yml}},
   389  	{Zil_, 1, argList{Yi32, Yax}},
   390  	{Zilo_m, 2, argList{Yi32, Yml}},
   391  	{Zr_m, 1, argList{Yrl, Yml}},
   392  	{Zm_r, 1, argList{Yml, Yrl}},
   393  }
   394  
   395  var yincl = []ytab{
   396  	{Z_rp, 1, argList{Yrl}},
   397  	{Zo_m, 2, argList{Yml}},
   398  }
   399  
   400  var yincq = []ytab{
   401  	{Zo_m, 2, argList{Yml}},
   402  }
   403  
   404  var ycmpb = []ytab{
   405  	{Z_ib, 1, argList{Yal, Yi32}},
   406  	{Zm_ibo, 2, argList{Ymb, Yi32}},
   407  	{Zm_r, 1, argList{Ymb, Yrb}},
   408  	{Zr_m, 1, argList{Yrb, Ymb}},
   409  }
   410  
   411  var ycmpl = []ytab{
   412  	{Zm_ibo, 2, argList{Yml, Yi8}},
   413  	{Z_il, 1, argList{Yax, Yi32}},
   414  	{Zm_ilo, 2, argList{Yml, Yi32}},
   415  	{Zm_r, 1, argList{Yml, Yrl}},
   416  	{Zr_m, 1, argList{Yrl, Yml}},
   417  }
   418  
   419  var yshb = []ytab{
   420  	{Zo_m, 2, argList{Yi1, Ymb}},
   421  	{Zibo_m, 2, argList{Yu8, Ymb}},
   422  	{Zo_m, 2, argList{Ycx, Ymb}},
   423  }
   424  
   425  var yshl = []ytab{
   426  	{Zo_m, 2, argList{Yi1, Yml}},
   427  	{Zibo_m, 2, argList{Yu8, Yml}},
   428  	{Zo_m, 2, argList{Ycl, Yml}},
   429  	{Zo_m, 2, argList{Ycx, Yml}},
   430  }
   431  
   432  var ytestl = []ytab{
   433  	{Zil_, 1, argList{Yi32, Yax}},
   434  	{Zilo_m, 2, argList{Yi32, Yml}},
   435  	{Zr_m, 1, argList{Yrl, Yml}},
   436  	{Zm_r, 1, argList{Yml, Yrl}},
   437  }
   438  
   439  var ymovb = []ytab{
   440  	{Zr_m, 1, argList{Yrb, Ymb}},
   441  	{Zm_r, 1, argList{Ymb, Yrb}},
   442  	{Zib_rp, 1, argList{Yi32, Yrb}},
   443  	{Zibo_m, 2, argList{Yi32, Ymb}},
   444  }
   445  
   446  var ybtl = []ytab{
   447  	{Zibo_m, 2, argList{Yi8, Yml}},
   448  	{Zr_m, 1, argList{Yrl, Yml}},
   449  }
   450  
   451  var ymovw = []ytab{
   452  	{Zr_m, 1, argList{Yrl, Yml}},
   453  	{Zm_r, 1, argList{Yml, Yrl}},
   454  	{Zil_rp, 1, argList{Yi32, Yrl}},
   455  	{Zilo_m, 2, argList{Yi32, Yml}},
   456  	{Zaut_r, 2, argList{Yiauto, Yrl}},
   457  }
   458  
   459  var ymovl = []ytab{
   460  	{Zr_m, 1, argList{Yrl, Yml}},
   461  	{Zm_r, 1, argList{Yml, Yrl}},
   462  	{Zil_rp, 1, argList{Yi32, Yrl}},
   463  	{Zilo_m, 2, argList{Yi32, Yml}},
   464  	{Zm_r_xm, 1, argList{Yml, Ymr}}, // MMX MOVD
   465  	{Zr_m_xm, 1, argList{Ymr, Yml}}, // MMX MOVD
   466  	{Zm_r_xm, 2, argList{Yml, Yxr}}, // XMM MOVD (32 bit)
   467  	{Zr_m_xm, 2, argList{Yxr, Yml}}, // XMM MOVD (32 bit)
   468  	{Zaut_r, 2, argList{Yiauto, Yrl}},
   469  }
   470  
   471  var yret = []ytab{
   472  	{Zo_iw, 1, argList{}},
   473  	{Zo_iw, 1, argList{Yi32}},
   474  }
   475  
   476  var ymovq = []ytab{
   477  	// valid in 32-bit mode
   478  	{Zm_r_xm_nr, 1, argList{Ym, Ymr}},  // 0x6f MMX MOVQ (shorter encoding)
   479  	{Zr_m_xm_nr, 1, argList{Ymr, Ym}},  // 0x7f MMX MOVQ
   480  	{Zm_r_xm_nr, 2, argList{Yxr, Ymr}}, // Pf2, 0xd6 MOVDQ2Q
   481  	{Zm_r_xm_nr, 2, argList{Yxm, Yxr}}, // Pf3, 0x7e MOVQ xmm1/m64 -> xmm2
   482  	{Zr_m_xm_nr, 2, argList{Yxr, Yxm}}, // Pe, 0xd6 MOVQ xmm1 -> xmm2/m64
   483  
   484  	// valid only in 64-bit mode, usually with 64-bit prefix
   485  	{Zr_m, 1, argList{Yrl, Yml}},      // 0x89
   486  	{Zm_r, 1, argList{Yml, Yrl}},      // 0x8b
   487  	{Ziq_rp, 1, argList{Yi64, Yrl}},   // 0xb8 -- 32/64 bit immediate (Ziq_rp picks 5/7/10-byte form)
   488  	{Zilo_m, 2, argList{Yi32, Yml}},   // 0xc7,(0)
   489  	{Zm_r_xm, 1, argList{Ymm, Ymr}},   // 0x6e MMX MOVD
   490  	{Zr_m_xm, 1, argList{Ymr, Ymm}},   // 0x7e MMX MOVD
   491  	{Zm_r_xm, 2, argList{Yml, Yxr}},   // Pe, 0x6e MOVD xmm load
   492  	{Zr_m_xm, 2, argList{Yxr, Yml}},   // Pe, 0x7e MOVD xmm store
   493  	{Zaut_r, 1, argList{Yiauto, Yrl}}, // 0 built-in LEAQ
   494  }
   495  
   496  var ymovbe = []ytab{
   497  	{Zlitm_r, 3, argList{Ym, Yrl}},
   498  	{Zlitr_m, 3, argList{Yrl, Ym}},
   499  }
   500  
   501  var ym_rl = []ytab{
   502  	{Zm_r, 1, argList{Ym, Yrl}},
   503  }
   504  
   505  var yrl_m = []ytab{
   506  	{Zr_m, 1, argList{Yrl, Ym}},
   507  }
   508  
   509  var ymb_rl = []ytab{
   510  	{Zmb_r, 1, argList{Ymb, Yrl}},
   511  }
   512  
   513  var yml_rl = []ytab{
   514  	{Zm_r, 1, argList{Yml, Yrl}},
   515  }
   516  
   517  var yrl_ml = []ytab{
   518  	{Zr_m, 1, argList{Yrl, Yml}},
   519  }
   520  
   521  var yml_mb = []ytab{
   522  	{Zr_m, 1, argList{Yrb, Ymb}},
   523  	{Zm_r, 1, argList{Ymb, Yrb}},
   524  }
   525  
   526  var yrb_mb = []ytab{
   527  	{Zr_m, 1, argList{Yrb, Ymb}},
   528  }
   529  
   530  var yxchg = []ytab{
   531  	{Z_rp, 1, argList{Yax, Yrl}},
   532  	{Zrp_, 1, argList{Yrl, Yax}},
   533  	{Zr_m, 1, argList{Yrl, Yml}},
   534  	{Zm_r, 1, argList{Yml, Yrl}},
   535  }
   536  
   537  var ydivl = []ytab{
   538  	{Zm_o, 2, argList{Yml}},
   539  }
   540  
   541  var ydivb = []ytab{
   542  	{Zm_o, 2, argList{Ymb}},
   543  }
   544  
   545  var yimul = []ytab{
   546  	{Zm_o, 2, argList{Yml}},
   547  	{Zib_rr, 1, argList{Yi8, Yrl}},
   548  	{Zil_rr, 1, argList{Yi32, Yrl}},
   549  	{Zm_r, 2, argList{Yml, Yrl}},
   550  }
   551  
   552  var yimul3 = []ytab{
   553  	{Zibm_r, 2, argList{Yi8, Yml, Yrl}},
   554  	{Zibm_r, 2, argList{Yi32, Yml, Yrl}},
   555  }
   556  
   557  var ybyte = []ytab{
   558  	{Zbyte, 1, argList{Yi64}},
   559  }
   560  
   561  var yin = []ytab{
   562  	{Zib_, 1, argList{Yi32}},
   563  	{Zlit, 1, argList{}},
   564  }
   565  
   566  var yint = []ytab{
   567  	{Zib_, 1, argList{Yi32}},
   568  }
   569  
   570  var ypushl = []ytab{
   571  	{Zrp_, 1, argList{Yrl}},
   572  	{Zm_o, 2, argList{Ym}},
   573  	{Zib_, 1, argList{Yi8}},
   574  	{Zil_, 1, argList{Yi32}},
   575  }
   576  
   577  var ypopl = []ytab{
   578  	{Z_rp, 1, argList{Yrl}},
   579  	{Zo_m, 2, argList{Ym}},
   580  }
   581  
   582  var ywrfsbase = []ytab{
   583  	{Zm_o, 2, argList{Yrl}},
   584  }
   585  
   586  var yrdrand = []ytab{
   587  	{Zo_m, 2, argList{Yrl}},
   588  }
   589  
   590  var yclflush = []ytab{
   591  	{Zo_m, 2, argList{Ym}},
   592  }
   593  
   594  var ybswap = []ytab{
   595  	{Z_rp, 2, argList{Yrl}},
   596  }
   597  
   598  var yscond = []ytab{
   599  	{Zo_m, 2, argList{Ymb}},
   600  }
   601  
   602  var yjcond = []ytab{
   603  	{Zbr, 0, argList{Ybr}},
   604  	{Zbr, 0, argList{Yi0, Ybr}},
   605  	{Zbr, 1, argList{Yi1, Ybr}},
   606  }
   607  
   608  var yloop = []ytab{
   609  	{Zloop, 1, argList{Ybr}},
   610  }
   611  
   612  var ycall = []ytab{
   613  	{Zcallindreg, 0, argList{Yml}},
   614  	{Zcallindreg, 2, argList{Yrx, Yrx}},
   615  	{Zcallind, 2, argList{Yindir}},
   616  	{Zcall, 0, argList{Ybr}},
   617  	{Zcallcon, 1, argList{Yi32}},
   618  }
   619  
   620  var yduff = []ytab{
   621  	{Zcallduff, 1, argList{Yi32}},
   622  }
   623  
   624  var yjmp = []ytab{
   625  	{Zo_m64, 2, argList{Yml}},
   626  	{Zjmp, 0, argList{Ybr}},
   627  	{Zjmpcon, 1, argList{Yi32}},
   628  }
   629  
   630  var yfmvd = []ytab{
   631  	{Zm_o, 2, argList{Ym, Yf0}},
   632  	{Zo_m, 2, argList{Yf0, Ym}},
   633  	{Zm_o, 2, argList{Yrf, Yf0}},
   634  	{Zo_m, 2, argList{Yf0, Yrf}},
   635  }
   636  
   637  var yfmvdp = []ytab{
   638  	{Zo_m, 2, argList{Yf0, Ym}},
   639  	{Zo_m, 2, argList{Yf0, Yrf}},
   640  }
   641  
   642  var yfmvf = []ytab{
   643  	{Zm_o, 2, argList{Ym, Yf0}},
   644  	{Zo_m, 2, argList{Yf0, Ym}},
   645  }
   646  
   647  var yfmvx = []ytab{
   648  	{Zm_o, 2, argList{Ym, Yf0}},
   649  }
   650  
   651  var yfmvp = []ytab{
   652  	{Zo_m, 2, argList{Yf0, Ym}},
   653  }
   654  
   655  var yfcmv = []ytab{
   656  	{Zm_o, 2, argList{Yrf, Yf0}},
   657  }
   658  
   659  var yfadd = []ytab{
   660  	{Zm_o, 2, argList{Ym, Yf0}},
   661  	{Zm_o, 2, argList{Yrf, Yf0}},
   662  	{Zo_m, 2, argList{Yf0, Yrf}},
   663  }
   664  
   665  var yfxch = []ytab{
   666  	{Zo_m, 2, argList{Yf0, Yrf}},
   667  	{Zm_o, 2, argList{Yrf, Yf0}},
   668  }
   669  
   670  var ycompp = []ytab{
   671  	{Zo_m, 2, argList{Yf0, Yrf}}, // botch is really f0,f1
   672  }
   673  
   674  var ystsw = []ytab{
   675  	{Zo_m, 2, argList{Ym}},
   676  	{Zlit, 1, argList{Yax}},
   677  }
   678  
   679  var ysvrs_mo = []ytab{
   680  	{Zm_o, 2, argList{Ym}},
   681  }
   682  
   683  // unaryDst version of "ysvrs_mo".
   684  var ysvrs_om = []ytab{
   685  	{Zo_m, 2, argList{Ym}},
   686  }
   687  
   688  var ymm = []ytab{
   689  	{Zm_r_xm, 1, argList{Ymm, Ymr}},
   690  	{Zm_r_xm, 2, argList{Yxm, Yxr}},
   691  }
   692  
   693  var yxm = []ytab{
   694  	{Zm_r_xm, 1, argList{Yxm, Yxr}},
   695  }
   696  
   697  var yxm_q4 = []ytab{
   698  	{Zm_r, 1, argList{Yxm, Yxr}},
   699  }
   700  
   701  var yxcvm1 = []ytab{
   702  	{Zm_r_xm, 2, argList{Yxm, Yxr}},
   703  	{Zm_r_xm, 2, argList{Yxm, Ymr}},
   704  }
   705  
   706  var yxcvm2 = []ytab{
   707  	{Zm_r_xm, 2, argList{Yxm, Yxr}},
   708  	{Zm_r_xm, 2, argList{Ymm, Yxr}},
   709  }
   710  
   711  var yxr = []ytab{
   712  	{Zm_r_xm, 1, argList{Yxr, Yxr}},
   713  }
   714  
   715  var yxr_ml = []ytab{
   716  	{Zr_m_xm, 1, argList{Yxr, Yml}},
   717  }
   718  
   719  var ymr = []ytab{
   720  	{Zm_r, 1, argList{Ymr, Ymr}},
   721  }
   722  
   723  var ymr_ml = []ytab{
   724  	{Zr_m_xm, 1, argList{Ymr, Yml}},
   725  }
   726  
   727  var yxcmpi = []ytab{
   728  	{Zm_r_i_xm, 2, argList{Yxm, Yxr, Yi8}},
   729  }
   730  
   731  var yxmov = []ytab{
   732  	{Zm_r_xm, 1, argList{Yxm, Yxr}},
   733  	{Zr_m_xm, 1, argList{Yxr, Yxm}},
   734  }
   735  
   736  var yxcvfl = []ytab{
   737  	{Zm_r_xm, 1, argList{Yxm, Yrl}},
   738  }
   739  
   740  var yxcvlf = []ytab{
   741  	{Zm_r_xm, 1, argList{Yml, Yxr}},
   742  }
   743  
   744  var yxcvfq = []ytab{
   745  	{Zm_r_xm, 2, argList{Yxm, Yrl}},
   746  }
   747  
   748  var yxcvqf = []ytab{
   749  	{Zm_r_xm, 2, argList{Yml, Yxr}},
   750  }
   751  
   752  var yps = []ytab{
   753  	{Zm_r_xm, 1, argList{Ymm, Ymr}},
   754  	{Zibo_m_xm, 2, argList{Yi8, Ymr}},
   755  	{Zm_r_xm, 2, argList{Yxm, Yxr}},
   756  	{Zibo_m_xm, 3, argList{Yi8, Yxr}},
   757  }
   758  
   759  var yxrrl = []ytab{
   760  	{Zm_r, 1, argList{Yxr, Yrl}},
   761  }
   762  
   763  var ymrxr = []ytab{
   764  	{Zm_r, 1, argList{Ymr, Yxr}},
   765  	{Zm_r_xm, 1, argList{Yxm, Yxr}},
   766  }
   767  
   768  var ymshuf = []ytab{
   769  	{Zibm_r, 2, argList{Yi8, Ymm, Ymr}},
   770  }
   771  
   772  var ymshufb = []ytab{
   773  	{Zm2_r, 2, argList{Yxm, Yxr}},
   774  }
   775  
   776  // It should never have more than 1 entry,
   777  // because some optab entries have opcode sequences that
   778  // are longer than 2 bytes (zoffset=2 here),
   779  // ROUNDPD and ROUNDPS and recently added BLENDPD,
   780  // to name a few.
   781  var yxshuf = []ytab{
   782  	{Zibm_r, 2, argList{Yu8, Yxm, Yxr}},
   783  }
   784  
   785  var yextrw = []ytab{
   786  	{Zibm_r, 2, argList{Yu8, Yxr, Yrl}},
   787  	{Zibr_m, 2, argList{Yu8, Yxr, Yml}},
   788  }
   789  
   790  var yextr = []ytab{
   791  	{Zibr_m, 3, argList{Yu8, Yxr, Ymm}},
   792  }
   793  
   794  var yinsrw = []ytab{
   795  	{Zibm_r, 2, argList{Yu8, Yml, Yxr}},
   796  }
   797  
   798  var yinsr = []ytab{
   799  	{Zibm_r, 3, argList{Yu8, Ymm, Yxr}},
   800  }
   801  
   802  var ypsdq = []ytab{
   803  	{Zibo_m, 2, argList{Yi8, Yxr}},
   804  }
   805  
   806  var ymskb = []ytab{
   807  	{Zm_r_xm, 2, argList{Yxr, Yrl}},
   808  	{Zm_r_xm, 1, argList{Ymr, Yrl}},
   809  }
   810  
   811  var ycrc32l = []ytab{
   812  	{Zlitm_r, 0, argList{Yml, Yrl}},
   813  }
   814  
   815  var ycrc32b = []ytab{
   816  	{Zlitm_r, 0, argList{Ymb, Yrl}},
   817  }
   818  
   819  var yprefetch = []ytab{
   820  	{Zm_o, 2, argList{Ym}},
   821  }
   822  
   823  var yaes = []ytab{
   824  	{Zlitm_r, 2, argList{Yxm, Yxr}},
   825  }
   826  
   827  var yxbegin = []ytab{
   828  	{Zjmp, 1, argList{Ybr}},
   829  }
   830  
   831  var yxabort = []ytab{
   832  	{Zib_, 1, argList{Yu8}},
   833  }
   834  
   835  var ylddqu = []ytab{
   836  	{Zm_r, 1, argList{Ym, Yxr}},
   837  }
   838  
   839  var ypalignr = []ytab{
   840  	{Zibm_r, 2, argList{Yu8, Yxm, Yxr}},
   841  }
   842  
   843  var ysha256rnds2 = []ytab{
   844  	{Zlit_m_r, 0, argList{Yxr0, Yxm, Yxr}},
   845  }
   846  
   847  var yblendvpd = []ytab{
   848  	{Z_m_r, 1, argList{Yxr0, Yxm, Yxr}},
   849  }
   850  
   851  var ymmxmm0f38 = []ytab{
   852  	{Zlitm_r, 3, argList{Ymm, Ymr}},
   853  	{Zlitm_r, 5, argList{Yxm, Yxr}},
   854  }
   855  
   856  var yextractps = []ytab{
   857  	{Zibr_m, 2, argList{Yu2, Yxr, Yml}},
   858  }
   859  
   860  var ysha1rnds4 = []ytab{
   861  	{Zibm_r, 2, argList{Yu2, Yxm, Yxr}},
   862  }
   863  
   864  // You are doasm, holding in your hand a *obj.Prog with p.As set to, say,
   865  // ACRC32, and p.From and p.To as operands (obj.Addr).  The linker scans optab
   866  // to find the entry with the given p.As and then looks through the ytable for
   867  // that instruction (the second field in the optab struct) for a line whose
   868  // first two values match the Ytypes of the p.From and p.To operands.  The
   869  // function oclass computes the specific Ytype of an operand and then the set
   870  // of more general Ytypes that it satisfies is implied by the ycover table, set
   871  // up in instinit.  For example, oclass distinguishes the constants 0 and 1
   872  // from the more general 8-bit constants, but instinit says
   873  //
   874  //	ycover[Yi0*Ymax+Ys32] = 1
   875  //	ycover[Yi1*Ymax+Ys32] = 1
   876  //	ycover[Yi8*Ymax+Ys32] = 1
   877  //
   878  // which means that Yi0, Yi1, and Yi8 all count as Ys32 (signed 32)
   879  // if that's what an instruction can handle.
   880  //
   881  // In parallel with the scan through the ytable for the appropriate line, there
   882  // is a z pointer that starts out pointing at the strange magic byte list in
   883  // the Optab struct.  With each step past a non-matching ytable line, z
   884  // advances by the 4th entry in the line.  When a matching line is found, that
   885  // z pointer has the extra data to use in laying down the instruction bytes.
   886  // The actual bytes laid down are a function of the 3rd entry in the line (that
   887  // is, the Ztype) and the z bytes.
   888  //
   889  // For example, let's look at AADDL.  The optab line says:
   890  //
   891  //	{AADDL, yaddl, Px, opBytes{0x83, 00, 0x05, 0x81, 00, 0x01, 0x03}},
   892  //
   893  // and yaddl says
   894  //
   895  //	var yaddl = []ytab{
   896  //	        {Yi8, Ynone, Yml, Zibo_m, 2},
   897  //	        {Yi32, Ynone, Yax, Zil_, 1},
   898  //	        {Yi32, Ynone, Yml, Zilo_m, 2},
   899  //	        {Yrl, Ynone, Yml, Zr_m, 1},
   900  //	        {Yml, Ynone, Yrl, Zm_r, 1},
   901  //	}
   902  //
   903  // so there are 5 possible types of ADDL instruction that can be laid down, and
   904  // possible states used to lay them down (Ztype and z pointer, assuming z
   905  // points at opBytes{0x83, 00, 0x05,0x81, 00, 0x01, 0x03}) are:
   906  //
   907  //	Yi8, Yml -> Zibo_m, z (0x83, 00)
   908  //	Yi32, Yax -> Zil_, z+2 (0x05)
   909  //	Yi32, Yml -> Zilo_m, z+2+1 (0x81, 0x00)
   910  //	Yrl, Yml -> Zr_m, z+2+1+2 (0x01)
   911  //	Yml, Yrl -> Zm_r, z+2+1+2+1 (0x03)
   912  //
   913  // The Pconstant in the optab line controls the prefix bytes to emit.  That's
   914  // relatively straightforward as this program goes.
   915  //
   916  // The switch on yt.zcase in doasm implements the various Z cases.  Zibo_m, for
   917  // example, is an opcode byte (z[0]) then an asmando (which is some kind of
   918  // encoded addressing mode for the Yml arg), and then a single immediate byte.
   919  // Zilo_m is the same but a long (32-bit) immediate.
   920  var optab =
   921  // as, ytab, andproto, opcode
   922  [...]Optab{
   923  	{obj.AXXX, nil, 0, opBytes{}},
   924  	{AAAA, ynone, P32, opBytes{0x37}},
   925  	{AAAD, ynone, P32, opBytes{0xd5, 0x0a}},
   926  	{AAAM, ynone, P32, opBytes{0xd4, 0x0a}},
   927  	{AAAS, ynone, P32, opBytes{0x3f}},
   928  	{AADCB, yxorb, Pb, opBytes{0x14, 0x80, 02, 0x10, 0x12}},
   929  	{AADCL, yaddl, Px, opBytes{0x83, 02, 0x15, 0x81, 02, 0x11, 0x13}},
   930  	{AADCQ, yaddl, Pw, opBytes{0x83, 02, 0x15, 0x81, 02, 0x11, 0x13}},
   931  	{AADCW, yaddl, Pe, opBytes{0x83, 02, 0x15, 0x81, 02, 0x11, 0x13}},
   932  	{AADCXL, yml_rl, Pq4, opBytes{0xf6}},
   933  	{AADCXQ, yml_rl, Pq4w, opBytes{0xf6}},
   934  	{AADDB, yxorb, Pb, opBytes{0x04, 0x80, 00, 0x00, 0x02}},
   935  	{AADDL, yaddl, Px, opBytes{0x83, 00, 0x05, 0x81, 00, 0x01, 0x03}},
   936  	{AADDPD, yxm, Pq, opBytes{0x58}},
   937  	{AADDPS, yxm, Pm, opBytes{0x58}},
   938  	{AADDQ, yaddl, Pw, opBytes{0x83, 00, 0x05, 0x81, 00, 0x01, 0x03}},
   939  	{AADDSD, yxm, Pf2, opBytes{0x58}},
   940  	{AADDSS, yxm, Pf3, opBytes{0x58}},
   941  	{AADDSUBPD, yxm, Pq, opBytes{0xd0}},
   942  	{AADDSUBPS, yxm, Pf2, opBytes{0xd0}},
   943  	{AADDW, yaddl, Pe, opBytes{0x83, 00, 0x05, 0x81, 00, 0x01, 0x03}},
   944  	{AADOXL, yml_rl, Pq5, opBytes{0xf6}},
   945  	{AADOXQ, yml_rl, Pq5w, opBytes{0xf6}},
   946  	{AADJSP, nil, 0, opBytes{}},
   947  	{AANDB, yxorb, Pb, opBytes{0x24, 0x80, 04, 0x20, 0x22}},
   948  	{AANDL, yaddl, Px, opBytes{0x83, 04, 0x25, 0x81, 04, 0x21, 0x23}},
   949  	{AANDNPD, yxm, Pq, opBytes{0x55}},
   950  	{AANDNPS, yxm, Pm, opBytes{0x55}},
   951  	{AANDPD, yxm, Pq, opBytes{0x54}},
   952  	{AANDPS, yxm, Pm, opBytes{0x54}},
   953  	{AANDQ, yaddl, Pw, opBytes{0x83, 04, 0x25, 0x81, 04, 0x21, 0x23}},
   954  	{AANDW, yaddl, Pe, opBytes{0x83, 04, 0x25, 0x81, 04, 0x21, 0x23}},
   955  	{AARPL, yrl_ml, P32, opBytes{0x63}},
   956  	{ABOUNDL, yrl_m, P32, opBytes{0x62}},
   957  	{ABOUNDW, yrl_m, Pe, opBytes{0x62}},
   958  	{ABSFL, yml_rl, Pm, opBytes{0xbc}},
   959  	{ABSFQ, yml_rl, Pw, opBytes{0x0f, 0xbc}},
   960  	{ABSFW, yml_rl, Pq, opBytes{0xbc}},
   961  	{ABSRL, yml_rl, Pm, opBytes{0xbd}},
   962  	{ABSRQ, yml_rl, Pw, opBytes{0x0f, 0xbd}},
   963  	{ABSRW, yml_rl, Pq, opBytes{0xbd}},
   964  	{ABSWAPL, ybswap, Px, opBytes{0x0f, 0xc8}},
   965  	{ABSWAPQ, ybswap, Pw, opBytes{0x0f, 0xc8}},
   966  	{ABTCL, ybtl, Pm, opBytes{0xba, 07, 0xbb}},
   967  	{ABTCQ, ybtl, Pw, opBytes{0x0f, 0xba, 07, 0x0f, 0xbb}},
   968  	{ABTCW, ybtl, Pq, opBytes{0xba, 07, 0xbb}},
   969  	{ABTL, ybtl, Pm, opBytes{0xba, 04, 0xa3}},
   970  	{ABTQ, ybtl, Pw, opBytes{0x0f, 0xba, 04, 0x0f, 0xa3}},
   971  	{ABTRL, ybtl, Pm, opBytes{0xba, 06, 0xb3}},
   972  	{ABTRQ, ybtl, Pw, opBytes{0x0f, 0xba, 06, 0x0f, 0xb3}},
   973  	{ABTRW, ybtl, Pq, opBytes{0xba, 06, 0xb3}},
   974  	{ABTSL, ybtl, Pm, opBytes{0xba, 05, 0xab}},
   975  	{ABTSQ, ybtl, Pw, opBytes{0x0f, 0xba, 05, 0x0f, 0xab}},
   976  	{ABTSW, ybtl, Pq, opBytes{0xba, 05, 0xab}},
   977  	{ABTW, ybtl, Pq, opBytes{0xba, 04, 0xa3}},
   978  	{ABYTE, ybyte, Px, opBytes{1}},
   979  	{obj.ACALL, ycall, Px, opBytes{0xff, 02, 0xff, 0x15, 0xe8}},
   980  	{ACBW, ynone, Pe, opBytes{0x98}},
   981  	{ACDQ, ynone, Px, opBytes{0x99}},
   982  	{ACDQE, ynone, Pw, opBytes{0x98}},
   983  	{ACLAC, ynone, Pm, opBytes{01, 0xca}},
   984  	{ACLC, ynone, Px, opBytes{0xf8}},
   985  	{ACLD, ynone, Px, opBytes{0xfc}},
   986  	{ACLDEMOTE, yclflush, Pm, opBytes{0x1c, 00}},
   987  	{ACLFLUSH, yclflush, Pm, opBytes{0xae, 07}},
   988  	{ACLFLUSHOPT, yclflush, Pq, opBytes{0xae, 07}},
   989  	{ACLI, ynone, Px, opBytes{0xfa}},
   990  	{ACLTS, ynone, Pm, opBytes{0x06}},
   991  	{ACLWB, yclflush, Pq, opBytes{0xae, 06}},
   992  	{ACMC, ynone, Px, opBytes{0xf5}},
   993  	{ACMOVLCC, yml_rl, Pm, opBytes{0x43}},
   994  	{ACMOVLCS, yml_rl, Pm, opBytes{0x42}},
   995  	{ACMOVLEQ, yml_rl, Pm, opBytes{0x44}},
   996  	{ACMOVLGE, yml_rl, Pm, opBytes{0x4d}},
   997  	{ACMOVLGT, yml_rl, Pm, opBytes{0x4f}},
   998  	{ACMOVLHI, yml_rl, Pm, opBytes{0x47}},
   999  	{ACMOVLLE, yml_rl, Pm, opBytes{0x4e}},
  1000  	{ACMOVLLS, yml_rl, Pm, opBytes{0x46}},
  1001  	{ACMOVLLT, yml_rl, Pm, opBytes{0x4c}},
  1002  	{ACMOVLMI, yml_rl, Pm, opBytes{0x48}},
  1003  	{ACMOVLNE, yml_rl, Pm, opBytes{0x45}},
  1004  	{ACMOVLOC, yml_rl, Pm, opBytes{0x41}},
  1005  	{ACMOVLOS, yml_rl, Pm, opBytes{0x40}},
  1006  	{ACMOVLPC, yml_rl, Pm, opBytes{0x4b}},
  1007  	{ACMOVLPL, yml_rl, Pm, opBytes{0x49}},
  1008  	{ACMOVLPS, yml_rl, Pm, opBytes{0x4a}},
  1009  	{ACMOVQCC, yml_rl, Pw, opBytes{0x0f, 0x43}},
  1010  	{ACMOVQCS, yml_rl, Pw, opBytes{0x0f, 0x42}},
  1011  	{ACMOVQEQ, yml_rl, Pw, opBytes{0x0f, 0x44}},
  1012  	{ACMOVQGE, yml_rl, Pw, opBytes{0x0f, 0x4d}},
  1013  	{ACMOVQGT, yml_rl, Pw, opBytes{0x0f, 0x4f}},
  1014  	{ACMOVQHI, yml_rl, Pw, opBytes{0x0f, 0x47}},
  1015  	{ACMOVQLE, yml_rl, Pw, opBytes{0x0f, 0x4e}},
  1016  	{ACMOVQLS, yml_rl, Pw, opBytes{0x0f, 0x46}},
  1017  	{ACMOVQLT, yml_rl, Pw, opBytes{0x0f, 0x4c}},
  1018  	{ACMOVQMI, yml_rl, Pw, opBytes{0x0f, 0x48}},
  1019  	{ACMOVQNE, yml_rl, Pw, opBytes{0x0f, 0x45}},
  1020  	{ACMOVQOC, yml_rl, Pw, opBytes{0x0f, 0x41}},
  1021  	{ACMOVQOS, yml_rl, Pw, opBytes{0x0f, 0x40}},
  1022  	{ACMOVQPC, yml_rl, Pw, opBytes{0x0f, 0x4b}},
  1023  	{ACMOVQPL, yml_rl, Pw, opBytes{0x0f, 0x49}},
  1024  	{ACMOVQPS, yml_rl, Pw, opBytes{0x0f, 0x4a}},
  1025  	{ACMOVWCC, yml_rl, Pq, opBytes{0x43}},
  1026  	{ACMOVWCS, yml_rl, Pq, opBytes{0x42}},
  1027  	{ACMOVWEQ, yml_rl, Pq, opBytes{0x44}},
  1028  	{ACMOVWGE, yml_rl, Pq, opBytes{0x4d}},
  1029  	{ACMOVWGT, yml_rl, Pq, opBytes{0x4f}},
  1030  	{ACMOVWHI, yml_rl, Pq, opBytes{0x47}},
  1031  	{ACMOVWLE, yml_rl, Pq, opBytes{0x4e}},
  1032  	{ACMOVWLS, yml_rl, Pq, opBytes{0x46}},
  1033  	{ACMOVWLT, yml_rl, Pq, opBytes{0x4c}},
  1034  	{ACMOVWMI, yml_rl, Pq, opBytes{0x48}},
  1035  	{ACMOVWNE, yml_rl, Pq, opBytes{0x45}},
  1036  	{ACMOVWOC, yml_rl, Pq, opBytes{0x41}},
  1037  	{ACMOVWOS, yml_rl, Pq, opBytes{0x40}},
  1038  	{ACMOVWPC, yml_rl, Pq, opBytes{0x4b}},
  1039  	{ACMOVWPL, yml_rl, Pq, opBytes{0x49}},
  1040  	{ACMOVWPS, yml_rl, Pq, opBytes{0x4a}},
  1041  	{ACMPB, ycmpb, Pb, opBytes{0x3c, 0x80, 07, 0x38, 0x3a}},
  1042  	{ACMPL, ycmpl, Px, opBytes{0x83, 07, 0x3d, 0x81, 07, 0x39, 0x3b}},
  1043  	{ACMPPD, yxcmpi, Px, opBytes{Pe, 0xc2}},
  1044  	{ACMPPS, yxcmpi, Pm, opBytes{0xc2, 0}},
  1045  	{ACMPQ, ycmpl, Pw, opBytes{0x83, 07, 0x3d, 0x81, 07, 0x39, 0x3b}},
  1046  	{ACMPSB, ynone, Pb, opBytes{0xa6}},
  1047  	{ACMPSD, yxcmpi, Px, opBytes{Pf2, 0xc2}},
  1048  	{ACMPSL, ynone, Px, opBytes{0xa7}},
  1049  	{ACMPSQ, ynone, Pw, opBytes{0xa7}},
  1050  	{ACMPSS, yxcmpi, Px, opBytes{Pf3, 0xc2}},
  1051  	{ACMPSW, ynone, Pe, opBytes{0xa7}},
  1052  	{ACMPW, ycmpl, Pe, opBytes{0x83, 07, 0x3d, 0x81, 07, 0x39, 0x3b}},
  1053  	{ACOMISD, yxm, Pe, opBytes{0x2f}},
  1054  	{ACOMISS, yxm, Pm, opBytes{0x2f}},
  1055  	{ACPUID, ynone, Pm, opBytes{0xa2}},
  1056  	{ACVTPL2PD, yxcvm2, Px, opBytes{Pf3, 0xe6, Pe, 0x2a}},
  1057  	{ACVTPL2PS, yxcvm2, Pm, opBytes{0x5b, 0, 0x2a, 0}},
  1058  	{ACVTPD2PL, yxcvm1, Px, opBytes{Pf2, 0xe6, Pe, 0x2d}},
  1059  	{ACVTPD2PS, yxm, Pe, opBytes{0x5a}},
  1060  	{ACVTPS2PL, yxcvm1, Px, opBytes{Pe, 0x5b, Pm, 0x2d}},
  1061  	{ACVTPS2PD, yxm, Pm, opBytes{0x5a}},
  1062  	{ACVTSD2SL, yxcvfl, Pf2, opBytes{0x2d}},
  1063  	{ACVTSD2SQ, yxcvfq, Pw, opBytes{Pf2, 0x2d}},
  1064  	{ACVTSD2SS, yxm, Pf2, opBytes{0x5a}},
  1065  	{ACVTSL2SD, yxcvlf, Pf2, opBytes{0x2a}},
  1066  	{ACVTSQ2SD, yxcvqf, Pw, opBytes{Pf2, 0x2a}},
  1067  	{ACVTSL2SS, yxcvlf, Pf3, opBytes{0x2a}},
  1068  	{ACVTSQ2SS, yxcvqf, Pw, opBytes{Pf3, 0x2a}},
  1069  	{ACVTSS2SD, yxm, Pf3, opBytes{0x5a}},
  1070  	{ACVTSS2SL, yxcvfl, Pf3, opBytes{0x2d}},
  1071  	{ACVTSS2SQ, yxcvfq, Pw, opBytes{Pf3, 0x2d}},
  1072  	{ACVTTPD2PL, yxcvm1, Px, opBytes{Pe, 0xe6, Pe, 0x2c}},
  1073  	{ACVTTPS2PL, yxcvm1, Px, opBytes{Pf3, 0x5b, Pm, 0x2c}},
  1074  	{ACVTTSD2SL, yxcvfl, Pf2, opBytes{0x2c}},
  1075  	{ACVTTSD2SQ, yxcvfq, Pw, opBytes{Pf2, 0x2c}},
  1076  	{ACVTTSS2SL, yxcvfl, Pf3, opBytes{0x2c}},
  1077  	{ACVTTSS2SQ, yxcvfq, Pw, opBytes{Pf3, 0x2c}},
  1078  	{ACWD, ynone, Pe, opBytes{0x99}},
  1079  	{ACWDE, ynone, Px, opBytes{0x98}},
  1080  	{ACQO, ynone, Pw, opBytes{0x99}},
  1081  	{ADAA, ynone, P32, opBytes{0x27}},
  1082  	{ADAS, ynone, P32, opBytes{0x2f}},
  1083  	{ADECB, yscond, Pb, opBytes{0xfe, 01}},
  1084  	{ADECL, yincl, Px1, opBytes{0x48, 0xff, 01}},
  1085  	{ADECQ, yincq, Pw, opBytes{0xff, 01}},
  1086  	{ADECW, yincq, Pe, opBytes{0xff, 01}},
  1087  	{ADIVB, ydivb, Pb, opBytes{0xf6, 06}},
  1088  	{ADIVL, ydivl, Px, opBytes{0xf7, 06}},
  1089  	{ADIVPD, yxm, Pe, opBytes{0x5e}},
  1090  	{ADIVPS, yxm, Pm, opBytes{0x5e}},
  1091  	{ADIVQ, ydivl, Pw, opBytes{0xf7, 06}},
  1092  	{ADIVSD, yxm, Pf2, opBytes{0x5e}},
  1093  	{ADIVSS, yxm, Pf3, opBytes{0x5e}},
  1094  	{ADIVW, ydivl, Pe, opBytes{0xf7, 06}},
  1095  	{ADPPD, yxshuf, Pq, opBytes{0x3a, 0x41, 0}},
  1096  	{ADPPS, yxshuf, Pq, opBytes{0x3a, 0x40, 0}},
  1097  	{AEMMS, ynone, Pm, opBytes{0x77}},
  1098  	{AENDBR64, ynone, Pf3, opBytes{0x1e, 0xfa}},
  1099  	{AEXTRACTPS, yextractps, Pq, opBytes{0x3a, 0x17, 0}},
  1100  	{AENTER, nil, 0, opBytes{}}, // botch
  1101  	{AFXRSTOR, ysvrs_mo, Pm, opBytes{0xae, 01, 0xae, 01}},
  1102  	{AFXSAVE, ysvrs_om, Pm, opBytes{0xae, 00, 0xae, 00}},
  1103  	{AFXRSTOR64, ysvrs_mo, Pw, opBytes{0x0f, 0xae, 01, 0x0f, 0xae, 01}},
  1104  	{AFXSAVE64, ysvrs_om, Pw, opBytes{0x0f, 0xae, 00, 0x0f, 0xae, 00}},
  1105  	{AHLT, ynone, Px, opBytes{0xf4}},
  1106  	{AIDIVB, ydivb, Pb, opBytes{0xf6, 07}},
  1107  	{AIDIVL, ydivl, Px, opBytes{0xf7, 07}},
  1108  	{AIDIVQ, ydivl, Pw, opBytes{0xf7, 07}},
  1109  	{AIDIVW, ydivl, Pe, opBytes{0xf7, 07}},
  1110  	{AIMULB, ydivb, Pb, opBytes{0xf6, 05}},
  1111  	{AIMULL, yimul, Px, opBytes{0xf7, 05, 0x6b, 0x69, Pm, 0xaf}},
  1112  	{AIMULQ, yimul, Pw, opBytes{0xf7, 05, 0x6b, 0x69, Pm, 0xaf}},
  1113  	{AIMULW, yimul, Pe, opBytes{0xf7, 05, 0x6b, 0x69, Pm, 0xaf}},
  1114  	{AIMUL3W, yimul3, Pe, opBytes{0x6b, 00, 0x69, 00}},
  1115  	{AIMUL3L, yimul3, Px, opBytes{0x6b, 00, 0x69, 00}},
  1116  	{AIMUL3Q, yimul3, Pw, opBytes{0x6b, 00, 0x69, 00}},
  1117  	{AINB, yin, Pb, opBytes{0xe4, 0xec}},
  1118  	{AINW, yin, Pe, opBytes{0xe5, 0xed}},
  1119  	{AINL, yin, Px, opBytes{0xe5, 0xed}},
  1120  	{AINCB, yscond, Pb, opBytes{0xfe, 00}},
  1121  	{AINCL, yincl, Px1, opBytes{0x40, 0xff, 00}},
  1122  	{AINCQ, yincq, Pw, opBytes{0xff, 00}},
  1123  	{AINCW, yincq, Pe, opBytes{0xff, 00}},
  1124  	{AINSB, ynone, Pb, opBytes{0x6c}},
  1125  	{AINSL, ynone, Px, opBytes{0x6d}},
  1126  	{AINSERTPS, yxshuf, Pq, opBytes{0x3a, 0x21, 0}},
  1127  	{AINSW, ynone, Pe, opBytes{0x6d}},
  1128  	{AICEBP, ynone, Px, opBytes{0xf1}},
  1129  	{AINT, yint, Px, opBytes{0xcd}},
  1130  	{AINTO, ynone, P32, opBytes{0xce}},
  1131  	{AIRETL, ynone, Px, opBytes{0xcf}},
  1132  	{AIRETQ, ynone, Pw, opBytes{0xcf}},
  1133  	{AIRETW, ynone, Pe, opBytes{0xcf}},
  1134  	{AJCC, yjcond, Px, opBytes{0x73, 0x83, 00}},
  1135  	{AJCS, yjcond, Px, opBytes{0x72, 0x82}},
  1136  	{AJCXZL, yloop, Px, opBytes{0xe3}},
  1137  	{AJCXZW, yloop, Px, opBytes{0xe3}},
  1138  	{AJCXZQ, yloop, Px, opBytes{0xe3}},
  1139  	{AJEQ, yjcond, Px, opBytes{0x74, 0x84}},
  1140  	{AJGE, yjcond, Px, opBytes{0x7d, 0x8d}},
  1141  	{AJGT, yjcond, Px, opBytes{0x7f, 0x8f}},
  1142  	{AJHI, yjcond, Px, opBytes{0x77, 0x87}},
  1143  	{AJLE, yjcond, Px, opBytes{0x7e, 0x8e}},
  1144  	{AJLS, yjcond, Px, opBytes{0x76, 0x86}},
  1145  	{AJLT, yjcond, Px, opBytes{0x7c, 0x8c}},
  1146  	{AJMI, yjcond, Px, opBytes{0x78, 0x88}},
  1147  	{obj.AJMP, yjmp, Px, opBytes{0xff, 04, 0xeb, 0xe9}},
  1148  	{AJNE, yjcond, Px, opBytes{0x75, 0x85}},
  1149  	{AJOC, yjcond, Px, opBytes{0x71, 0x81, 00}},
  1150  	{AJOS, yjcond, Px, opBytes{0x70, 0x80, 00}},
  1151  	{AJPC, yjcond, Px, opBytes{0x7b, 0x8b}},
  1152  	{AJPL, yjcond, Px, opBytes{0x79, 0x89}},
  1153  	{AJPS, yjcond, Px, opBytes{0x7a, 0x8a}},
  1154  	{AHADDPD, yxm, Pq, opBytes{0x7c}},
  1155  	{AHADDPS, yxm, Pf2, opBytes{0x7c}},
  1156  	{AHSUBPD, yxm, Pq, opBytes{0x7d}},
  1157  	{AHSUBPS, yxm, Pf2, opBytes{0x7d}},
  1158  	{ALAHF, ynone, Px, opBytes{0x9f}},
  1159  	{ALARL, yml_rl, Pm, opBytes{0x02}},
  1160  	{ALARQ, yml_rl, Pw, opBytes{0x0f, 0x02}},
  1161  	{ALARW, yml_rl, Pq, opBytes{0x02}},
  1162  	{ALDDQU, ylddqu, Pf2, opBytes{0xf0}},
  1163  	{ALDMXCSR, ysvrs_mo, Pm, opBytes{0xae, 02, 0xae, 02}},
  1164  	{ALEAL, ym_rl, Px, opBytes{0x8d}},
  1165  	{ALEAQ, ym_rl, Pw, opBytes{0x8d}},
  1166  	{ALEAVEL, ynone, P32, opBytes{0xc9}},
  1167  	{ALEAVEQ, ynone, Py, opBytes{0xc9}},
  1168  	{ALEAVEW, ynone, Pe, opBytes{0xc9}},
  1169  	{ALEAW, ym_rl, Pe, opBytes{0x8d}},
  1170  	{ALOCK, ynone, Px, opBytes{0xf0}},
  1171  	{ALODSB, ynone, Pb, opBytes{0xac}},
  1172  	{ALODSL, ynone, Px, opBytes{0xad}},
  1173  	{ALODSQ, ynone, Pw, opBytes{0xad}},
  1174  	{ALODSW, ynone, Pe, opBytes{0xad}},
  1175  	{ALONG, ybyte, Px, opBytes{4}},
  1176  	{ALOOP, yloop, Px, opBytes{0xe2}},
  1177  	{ALOOPEQ, yloop, Px, opBytes{0xe1}},
  1178  	{ALOOPNE, yloop, Px, opBytes{0xe0}},
  1179  	{ALTR, ydivl, Pm, opBytes{0x00, 03}},
  1180  	{ALZCNTL, yml_rl, Pf3, opBytes{0xbd}},
  1181  	{ALZCNTQ, yml_rl, Pfw, opBytes{0xbd}},
  1182  	{ALZCNTW, yml_rl, Pef3, opBytes{0xbd}},
  1183  	{ALSLL, yml_rl, Pm, opBytes{0x03}},
  1184  	{ALSLW, yml_rl, Pq, opBytes{0x03}},
  1185  	{ALSLQ, yml_rl, Pw, opBytes{0x0f, 0x03}},
  1186  	{AMASKMOVOU, yxr, Pe, opBytes{0xf7}},
  1187  	{AMASKMOVQ, ymr, Pm, opBytes{0xf7}},
  1188  	{AMAXPD, yxm, Pe, opBytes{0x5f}},
  1189  	{AMAXPS, yxm, Pm, opBytes{0x5f}},
  1190  	{AMAXSD, yxm, Pf2, opBytes{0x5f}},
  1191  	{AMAXSS, yxm, Pf3, opBytes{0x5f}},
  1192  	{AMINPD, yxm, Pe, opBytes{0x5d}},
  1193  	{AMINPS, yxm, Pm, opBytes{0x5d}},
  1194  	{AMINSD, yxm, Pf2, opBytes{0x5d}},
  1195  	{AMINSS, yxm, Pf3, opBytes{0x5d}},
  1196  	{AMONITOR, ynone, Px, opBytes{0x0f, 0x01, 0xc8, 0}},
  1197  	{AMWAIT, ynone, Px, opBytes{0x0f, 0x01, 0xc9, 0}},
  1198  	{AMOVAPD, yxmov, Pe, opBytes{0x28, 0x29}},
  1199  	{AMOVAPS, yxmov, Pm, opBytes{0x28, 0x29}},
  1200  	{AMOVB, ymovb, Pb, opBytes{0x88, 0x8a, 0xb0, 0xc6, 00}},
  1201  	{AMOVBLSX, ymb_rl, Pm, opBytes{0xbe}},
  1202  	{AMOVBLZX, ymb_rl, Pm, opBytes{0xb6}},
  1203  	{AMOVBQSX, ymb_rl, Pw, opBytes{0x0f, 0xbe}},
  1204  	{AMOVBQZX, ymb_rl, Pw, opBytes{0x0f, 0xb6}},
  1205  	{AMOVBWSX, ymb_rl, Pq, opBytes{0xbe}},
  1206  	{AMOVSWW, ymb_rl, Pe, opBytes{0x0f, 0xbf}},
  1207  	{AMOVBWZX, ymb_rl, Pq, opBytes{0xb6}},
  1208  	{AMOVZWW, ymb_rl, Pe, opBytes{0x0f, 0xb7}},
  1209  	{AMOVO, yxmov, Pe, opBytes{0x6f, 0x7f}},
  1210  	{AMOVOU, yxmov, Pf3, opBytes{0x6f, 0x7f}},
  1211  	{AMOVHLPS, yxr, Pm, opBytes{0x12}},
  1212  	{AMOVHPD, yxmov, Pe, opBytes{0x16, 0x17}},
  1213  	{AMOVHPS, yxmov, Pm, opBytes{0x16, 0x17}},
  1214  	{AMOVL, ymovl, Px, opBytes{0x89, 0x8b, 0xb8, 0xc7, 00, 0x6e, 0x7e, Pe, 0x6e, Pe, 0x7e, 0}},
  1215  	{AMOVLHPS, yxr, Pm, opBytes{0x16}},
  1216  	{AMOVLPD, yxmov, Pe, opBytes{0x12, 0x13}},
  1217  	{AMOVLPS, yxmov, Pm, opBytes{0x12, 0x13}},
  1218  	{AMOVLQSX, yml_rl, Pw, opBytes{0x63}},
  1219  	{AMOVLQZX, yml_rl, Px, opBytes{0x8b}},
  1220  	{AMOVMSKPD, yxrrl, Pq, opBytes{0x50}},
  1221  	{AMOVMSKPS, yxrrl, Pm, opBytes{0x50}},
  1222  	{AMOVNTO, yxr_ml, Pe, opBytes{0xe7}},
  1223  	{AMOVNTDQA, ylddqu, Pq4, opBytes{0x2a}},
  1224  	{AMOVNTPD, yxr_ml, Pe, opBytes{0x2b}},
  1225  	{AMOVNTPS, yxr_ml, Pm, opBytes{0x2b}},
  1226  	{AMOVNTQ, ymr_ml, Pm, opBytes{0xe7}},
  1227  	{AMOVQ, ymovq, Pw8, opBytes{0x6f, 0x7f, Pf2, 0xd6, Pf3, 0x7e, Pe, 0xd6, 0x89, 0x8b, 0xb8, 0xc7, 00, 0x6e, 0x7e, Pe, 0x6e, Pe, 0x7e, 0}},
  1228  	{AMOVQOZX, ymrxr, Pf3, opBytes{0xd6, 0x7e}},
  1229  	{AMOVSB, ynone, Pb, opBytes{0xa4}},
  1230  	{AMOVSD, yxmov, Pf2, opBytes{0x10, 0x11}},
  1231  	{AMOVSL, ynone, Px, opBytes{0xa5}},
  1232  	{AMOVSQ, ynone, Pw, opBytes{0xa5}},
  1233  	{AMOVSS, yxmov, Pf3, opBytes{0x10, 0x11}},
  1234  	{AMOVSW, ynone, Pe, opBytes{0xa5}},
  1235  	{AMOVUPD, yxmov, Pe, opBytes{0x10, 0x11}},
  1236  	{AMOVUPS, yxmov, Pm, opBytes{0x10, 0x11}},
  1237  	{AMOVW, ymovw, Pe, opBytes{0x89, 0x8b, 0xb8, 0xc7, 00, 0}},
  1238  	{AMOVWLSX, yml_rl, Pm, opBytes{0xbf}},
  1239  	{AMOVWLZX, yml_rl, Pm, opBytes{0xb7}},
  1240  	{AMOVWQSX, yml_rl, Pw, opBytes{0x0f, 0xbf}},
  1241  	{AMOVWQZX, yml_rl, Pw, opBytes{0x0f, 0xb7}},
  1242  	{AMPSADBW, yxshuf, Pq, opBytes{0x3a, 0x42, 0}},
  1243  	{AMULB, ydivb, Pb, opBytes{0xf6, 04}},
  1244  	{AMULL, ydivl, Px, opBytes{0xf7, 04}},
  1245  	{AMULPD, yxm, Pe, opBytes{0x59}},
  1246  	{AMULPS, yxm, Ym, opBytes{0x59}},
  1247  	{AMULQ, ydivl, Pw, opBytes{0xf7, 04}},
  1248  	{AMULSD, yxm, Pf2, opBytes{0x59}},
  1249  	{AMULSS, yxm, Pf3, opBytes{0x59}},
  1250  	{AMULW, ydivl, Pe, opBytes{0xf7, 04}},
  1251  	{ANEGB, yscond, Pb, opBytes{0xf6, 03}},
  1252  	{ANEGL, yscond, Px, opBytes{0xf7, 03}},
  1253  	{ANEGQ, yscond, Pw, opBytes{0xf7, 03}},
  1254  	{ANEGW, yscond, Pe, opBytes{0xf7, 03}},
  1255  	{obj.ANOP, ynop, Px, opBytes{0, 0}},
  1256  	{ANOTB, yscond, Pb, opBytes{0xf6, 02}},
  1257  	{ANOTL, yscond, Px, opBytes{0xf7, 02}}, // TODO(rsc): yscond is wrong here.
  1258  	{ANOTQ, yscond, Pw, opBytes{0xf7, 02}},
  1259  	{ANOTW, yscond, Pe, opBytes{0xf7, 02}},
  1260  	{AORB, yxorb, Pb, opBytes{0x0c, 0x80, 01, 0x08, 0x0a}},
  1261  	{AORL, yaddl, Px, opBytes{0x83, 01, 0x0d, 0x81, 01, 0x09, 0x0b}},
  1262  	{AORPD, yxm, Pq, opBytes{0x56}},
  1263  	{AORPS, yxm, Pm, opBytes{0x56}},
  1264  	{AORQ, yaddl, Pw, opBytes{0x83, 01, 0x0d, 0x81, 01, 0x09, 0x0b}},
  1265  	{AORW, yaddl, Pe, opBytes{0x83, 01, 0x0d, 0x81, 01, 0x09, 0x0b}},
  1266  	{AOUTB, yin, Pb, opBytes{0xe6, 0xee}},
  1267  	{AOUTL, yin, Px, opBytes{0xe7, 0xef}},
  1268  	{AOUTW, yin, Pe, opBytes{0xe7, 0xef}},
  1269  	{AOUTSB, ynone, Pb, opBytes{0x6e}},
  1270  	{AOUTSL, ynone, Px, opBytes{0x6f}},
  1271  	{AOUTSW, ynone, Pe, opBytes{0x6f}},
  1272  	{APABSB, yxm_q4, Pq4, opBytes{0x1c}},
  1273  	{APABSD, yxm_q4, Pq4, opBytes{0x1e}},
  1274  	{APABSW, yxm_q4, Pq4, opBytes{0x1d}},
  1275  	{APACKSSLW, ymm, Py1, opBytes{0x6b, Pe, 0x6b}},
  1276  	{APACKSSWB, ymm, Py1, opBytes{0x63, Pe, 0x63}},
  1277  	{APACKUSDW, yxm_q4, Pq4, opBytes{0x2b}},
  1278  	{APACKUSWB, ymm, Py1, opBytes{0x67, Pe, 0x67}},
  1279  	{APADDB, ymm, Py1, opBytes{0xfc, Pe, 0xfc}},
  1280  	{APADDL, ymm, Py1, opBytes{0xfe, Pe, 0xfe}},
  1281  	{APADDQ, yxm, Pe, opBytes{0xd4}},
  1282  	{APADDSB, ymm, Py1, opBytes{0xec, Pe, 0xec}},
  1283  	{APADDSW, ymm, Py1, opBytes{0xed, Pe, 0xed}},
  1284  	{APADDUSB, ymm, Py1, opBytes{0xdc, Pe, 0xdc}},
  1285  	{APADDUSW, ymm, Py1, opBytes{0xdd, Pe, 0xdd}},
  1286  	{APADDW, ymm, Py1, opBytes{0xfd, Pe, 0xfd}},
  1287  	{APALIGNR, ypalignr, Pq, opBytes{0x3a, 0x0f}},
  1288  	{APAND, ymm, Py1, opBytes{0xdb, Pe, 0xdb}},
  1289  	{APANDN, ymm, Py1, opBytes{0xdf, Pe, 0xdf}},
  1290  	{APAUSE, ynone, Px, opBytes{0xf3, 0x90}},
  1291  	{APAVGB, ymm, Py1, opBytes{0xe0, Pe, 0xe0}},
  1292  	{APAVGW, ymm, Py1, opBytes{0xe3, Pe, 0xe3}},
  1293  	{APBLENDW, yxshuf, Pq, opBytes{0x3a, 0x0e, 0}},
  1294  	{APCMPEQB, ymm, Py1, opBytes{0x74, Pe, 0x74}},
  1295  	{APCMPEQL, ymm, Py1, opBytes{0x76, Pe, 0x76}},
  1296  	{APCMPEQQ, yxm_q4, Pq4, opBytes{0x29}},
  1297  	{APCMPEQW, ymm, Py1, opBytes{0x75, Pe, 0x75}},
  1298  	{APCMPGTB, ymm, Py1, opBytes{0x64, Pe, 0x64}},
  1299  	{APCMPGTL, ymm, Py1, opBytes{0x66, Pe, 0x66}},
  1300  	{APCMPGTQ, yxm_q4, Pq4, opBytes{0x37}},
  1301  	{APCMPGTW, ymm, Py1, opBytes{0x65, Pe, 0x65}},
  1302  	{APCMPISTRI, yxshuf, Pq, opBytes{0x3a, 0x63, 0}},
  1303  	{APCMPISTRM, yxshuf, Pq, opBytes{0x3a, 0x62, 0}},
  1304  	{APEXTRW, yextrw, Pq, opBytes{0xc5, 0, 0x3a, 0x15, 0}},
  1305  	{APEXTRB, yextr, Pq, opBytes{0x3a, 0x14, 00}},
  1306  	{APEXTRD, yextr, Pq, opBytes{0x3a, 0x16, 00}},
  1307  	{APEXTRQ, yextr, Pq3, opBytes{0x3a, 0x16, 00}},
  1308  	{APHADDD, ymmxmm0f38, Px, opBytes{0x0F, 0x38, 0x02, 0, 0x66, 0x0F, 0x38, 0x02, 0}},
  1309  	{APHADDSW, yxm_q4, Pq4, opBytes{0x03}},
  1310  	{APHADDW, yxm_q4, Pq4, opBytes{0x01}},
  1311  	{APHMINPOSUW, yxm_q4, Pq4, opBytes{0x41}},
  1312  	{APHSUBD, yxm_q4, Pq4, opBytes{0x06}},
  1313  	{APHSUBSW, yxm_q4, Pq4, opBytes{0x07}},
  1314  	{APHSUBW, yxm_q4, Pq4, opBytes{0x05}},
  1315  	{APINSRW, yinsrw, Pq, opBytes{0xc4, 00}},
  1316  	{APINSRB, yinsr, Pq, opBytes{0x3a, 0x20, 00}},
  1317  	{APINSRD, yinsr, Pq, opBytes{0x3a, 0x22, 00}},
  1318  	{APINSRQ, yinsr, Pq3, opBytes{0x3a, 0x22, 00}},
  1319  	{APMADDUBSW, yxm_q4, Pq4, opBytes{0x04}},
  1320  	{APMADDWL, ymm, Py1, opBytes{0xf5, Pe, 0xf5}},
  1321  	{APMAXSB, yxm_q4, Pq4, opBytes{0x3c}},
  1322  	{APMAXSD, yxm_q4, Pq4, opBytes{0x3d}},
  1323  	{APMAXSW, yxm, Pe, opBytes{0xee}},
  1324  	{APMAXUB, yxm, Pe, opBytes{0xde}},
  1325  	{APMAXUD, yxm_q4, Pq4, opBytes{0x3f}},
  1326  	{APMAXUW, yxm_q4, Pq4, opBytes{0x3e}},
  1327  	{APMINSB, yxm_q4, Pq4, opBytes{0x38}},
  1328  	{APMINSD, yxm_q4, Pq4, opBytes{0x39}},
  1329  	{APMINSW, yxm, Pe, opBytes{0xea}},
  1330  	{APMINUB, yxm, Pe, opBytes{0xda}},
  1331  	{APMINUD, yxm_q4, Pq4, opBytes{0x3b}},
  1332  	{APMINUW, yxm_q4, Pq4, opBytes{0x3a}},
  1333  	{APMOVMSKB, ymskb, Px, opBytes{Pe, 0xd7, 0xd7}},
  1334  	{APMOVSXBD, yxm_q4, Pq4, opBytes{0x21}},
  1335  	{APMOVSXBQ, yxm_q4, Pq4, opBytes{0x22}},
  1336  	{APMOVSXBW, yxm_q4, Pq4, opBytes{0x20}},
  1337  	{APMOVSXDQ, yxm_q4, Pq4, opBytes{0x25}},
  1338  	{APMOVSXWD, yxm_q4, Pq4, opBytes{0x23}},
  1339  	{APMOVSXWQ, yxm_q4, Pq4, opBytes{0x24}},
  1340  	{APMOVZXBD, yxm_q4, Pq4, opBytes{0x31}},
  1341  	{APMOVZXBQ, yxm_q4, Pq4, opBytes{0x32}},
  1342  	{APMOVZXBW, yxm_q4, Pq4, opBytes{0x30}},
  1343  	{APMOVZXDQ, yxm_q4, Pq4, opBytes{0x35}},
  1344  	{APMOVZXWD, yxm_q4, Pq4, opBytes{0x33}},
  1345  	{APMOVZXWQ, yxm_q4, Pq4, opBytes{0x34}},
  1346  	{APMULDQ, yxm_q4, Pq4, opBytes{0x28}},
  1347  	{APMULHRSW, yxm_q4, Pq4, opBytes{0x0b}},
  1348  	{APMULHUW, ymm, Py1, opBytes{0xe4, Pe, 0xe4}},
  1349  	{APMULHW, ymm, Py1, opBytes{0xe5, Pe, 0xe5}},
  1350  	{APMULLD, yxm_q4, Pq4, opBytes{0x40}},
  1351  	{APMULLW, ymm, Py1, opBytes{0xd5, Pe, 0xd5}},
  1352  	{APMULULQ, ymm, Py1, opBytes{0xf4, Pe, 0xf4}},
  1353  	{APOPAL, ynone, P32, opBytes{0x61}},
  1354  	{APOPAW, ynone, Pe, opBytes{0x61}},
  1355  	{APOPCNTW, yml_rl, Pef3, opBytes{0xb8}},
  1356  	{APOPCNTL, yml_rl, Pf3, opBytes{0xb8}},
  1357  	{APOPCNTQ, yml_rl, Pfw, opBytes{0xb8}},
  1358  	{APOPFL, ynone, P32, opBytes{0x9d}},
  1359  	{APOPFQ, ynone, Py, opBytes{0x9d}},
  1360  	{APOPFW, ynone, Pe, opBytes{0x9d}},
  1361  	{APOPL, ypopl, P32, opBytes{0x58, 0x8f, 00}},
  1362  	{APOPQ, ypopl, Py, opBytes{0x58, 0x8f, 00}},
  1363  	{APOPW, ypopl, Pe, opBytes{0x58, 0x8f, 00}},
  1364  	{APOR, ymm, Py1, opBytes{0xeb, Pe, 0xeb}},
  1365  	{APSADBW, yxm, Pq, opBytes{0xf6}},
  1366  	{APSHUFHW, yxshuf, Pf3, opBytes{0x70, 00}},
  1367  	{APSHUFL, yxshuf, Pq, opBytes{0x70, 00}},
  1368  	{APSHUFLW, yxshuf, Pf2, opBytes{0x70, 00}},
  1369  	{APSHUFW, ymshuf, Pm, opBytes{0x70, 00}},
  1370  	{APSHUFB, ymshufb, Pq, opBytes{0x38, 0x00}},
  1371  	{APSIGNB, yxm_q4, Pq4, opBytes{0x08}},
  1372  	{APSIGND, yxm_q4, Pq4, opBytes{0x0a}},
  1373  	{APSIGNW, yxm_q4, Pq4, opBytes{0x09}},
  1374  	{APSLLO, ypsdq, Pq, opBytes{0x73, 07}},
  1375  	{APSLLL, yps, Py3, opBytes{0xf2, 0x72, 06, Pe, 0xf2, Pe, 0x72, 06}},
  1376  	{APSLLQ, yps, Py3, opBytes{0xf3, 0x73, 06, Pe, 0xf3, Pe, 0x73, 06}},
  1377  	{APSLLW, yps, Py3, opBytes{0xf1, 0x71, 06, Pe, 0xf1, Pe, 0x71, 06}},
  1378  	{APSRAL, yps, Py3, opBytes{0xe2, 0x72, 04, Pe, 0xe2, Pe, 0x72, 04}},
  1379  	{APSRAW, yps, Py3, opBytes{0xe1, 0x71, 04, Pe, 0xe1, Pe, 0x71, 04}},
  1380  	{APSRLO, ypsdq, Pq, opBytes{0x73, 03}},
  1381  	{APSRLL, yps, Py3, opBytes{0xd2, 0x72, 02, Pe, 0xd2, Pe, 0x72, 02}},
  1382  	{APSRLQ, yps, Py3, opBytes{0xd3, 0x73, 02, Pe, 0xd3, Pe, 0x73, 02}},
  1383  	{APSRLW, yps, Py3, opBytes{0xd1, 0x71, 02, Pe, 0xd1, Pe, 0x71, 02}},
  1384  	{APSUBB, yxm, Pe, opBytes{0xf8}},
  1385  	{APSUBL, yxm, Pe, opBytes{0xfa}},
  1386  	{APSUBQ, yxm, Pe, opBytes{0xfb}},
  1387  	{APSUBSB, yxm, Pe, opBytes{0xe8}},
  1388  	{APSUBSW, yxm, Pe, opBytes{0xe9}},
  1389  	{APSUBUSB, yxm, Pe, opBytes{0xd8}},
  1390  	{APSUBUSW, yxm, Pe, opBytes{0xd9}},
  1391  	{APSUBW, yxm, Pe, opBytes{0xf9}},
  1392  	{APTEST, yxm_q4, Pq4, opBytes{0x17}},
  1393  	{APUNPCKHBW, ymm, Py1, opBytes{0x68, Pe, 0x68}},
  1394  	{APUNPCKHLQ, ymm, Py1, opBytes{0x6a, Pe, 0x6a}},
  1395  	{APUNPCKHQDQ, yxm, Pe, opBytes{0x6d}},
  1396  	{APUNPCKHWL, ymm, Py1, opBytes{0x69, Pe, 0x69}},
  1397  	{APUNPCKLBW, ymm, Py1, opBytes{0x60, Pe, 0x60}},
  1398  	{APUNPCKLLQ, ymm, Py1, opBytes{0x62, Pe, 0x62}},
  1399  	{APUNPCKLQDQ, yxm, Pe, opBytes{0x6c}},
  1400  	{APUNPCKLWL, ymm, Py1, opBytes{0x61, Pe, 0x61}},
  1401  	{APUSHAL, ynone, P32, opBytes{0x60}},
  1402  	{APUSHAW, ynone, Pe, opBytes{0x60}},
  1403  	{APUSHFL, ynone, P32, opBytes{0x9c}},
  1404  	{APUSHFQ, ynone, Py, opBytes{0x9c}},
  1405  	{APUSHFW, ynone, Pe, opBytes{0x9c}},
  1406  	{APUSHL, ypushl, P32, opBytes{0x50, 0xff, 06, 0x6a, 0x68}},
  1407  	{APUSHQ, ypushl, Py, opBytes{0x50, 0xff, 06, 0x6a, 0x68}},
  1408  	{APUSHW, ypushl, Pe, opBytes{0x50, 0xff, 06, 0x6a, 0x68}},
  1409  	{APXOR, ymm, Py1, opBytes{0xef, Pe, 0xef}},
  1410  	{AQUAD, ybyte, Px, opBytes{8}},
  1411  	{ARCLB, yshb, Pb, opBytes{0xd0, 02, 0xc0, 02, 0xd2, 02}},
  1412  	{ARCLL, yshl, Px, opBytes{0xd1, 02, 0xc1, 02, 0xd3, 02, 0xd3, 02}},
  1413  	{ARCLQ, yshl, Pw, opBytes{0xd1, 02, 0xc1, 02, 0xd3, 02, 0xd3, 02}},
  1414  	{ARCLW, yshl, Pe, opBytes{0xd1, 02, 0xc1, 02, 0xd3, 02, 0xd3, 02}},
  1415  	{ARCPPS, yxm, Pm, opBytes{0x53}},
  1416  	{ARCPSS, yxm, Pf3, opBytes{0x53}},
  1417  	{ARCRB, yshb, Pb, opBytes{0xd0, 03, 0xc0, 03, 0xd2, 03}},
  1418  	{ARCRL, yshl, Px, opBytes{0xd1, 03, 0xc1, 03, 0xd3, 03, 0xd3, 03}},
  1419  	{ARCRQ, yshl, Pw, opBytes{0xd1, 03, 0xc1, 03, 0xd3, 03, 0xd3, 03}},
  1420  	{ARCRW, yshl, Pe, opBytes{0xd1, 03, 0xc1, 03, 0xd3, 03, 0xd3, 03}},
  1421  	{AREP, ynone, Px, opBytes{0xf3}},
  1422  	{AREPN, ynone, Px, opBytes{0xf2}},
  1423  	{obj.ARET, ynone, Px, opBytes{0xc3}},
  1424  	{ARETFW, yret, Pe, opBytes{0xcb, 0xca}},
  1425  	{ARETFL, yret, Px, opBytes{0xcb, 0xca}},
  1426  	{ARETFQ, yret, Pw, opBytes{0xcb, 0xca}},
  1427  	{AROLB, yshb, Pb, opBytes{0xd0, 00, 0xc0, 00, 0xd2, 00}},
  1428  	{AROLL, yshl, Px, opBytes{0xd1, 00, 0xc1, 00, 0xd3, 00, 0xd3, 00}},
  1429  	{AROLQ, yshl, Pw, opBytes{0xd1, 00, 0xc1, 00, 0xd3, 00, 0xd3, 00}},
  1430  	{AROLW, yshl, Pe, opBytes{0xd1, 00, 0xc1, 00, 0xd3, 00, 0xd3, 00}},
  1431  	{ARORB, yshb, Pb, opBytes{0xd0, 01, 0xc0, 01, 0xd2, 01}},
  1432  	{ARORL, yshl, Px, opBytes{0xd1, 01, 0xc1, 01, 0xd3, 01, 0xd3, 01}},
  1433  	{ARORQ, yshl, Pw, opBytes{0xd1, 01, 0xc1, 01, 0xd3, 01, 0xd3, 01}},
  1434  	{ARORW, yshl, Pe, opBytes{0xd1, 01, 0xc1, 01, 0xd3, 01, 0xd3, 01}},
  1435  	{ARSQRTPS, yxm, Pm, opBytes{0x52}},
  1436  	{ARSQRTSS, yxm, Pf3, opBytes{0x52}},
  1437  	{ASAHF, ynone, Px, opBytes{0x9e, 00, 0x86, 0xe0, 0x50, 0x9d}}, // XCHGB AH,AL; PUSH AX; POPFL
  1438  	{ASALB, yshb, Pb, opBytes{0xd0, 04, 0xc0, 04, 0xd2, 04}},
  1439  	{ASALL, yshl, Px, opBytes{0xd1, 04, 0xc1, 04, 0xd3, 04, 0xd3, 04}},
  1440  	{ASALQ, yshl, Pw, opBytes{0xd1, 04, 0xc1, 04, 0xd3, 04, 0xd3, 04}},
  1441  	{ASALW, yshl, Pe, opBytes{0xd1, 04, 0xc1, 04, 0xd3, 04, 0xd3, 04}},
  1442  	{ASARB, yshb, Pb, opBytes{0xd0, 07, 0xc0, 07, 0xd2, 07}},
  1443  	{ASARL, yshl, Px, opBytes{0xd1, 07, 0xc1, 07, 0xd3, 07, 0xd3, 07}},
  1444  	{ASARQ, yshl, Pw, opBytes{0xd1, 07, 0xc1, 07, 0xd3, 07, 0xd3, 07}},
  1445  	{ASARW, yshl, Pe, opBytes{0xd1, 07, 0xc1, 07, 0xd3, 07, 0xd3, 07}},
  1446  	{ASBBB, yxorb, Pb, opBytes{0x1c, 0x80, 03, 0x18, 0x1a}},
  1447  	{ASBBL, yaddl, Px, opBytes{0x83, 03, 0x1d, 0x81, 03, 0x19, 0x1b}},
  1448  	{ASBBQ, yaddl, Pw, opBytes{0x83, 03, 0x1d, 0x81, 03, 0x19, 0x1b}},
  1449  	{ASBBW, yaddl, Pe, opBytes{0x83, 03, 0x1d, 0x81, 03, 0x19, 0x1b}},
  1450  	{ASCASB, ynone, Pb, opBytes{0xae}},
  1451  	{ASCASL, ynone, Px, opBytes{0xaf}},
  1452  	{ASCASQ, ynone, Pw, opBytes{0xaf}},
  1453  	{ASCASW, ynone, Pe, opBytes{0xaf}},
  1454  	{ASETCC, yscond, Pb, opBytes{0x0f, 0x93, 00}},
  1455  	{ASETCS, yscond, Pb, opBytes{0x0f, 0x92, 00}},
  1456  	{ASETEQ, yscond, Pb, opBytes{0x0f, 0x94, 00}},
  1457  	{ASETGE, yscond, Pb, opBytes{0x0f, 0x9d, 00}},
  1458  	{ASETGT, yscond, Pb, opBytes{0x0f, 0x9f, 00}},
  1459  	{ASETHI, yscond, Pb, opBytes{0x0f, 0x97, 00}},
  1460  	{ASETLE, yscond, Pb, opBytes{0x0f, 0x9e, 00}},
  1461  	{ASETLS, yscond, Pb, opBytes{0x0f, 0x96, 00}},
  1462  	{ASETLT, yscond, Pb, opBytes{0x0f, 0x9c, 00}},
  1463  	{ASETMI, yscond, Pb, opBytes{0x0f, 0x98, 00}},
  1464  	{ASETNE, yscond, Pb, opBytes{0x0f, 0x95, 00}},
  1465  	{ASETOC, yscond, Pb, opBytes{0x0f, 0x91, 00}},
  1466  	{ASETOS, yscond, Pb, opBytes{0x0f, 0x90, 00}},
  1467  	{ASETPC, yscond, Pb, opBytes{0x0f, 0x9b, 00}},
  1468  	{ASETPL, yscond, Pb, opBytes{0x0f, 0x99, 00}},
  1469  	{ASETPS, yscond, Pb, opBytes{0x0f, 0x9a, 00}},
  1470  	{ASHLB, yshb, Pb, opBytes{0xd0, 04, 0xc0, 04, 0xd2, 04}},
  1471  	{ASHLL, yshl, Px, opBytes{0xd1, 04, 0xc1, 04, 0xd3, 04, 0xd3, 04}},
  1472  	{ASHLQ, yshl, Pw, opBytes{0xd1, 04, 0xc1, 04, 0xd3, 04, 0xd3, 04}},
  1473  	{ASHLW, yshl, Pe, opBytes{0xd1, 04, 0xc1, 04, 0xd3, 04, 0xd3, 04}},
  1474  	{ASHRB, yshb, Pb, opBytes{0xd0, 05, 0xc0, 05, 0xd2, 05}},
  1475  	{ASHRL, yshl, Px, opBytes{0xd1, 05, 0xc1, 05, 0xd3, 05, 0xd3, 05}},
  1476  	{ASHRQ, yshl, Pw, opBytes{0xd1, 05, 0xc1, 05, 0xd3, 05, 0xd3, 05}},
  1477  	{ASHRW, yshl, Pe, opBytes{0xd1, 05, 0xc1, 05, 0xd3, 05, 0xd3, 05}},
  1478  	{ASHUFPD, yxshuf, Pq, opBytes{0xc6, 00}},
  1479  	{ASHUFPS, yxshuf, Pm, opBytes{0xc6, 00}},
  1480  	{ASQRTPD, yxm, Pe, opBytes{0x51}},
  1481  	{ASQRTPS, yxm, Pm, opBytes{0x51}},
  1482  	{ASQRTSD, yxm, Pf2, opBytes{0x51}},
  1483  	{ASQRTSS, yxm, Pf3, opBytes{0x51}},
  1484  	{ASTC, ynone, Px, opBytes{0xf9}},
  1485  	{ASTD, ynone, Px, opBytes{0xfd}},
  1486  	{ASTI, ynone, Px, opBytes{0xfb}},
  1487  	{ASTMXCSR, ysvrs_om, Pm, opBytes{0xae, 03, 0xae, 03}},
  1488  	{ASTOSB, ynone, Pb, opBytes{0xaa}},
  1489  	{ASTOSL, ynone, Px, opBytes{0xab}},
  1490  	{ASTOSQ, ynone, Pw, opBytes{0xab}},
  1491  	{ASTOSW, ynone, Pe, opBytes{0xab}},
  1492  	{ASUBB, yxorb, Pb, opBytes{0x2c, 0x80, 05, 0x28, 0x2a}},
  1493  	{ASUBL, yaddl, Px, opBytes{0x83, 05, 0x2d, 0x81, 05, 0x29, 0x2b}},
  1494  	{ASUBPD, yxm, Pe, opBytes{0x5c}},
  1495  	{ASUBPS, yxm, Pm, opBytes{0x5c}},
  1496  	{ASUBQ, yaddl, Pw, opBytes{0x83, 05, 0x2d, 0x81, 05, 0x29, 0x2b}},
  1497  	{ASUBSD, yxm, Pf2, opBytes{0x5c}},
  1498  	{ASUBSS, yxm, Pf3, opBytes{0x5c}},
  1499  	{ASUBW, yaddl, Pe, opBytes{0x83, 05, 0x2d, 0x81, 05, 0x29, 0x2b}},
  1500  	{ASWAPGS, ynone, Pm, opBytes{0x01, 0xf8}},
  1501  	{ASYSCALL, ynone, Px, opBytes{0x0f, 0x05}}, // fast syscall
  1502  	{ATESTB, yxorb, Pb, opBytes{0xa8, 0xf6, 00, 0x84, 0x84}},
  1503  	{ATESTL, ytestl, Px, opBytes{0xa9, 0xf7, 00, 0x85, 0x85}},
  1504  	{ATESTQ, ytestl, Pw, opBytes{0xa9, 0xf7, 00, 0x85, 0x85}},
  1505  	{ATESTW, ytestl, Pe, opBytes{0xa9, 0xf7, 00, 0x85, 0x85}},
  1506  	{ATPAUSE, ywrfsbase, Pq, opBytes{0xae, 06}},
  1507  	{obj.ATEXT, ytext, Px, opBytes{}},
  1508  	{AUCOMISD, yxm, Pe, opBytes{0x2e}},
  1509  	{AUCOMISS, yxm, Pm, opBytes{0x2e}},
  1510  	{AUNPCKHPD, yxm, Pe, opBytes{0x15}},
  1511  	{AUNPCKHPS, yxm, Pm, opBytes{0x15}},
  1512  	{AUNPCKLPD, yxm, Pe, opBytes{0x14}},
  1513  	{AUNPCKLPS, yxm, Pm, opBytes{0x14}},
  1514  	{AUMONITOR, ywrfsbase, Pf3, opBytes{0xae, 06}},
  1515  	{AVERR, ydivl, Pm, opBytes{0x00, 04}},
  1516  	{AVERW, ydivl, Pm, opBytes{0x00, 05}},
  1517  	{AWAIT, ynone, Px, opBytes{0x9b}},
  1518  	{AWORD, ybyte, Px, opBytes{2}},
  1519  	{AXCHGB, yml_mb, Pb, opBytes{0x86, 0x86}},
  1520  	{AXCHGL, yxchg, Px, opBytes{0x90, 0x90, 0x87, 0x87}},
  1521  	{AXCHGQ, yxchg, Pw, opBytes{0x90, 0x90, 0x87, 0x87}},
  1522  	{AXCHGW, yxchg, Pe, opBytes{0x90, 0x90, 0x87, 0x87}},
  1523  	{AXLAT, ynone, Px, opBytes{0xd7}},
  1524  	{AXORB, yxorb, Pb, opBytes{0x34, 0x80, 06, 0x30, 0x32}},
  1525  	{AXORL, yaddl, Px, opBytes{0x83, 06, 0x35, 0x81, 06, 0x31, 0x33}},
  1526  	{AXORPD, yxm, Pe, opBytes{0x57}},
  1527  	{AXORPS, yxm, Pm, opBytes{0x57}},
  1528  	{AXORQ, yaddl, Pw, opBytes{0x83, 06, 0x35, 0x81, 06, 0x31, 0x33}},
  1529  	{AXORW, yaddl, Pe, opBytes{0x83, 06, 0x35, 0x81, 06, 0x31, 0x33}},
  1530  	{AFMOVB, yfmvx, Px, opBytes{0xdf, 04}},
  1531  	{AFMOVBP, yfmvp, Px, opBytes{0xdf, 06}},
  1532  	{AFMOVD, yfmvd, Px, opBytes{0xdd, 00, 0xdd, 02, 0xd9, 00, 0xdd, 02}},
  1533  	{AFMOVDP, yfmvdp, Px, opBytes{0xdd, 03, 0xdd, 03}},
  1534  	{AFMOVF, yfmvf, Px, opBytes{0xd9, 00, 0xd9, 02}},
  1535  	{AFMOVFP, yfmvp, Px, opBytes{0xd9, 03}},
  1536  	{AFMOVL, yfmvf, Px, opBytes{0xdb, 00, 0xdb, 02}},
  1537  	{AFMOVLP, yfmvp, Px, opBytes{0xdb, 03}},
  1538  	{AFMOVV, yfmvx, Px, opBytes{0xdf, 05}},
  1539  	{AFMOVVP, yfmvp, Px, opBytes{0xdf, 07}},
  1540  	{AFMOVW, yfmvf, Px, opBytes{0xdf, 00, 0xdf, 02}},
  1541  	{AFMOVWP, yfmvp, Px, opBytes{0xdf, 03}},
  1542  	{AFMOVX, yfmvx, Px, opBytes{0xdb, 05}},
  1543  	{AFMOVXP, yfmvp, Px, opBytes{0xdb, 07}},
  1544  	{AFCMOVCC, yfcmv, Px, opBytes{0xdb, 00}},
  1545  	{AFCMOVCS, yfcmv, Px, opBytes{0xda, 00}},
  1546  	{AFCMOVEQ, yfcmv, Px, opBytes{0xda, 01}},
  1547  	{AFCMOVHI, yfcmv, Px, opBytes{0xdb, 02}},
  1548  	{AFCMOVLS, yfcmv, Px, opBytes{0xda, 02}},
  1549  	{AFCMOVB, yfcmv, Px, opBytes{0xda, 00}},
  1550  	{AFCMOVBE, yfcmv, Px, opBytes{0xda, 02}},
  1551  	{AFCMOVNB, yfcmv, Px, opBytes{0xdb, 00}},
  1552  	{AFCMOVNBE, yfcmv, Px, opBytes{0xdb, 02}},
  1553  	{AFCMOVE, yfcmv, Px, opBytes{0xda, 01}},
  1554  	{AFCMOVNE, yfcmv, Px, opBytes{0xdb, 01}},
  1555  	{AFCMOVNU, yfcmv, Px, opBytes{0xdb, 03}},
  1556  	{AFCMOVU, yfcmv, Px, opBytes{0xda, 03}},
  1557  	{AFCMOVUN, yfcmv, Px, opBytes{0xda, 03}},
  1558  	{AFCOMD, yfadd, Px, opBytes{0xdc, 02, 0xd8, 02, 0xdc, 02}},  // botch
  1559  	{AFCOMDP, yfadd, Px, opBytes{0xdc, 03, 0xd8, 03, 0xdc, 03}}, // botch
  1560  	{AFCOMDPP, ycompp, Px, opBytes{0xde, 03}},
  1561  	{AFCOMF, yfmvx, Px, opBytes{0xd8, 02}},
  1562  	{AFCOMFP, yfmvx, Px, opBytes{0xd8, 03}},
  1563  	{AFCOMI, yfcmv, Px, opBytes{0xdb, 06}},
  1564  	{AFCOMIP, yfcmv, Px, opBytes{0xdf, 06}},
  1565  	{AFCOML, yfmvx, Px, opBytes{0xda, 02}},
  1566  	{AFCOMLP, yfmvx, Px, opBytes{0xda, 03}},
  1567  	{AFCOMW, yfmvx, Px, opBytes{0xde, 02}},
  1568  	{AFCOMWP, yfmvx, Px, opBytes{0xde, 03}},
  1569  	{AFUCOM, ycompp, Px, opBytes{0xdd, 04}},
  1570  	{AFUCOMI, ycompp, Px, opBytes{0xdb, 05}},
  1571  	{AFUCOMIP, ycompp, Px, opBytes{0xdf, 05}},
  1572  	{AFUCOMP, ycompp, Px, opBytes{0xdd, 05}},
  1573  	{AFUCOMPP, ycompp, Px, opBytes{0xda, 13}},
  1574  	{AFADDDP, ycompp, Px, opBytes{0xde, 00}},
  1575  	{AFADDW, yfmvx, Px, opBytes{0xde, 00}},
  1576  	{AFADDL, yfmvx, Px, opBytes{0xda, 00}},
  1577  	{AFADDF, yfmvx, Px, opBytes{0xd8, 00}},
  1578  	{AFADDD, yfadd, Px, opBytes{0xdc, 00, 0xd8, 00, 0xdc, 00}},
  1579  	{AFMULDP, ycompp, Px, opBytes{0xde, 01}},
  1580  	{AFMULW, yfmvx, Px, opBytes{0xde, 01}},
  1581  	{AFMULL, yfmvx, Px, opBytes{0xda, 01}},
  1582  	{AFMULF, yfmvx, Px, opBytes{0xd8, 01}},
  1583  	{AFMULD, yfadd, Px, opBytes{0xdc, 01, 0xd8, 01, 0xdc, 01}},
  1584  	{AFSUBDP, ycompp, Px, opBytes{0xde, 05}},
  1585  	{AFSUBW, yfmvx, Px, opBytes{0xde, 04}},
  1586  	{AFSUBL, yfmvx, Px, opBytes{0xda, 04}},
  1587  	{AFSUBF, yfmvx, Px, opBytes{0xd8, 04}},
  1588  	{AFSUBD, yfadd, Px, opBytes{0xdc, 04, 0xd8, 04, 0xdc, 05}},
  1589  	{AFSUBRDP, ycompp, Px, opBytes{0xde, 04}},
  1590  	{AFSUBRW, yfmvx, Px, opBytes{0xde, 05}},
  1591  	{AFSUBRL, yfmvx, Px, opBytes{0xda, 05}},
  1592  	{AFSUBRF, yfmvx, Px, opBytes{0xd8, 05}},
  1593  	{AFSUBRD, yfadd, Px, opBytes{0xdc, 05, 0xd8, 05, 0xdc, 04}},
  1594  	{AFDIVDP, ycompp, Px, opBytes{0xde, 07}},
  1595  	{AFDIVW, yfmvx, Px, opBytes{0xde, 06}},
  1596  	{AFDIVL, yfmvx, Px, opBytes{0xda, 06}},
  1597  	{AFDIVF, yfmvx, Px, opBytes{0xd8, 06}},
  1598  	{AFDIVD, yfadd, Px, opBytes{0xdc, 06, 0xd8, 06, 0xdc, 07}},
  1599  	{AFDIVRDP, ycompp, Px, opBytes{0xde, 06}},
  1600  	{AFDIVRW, yfmvx, Px, opBytes{0xde, 07}},
  1601  	{AFDIVRL, yfmvx, Px, opBytes{0xda, 07}},
  1602  	{AFDIVRF, yfmvx, Px, opBytes{0xd8, 07}},
  1603  	{AFDIVRD, yfadd, Px, opBytes{0xdc, 07, 0xd8, 07, 0xdc, 06}},
  1604  	{AFXCHD, yfxch, Px, opBytes{0xd9, 01, 0xd9, 01}},
  1605  	{AFFREE, nil, 0, opBytes{}},
  1606  	{AFLDCW, ysvrs_mo, Px, opBytes{0xd9, 05, 0xd9, 05}},
  1607  	{AFLDENV, ysvrs_mo, Px, opBytes{0xd9, 04, 0xd9, 04}},
  1608  	{AFRSTOR, ysvrs_mo, Px, opBytes{0xdd, 04, 0xdd, 04}},
  1609  	{AFSAVE, ysvrs_om, Px, opBytes{0xdd, 06, 0xdd, 06}},
  1610  	{AFSTCW, ysvrs_om, Px, opBytes{0xd9, 07, 0xd9, 07}},
  1611  	{AFSTENV, ysvrs_om, Px, opBytes{0xd9, 06, 0xd9, 06}},
  1612  	{AFSTSW, ystsw, Px, opBytes{0xdd, 07, 0xdf, 0xe0}},
  1613  	{AF2XM1, ynone, Px, opBytes{0xd9, 0xf0}},
  1614  	{AFABS, ynone, Px, opBytes{0xd9, 0xe1}},
  1615  	{AFBLD, ysvrs_mo, Px, opBytes{0xdf, 04}},
  1616  	{AFBSTP, yclflush, Px, opBytes{0xdf, 06}},
  1617  	{AFCHS, ynone, Px, opBytes{0xd9, 0xe0}},
  1618  	{AFCLEX, ynone, Px, opBytes{0xdb, 0xe2}},
  1619  	{AFCOS, ynone, Px, opBytes{0xd9, 0xff}},
  1620  	{AFDECSTP, ynone, Px, opBytes{0xd9, 0xf6}},
  1621  	{AFINCSTP, ynone, Px, opBytes{0xd9, 0xf7}},
  1622  	{AFINIT, ynone, Px, opBytes{0xdb, 0xe3}},
  1623  	{AFLD1, ynone, Px, opBytes{0xd9, 0xe8}},
  1624  	{AFLDL2E, ynone, Px, opBytes{0xd9, 0xea}},
  1625  	{AFLDL2T, ynone, Px, opBytes{0xd9, 0xe9}},
  1626  	{AFLDLG2, ynone, Px, opBytes{0xd9, 0xec}},
  1627  	{AFLDLN2, ynone, Px, opBytes{0xd9, 0xed}},
  1628  	{AFLDPI, ynone, Px, opBytes{0xd9, 0xeb}},
  1629  	{AFLDZ, ynone, Px, opBytes{0xd9, 0xee}},
  1630  	{AFNOP, ynone, Px, opBytes{0xd9, 0xd0}},
  1631  	{AFPATAN, ynone, Px, opBytes{0xd9, 0xf3}},
  1632  	{AFPREM, ynone, Px, opBytes{0xd9, 0xf8}},
  1633  	{AFPREM1, ynone, Px, opBytes{0xd9, 0xf5}},
  1634  	{AFPTAN, ynone, Px, opBytes{0xd9, 0xf2}},
  1635  	{AFRNDINT, ynone, Px, opBytes{0xd9, 0xfc}},
  1636  	{AFSCALE, ynone, Px, opBytes{0xd9, 0xfd}},
  1637  	{AFSIN, ynone, Px, opBytes{0xd9, 0xfe}},
  1638  	{AFSINCOS, ynone, Px, opBytes{0xd9, 0xfb}},
  1639  	{AFSQRT, ynone, Px, opBytes{0xd9, 0xfa}},
  1640  	{AFTST, ynone, Px, opBytes{0xd9, 0xe4}},
  1641  	{AFXAM, ynone, Px, opBytes{0xd9, 0xe5}},
  1642  	{AFXTRACT, ynone, Px, opBytes{0xd9, 0xf4}},
  1643  	{AFYL2X, ynone, Px, opBytes{0xd9, 0xf1}},
  1644  	{AFYL2XP1, ynone, Px, opBytes{0xd9, 0xf9}},
  1645  	{ACMPXCHGB, yrb_mb, Pb, opBytes{0x0f, 0xb0}},
  1646  	{ACMPXCHGL, yrl_ml, Px, opBytes{0x0f, 0xb1}},
  1647  	{ACMPXCHGW, yrl_ml, Pe, opBytes{0x0f, 0xb1}},
  1648  	{ACMPXCHGQ, yrl_ml, Pw, opBytes{0x0f, 0xb1}},
  1649  	{ACMPXCHG8B, yscond, Pm, opBytes{0xc7, 01}},
  1650  	{ACMPXCHG16B, yscond, Pw, opBytes{0x0f, 0xc7, 01}},
  1651  	{AINVD, ynone, Pm, opBytes{0x08}},
  1652  	{AINVLPG, ydivb, Pm, opBytes{0x01, 07}},
  1653  	{AINVPCID, ycrc32l, Pe, opBytes{0x0f, 0x38, 0x82, 0}},
  1654  	{ALFENCE, ynone, Pm, opBytes{0xae, 0xe8}},
  1655  	{AMFENCE, ynone, Pm, opBytes{0xae, 0xf0}},
  1656  	{AMOVNTIL, yrl_ml, Pm, opBytes{0xc3}},
  1657  	{AMOVNTIQ, yrl_ml, Pw, opBytes{0x0f, 0xc3}},
  1658  	{ARDPKRU, ynone, Pm, opBytes{0x01, 0xee, 0}},
  1659  	{ARDMSR, ynone, Pm, opBytes{0x32}},
  1660  	{ARDPMC, ynone, Pm, opBytes{0x33}},
  1661  	{ARDTSC, ynone, Pm, opBytes{0x31}},
  1662  	{ARSM, ynone, Pm, opBytes{0xaa}},
  1663  	{ASFENCE, ynone, Pm, opBytes{0xae, 0xf8}},
  1664  	{ASYSRET, ynone, Pm, opBytes{0x07}},
  1665  	{AWBINVD, ynone, Pm, opBytes{0x09}},
  1666  	{AWRMSR, ynone, Pm, opBytes{0x30}},
  1667  	{AWRPKRU, ynone, Pm, opBytes{0x01, 0xef, 0}},
  1668  	{AXADDB, yrb_mb, Pb, opBytes{0x0f, 0xc0}},
  1669  	{AXADDL, yrl_ml, Px, opBytes{0x0f, 0xc1}},
  1670  	{AXADDQ, yrl_ml, Pw, opBytes{0x0f, 0xc1}},
  1671  	{AXADDW, yrl_ml, Pe, opBytes{0x0f, 0xc1}},
  1672  	{ACRC32B, ycrc32b, Px, opBytes{0xf2, 0x0f, 0x38, 0xf0, 0}},
  1673  	{ACRC32L, ycrc32l, Px, opBytes{0xf2, 0x0f, 0x38, 0xf1, 0}},
  1674  	{ACRC32Q, ycrc32l, Pw, opBytes{0xf2, 0x0f, 0x38, 0xf1, 0}},
  1675  	{ACRC32W, ycrc32l, Pe, opBytes{0xf2, 0x0f, 0x38, 0xf1, 0}},
  1676  	{APREFETCHT0, yprefetch, Pm, opBytes{0x18, 01}},
  1677  	{APREFETCHT1, yprefetch, Pm, opBytes{0x18, 02}},
  1678  	{APREFETCHT2, yprefetch, Pm, opBytes{0x18, 03}},
  1679  	{APREFETCHNTA, yprefetch, Pm, opBytes{0x18, 00}},
  1680  	{AMOVQL, yrl_ml, Px, opBytes{0x89}},
  1681  	{obj.AUNDEF, ynone, Px, opBytes{0x0f, 0x0b}},
  1682  	{AAESENC, yaes, Pq, opBytes{0x38, 0xdc, 0}},
  1683  	{AAESENCLAST, yaes, Pq, opBytes{0x38, 0xdd, 0}},
  1684  	{AAESDEC, yaes, Pq, opBytes{0x38, 0xde, 0}},
  1685  	{AAESDECLAST, yaes, Pq, opBytes{0x38, 0xdf, 0}},
  1686  	{AAESIMC, yaes, Pq, opBytes{0x38, 0xdb, 0}},
  1687  	{AAESKEYGENASSIST, yxshuf, Pq, opBytes{0x3a, 0xdf, 0}},
  1688  	{AROUNDPD, yxshuf, Pq, opBytes{0x3a, 0x09, 0}},
  1689  	{AROUNDPS, yxshuf, Pq, opBytes{0x3a, 0x08, 0}},
  1690  	{AROUNDSD, yxshuf, Pq, opBytes{0x3a, 0x0b, 0}},
  1691  	{AROUNDSS, yxshuf, Pq, opBytes{0x3a, 0x0a, 0}},
  1692  	{APSHUFD, yxshuf, Pq, opBytes{0x70, 0}},
  1693  	{APCLMULQDQ, yxshuf, Pq, opBytes{0x3a, 0x44, 0}},
  1694  	{APCMPESTRI, yxshuf, Pq, opBytes{0x3a, 0x61, 0}},
  1695  	{APCMPESTRM, yxshuf, Pq, opBytes{0x3a, 0x60, 0}},
  1696  	{AMOVDDUP, yxm, Pf2, opBytes{0x12}},
  1697  	{AMOVSHDUP, yxm, Pf3, opBytes{0x16}},
  1698  	{AMOVSLDUP, yxm, Pf3, opBytes{0x12}},
  1699  	{ARDTSCP, ynone, Pm, opBytes{0x01, 0xf9, 0}},
  1700  	{ASTAC, ynone, Pm, opBytes{0x01, 0xcb, 0}},
  1701  	{AUD1, ynone, Pm, opBytes{0xb9, 0}},
  1702  	{AUD2, ynone, Pm, opBytes{0x0b, 0}},
  1703  	{AUMWAIT, ywrfsbase, Pf2, opBytes{0xae, 06}},
  1704  	{ASYSENTER, ynone, Px, opBytes{0x0f, 0x34, 0}},
  1705  	{ASYSENTER64, ynone, Pw, opBytes{0x0f, 0x34, 0}},
  1706  	{ASYSEXIT, ynone, Px, opBytes{0x0f, 0x35, 0}},
  1707  	{ASYSEXIT64, ynone, Pw, opBytes{0x0f, 0x35, 0}},
  1708  	{ALMSW, ydivl, Pm, opBytes{0x01, 06}},
  1709  	{ALLDT, ydivl, Pm, opBytes{0x00, 02}},
  1710  	{ALIDT, ysvrs_mo, Pm, opBytes{0x01, 03}},
  1711  	{ALGDT, ysvrs_mo, Pm, opBytes{0x01, 02}},
  1712  	{ATZCNTW, ycrc32l, Pe, opBytes{0xf3, 0x0f, 0xbc, 0}},
  1713  	{ATZCNTL, ycrc32l, Px, opBytes{0xf3, 0x0f, 0xbc, 0}},
  1714  	{ATZCNTQ, ycrc32l, Pw, opBytes{0xf3, 0x0f, 0xbc, 0}},
  1715  	{AXRSTOR, ydivl, Px, opBytes{0x0f, 0xae, 05}},
  1716  	{AXRSTOR64, ydivl, Pw, opBytes{0x0f, 0xae, 05}},
  1717  	{AXRSTORS, ydivl, Px, opBytes{0x0f, 0xc7, 03}},
  1718  	{AXRSTORS64, ydivl, Pw, opBytes{0x0f, 0xc7, 03}},
  1719  	{AXSAVE, yclflush, Px, opBytes{0x0f, 0xae, 04}},
  1720  	{AXSAVE64, yclflush, Pw, opBytes{0x0f, 0xae, 04}},
  1721  	{AXSAVEOPT, yclflush, Px, opBytes{0x0f, 0xae, 06}},
  1722  	{AXSAVEOPT64, yclflush, Pw, opBytes{0x0f, 0xae, 06}},
  1723  	{AXSAVEC, yclflush, Px, opBytes{0x0f, 0xc7, 04}},
  1724  	{AXSAVEC64, yclflush, Pw, opBytes{0x0f, 0xc7, 04}},
  1725  	{AXSAVES, yclflush, Px, opBytes{0x0f, 0xc7, 05}},
  1726  	{AXSAVES64, yclflush, Pw, opBytes{0x0f, 0xc7, 05}},
  1727  	{ASGDT, yclflush, Pm, opBytes{0x01, 00}},
  1728  	{ASIDT, yclflush, Pm, opBytes{0x01, 01}},
  1729  	{ARDRANDW, yrdrand, Pe, opBytes{0x0f, 0xc7, 06}},
  1730  	{ARDRANDL, yrdrand, Px, opBytes{0x0f, 0xc7, 06}},
  1731  	{ARDRANDQ, yrdrand, Pw, opBytes{0x0f, 0xc7, 06}},
  1732  	{ARDSEEDW, yrdrand, Pe, opBytes{0x0f, 0xc7, 07}},
  1733  	{ARDSEEDL, yrdrand, Px, opBytes{0x0f, 0xc7, 07}},
  1734  	{ARDSEEDQ, yrdrand, Pw, opBytes{0x0f, 0xc7, 07}},
  1735  	{ASTRW, yincq, Pe, opBytes{0x0f, 0x00, 01}},
  1736  	{ASTRL, yincq, Px, opBytes{0x0f, 0x00, 01}},
  1737  	{ASTRQ, yincq, Pw, opBytes{0x0f, 0x00, 01}},
  1738  	{AXSETBV, ynone, Pm, opBytes{0x01, 0xd1, 0}},
  1739  	{AMOVBEW, ymovbe, Pq, opBytes{0x38, 0xf0, 0, 0x38, 0xf1, 0}},
  1740  	{AMOVBEL, ymovbe, Pm, opBytes{0x38, 0xf0, 0, 0x38, 0xf1, 0}},
  1741  	{AMOVBEQ, ymovbe, Pw, opBytes{0x0f, 0x38, 0xf0, 0, 0x0f, 0x38, 0xf1, 0}},
  1742  	{ANOPW, ydivl, Pe, opBytes{0x0f, 0x1f, 00}},
  1743  	{ANOPL, ydivl, Px, opBytes{0x0f, 0x1f, 00}},
  1744  	{ASLDTW, yincq, Pe, opBytes{0x0f, 0x00, 00}},
  1745  	{ASLDTL, yincq, Px, opBytes{0x0f, 0x00, 00}},
  1746  	{ASLDTQ, yincq, Pw, opBytes{0x0f, 0x00, 00}},
  1747  	{ASMSWW, yincq, Pe, opBytes{0x0f, 0x01, 04}},
  1748  	{ASMSWL, yincq, Px, opBytes{0x0f, 0x01, 04}},
  1749  	{ASMSWQ, yincq, Pw, opBytes{0x0f, 0x01, 04}},
  1750  	{ABLENDVPS, yblendvpd, Pq4, opBytes{0x14}},
  1751  	{ABLENDVPD, yblendvpd, Pq4, opBytes{0x15}},
  1752  	{APBLENDVB, yblendvpd, Pq4, opBytes{0x10}},
  1753  	{ASHA1MSG1, yaes, Px, opBytes{0x0f, 0x38, 0xc9, 0}},
  1754  	{ASHA1MSG2, yaes, Px, opBytes{0x0f, 0x38, 0xca, 0}},
  1755  	{ASHA1NEXTE, yaes, Px, opBytes{0x0f, 0x38, 0xc8, 0}},
  1756  	{ASHA256MSG1, yaes, Px, opBytes{0x0f, 0x38, 0xcc, 0}},
  1757  	{ASHA256MSG2, yaes, Px, opBytes{0x0f, 0x38, 0xcd, 0}},
  1758  	{ASHA1RNDS4, ysha1rnds4, Pm, opBytes{0x3a, 0xcc, 0}},
  1759  	{ASHA256RNDS2, ysha256rnds2, Px, opBytes{0x0f, 0x38, 0xcb, 0}},
  1760  	{ARDFSBASEL, yrdrand, Pf3, opBytes{0xae, 00}},
  1761  	{ARDFSBASEQ, yrdrand, Pfw, opBytes{0xae, 00}},
  1762  	{ARDGSBASEL, yrdrand, Pf3, opBytes{0xae, 01}},
  1763  	{ARDGSBASEQ, yrdrand, Pfw, opBytes{0xae, 01}},
  1764  	{AWRFSBASEL, ywrfsbase, Pf3, opBytes{0xae, 02}},
  1765  	{AWRFSBASEQ, ywrfsbase, Pfw, opBytes{0xae, 02}},
  1766  	{AWRGSBASEL, ywrfsbase, Pf3, opBytes{0xae, 03}},
  1767  	{AWRGSBASEQ, ywrfsbase, Pfw, opBytes{0xae, 03}},
  1768  	{ALFSW, ym_rl, Pe, opBytes{0x0f, 0xb4}},
  1769  	{ALFSL, ym_rl, Px, opBytes{0x0f, 0xb4}},
  1770  	{ALFSQ, ym_rl, Pw, opBytes{0x0f, 0xb4}},
  1771  	{ALGSW, ym_rl, Pe, opBytes{0x0f, 0xb5}},
  1772  	{ALGSL, ym_rl, Px, opBytes{0x0f, 0xb5}},
  1773  	{ALGSQ, ym_rl, Pw, opBytes{0x0f, 0xb5}},
  1774  	{ALSSW, ym_rl, Pe, opBytes{0x0f, 0xb2}},
  1775  	{ALSSL, ym_rl, Px, opBytes{0x0f, 0xb2}},
  1776  	{ALSSQ, ym_rl, Pw, opBytes{0x0f, 0xb2}},
  1777  	{ARDPID, yrdrand, Pf3, opBytes{0xc7, 07}},
  1778  
  1779  	{ABLENDPD, yxshuf, Pq, opBytes{0x3a, 0x0d, 0}},
  1780  	{ABLENDPS, yxshuf, Pq, opBytes{0x3a, 0x0c, 0}},
  1781  	{AXACQUIRE, ynone, Px, opBytes{0xf2}},
  1782  	{AXRELEASE, ynone, Px, opBytes{0xf3}},
  1783  	{AXBEGIN, yxbegin, Px, opBytes{0xc7, 0xf8}},
  1784  	{AXABORT, yxabort, Px, opBytes{0xc6, 0xf8}},
  1785  	{AXEND, ynone, Px, opBytes{0x0f, 01, 0xd5}},
  1786  	{AXTEST, ynone, Px, opBytes{0x0f, 01, 0xd6}},
  1787  	{AXGETBV, ynone, Pm, opBytes{01, 0xd0}},
  1788  	{obj.AFUNCDATA, yfuncdata, Px, opBytes{0, 0}},
  1789  	{obj.APCDATA, ypcdata, Px, opBytes{0, 0}},
  1790  	{obj.ADUFFCOPY, yduff, Px, opBytes{0xe8}},
  1791  	{obj.ADUFFZERO, yduff, Px, opBytes{0xe8}},
  1792  
  1793  	{obj.AEND, nil, 0, opBytes{}},
  1794  	{0, nil, 0, opBytes{}},
  1795  }
  1796  
  1797  var opindex [(ALAST + 1) & obj.AMask]*Optab
  1798  
  1799  // useAbs reports whether s describes a symbol that must avoid pc-relative addressing.
  1800  // This happens on systems like Solaris that call .so functions instead of system calls.
  1801  // It does not seem to be necessary for any other systems. This is probably working
  1802  // around a Solaris-specific bug that should be fixed differently, but we don't know
  1803  // what that bug is. And this does fix it.
  1804  func useAbs(ctxt *obj.Link, s *obj.LSym) bool {
  1805  	if ctxt.Headtype == objabi.Hsolaris {
  1806  		// All the Solaris dynamic imports from libc.so begin with "libc_".
  1807  		return strings.HasPrefix(s.Name, "libc_")
  1808  	}
  1809  	return ctxt.Arch.Family == sys.I386 && !ctxt.Flag_shared
  1810  }
  1811  
  1812  // single-instruction no-ops of various lengths.
  1813  // constructed by hand and disassembled with gdb to verify.
  1814  // see http://www.agner.org/optimize/optimizing_assembly.pdf for discussion.
  1815  var nop = [][16]uint8{
  1816  	{0x90},
  1817  	{0x66, 0x90},
  1818  	{0x0F, 0x1F, 0x00},
  1819  	{0x0F, 0x1F, 0x40, 0x00},
  1820  	{0x0F, 0x1F, 0x44, 0x00, 0x00},
  1821  	{0x66, 0x0F, 0x1F, 0x44, 0x00, 0x00},
  1822  	{0x0F, 0x1F, 0x80, 0x00, 0x00, 0x00, 0x00},
  1823  	{0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
  1824  	{0x66, 0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
  1825  }
  1826  
  1827  // Native Client rejects the repeated 0x66 prefix.
  1828  // {0x66, 0x66, 0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
  1829  func fillnop(p []byte, n int) {
  1830  	var m int
  1831  
  1832  	for n > 0 {
  1833  		m = n
  1834  		if m > len(nop) {
  1835  			m = len(nop)
  1836  		}
  1837  		copy(p[:m], nop[m-1][:m])
  1838  		p = p[m:]
  1839  		n -= m
  1840  	}
  1841  }
  1842  
  1843  func noppad(ctxt *obj.Link, s *obj.LSym, c int32, pad int32) int32 {
  1844  	s.Grow(int64(c) + int64(pad))
  1845  	fillnop(s.P[c:], int(pad))
  1846  	return c + pad
  1847  }
  1848  
  1849  func spadjop(ctxt *obj.Link, l, q obj.As) obj.As {
  1850  	if ctxt.Arch.Family != sys.AMD64 || ctxt.Arch.PtrSize == 4 {
  1851  		return l
  1852  	}
  1853  	return q
  1854  }
  1855  
  1856  // isJump returns whether p is a jump instruction.
  1857  // It is used to ensure that no standalone or macro-fused jump will straddle
  1858  // or end on a 32 byte boundary by inserting NOPs before the jumps.
  1859  func isJump(p *obj.Prog) bool {
  1860  	return p.To.Target() != nil || p.As == obj.AJMP || p.As == obj.ACALL ||
  1861  		p.As == obj.ARET || p.As == obj.ADUFFCOPY || p.As == obj.ADUFFZERO
  1862  }
  1863  
  1864  // lookForJCC returns the first real instruction starting from p, if that instruction is a conditional
  1865  // jump. Otherwise, nil is returned.
  1866  func lookForJCC(p *obj.Prog) *obj.Prog {
  1867  	// Skip any PCDATA, FUNCDATA or NOP instructions
  1868  	var q *obj.Prog
  1869  	for q = p.Link; q != nil && (q.As == obj.APCDATA || q.As == obj.AFUNCDATA || q.As == obj.ANOP); q = q.Link {
  1870  	}
  1871  
  1872  	if q == nil || q.To.Target() == nil || p.As == obj.AJMP || p.As == obj.ACALL {
  1873  		return nil
  1874  	}
  1875  
  1876  	switch q.As {
  1877  	case AJOS, AJOC, AJCS, AJCC, AJEQ, AJNE, AJLS, AJHI,
  1878  		AJMI, AJPL, AJPS, AJPC, AJLT, AJGE, AJLE, AJGT:
  1879  	default:
  1880  		return nil
  1881  	}
  1882  
  1883  	return q
  1884  }
  1885  
  1886  // fusedJump determines whether p can be fused with a subsequent conditional jump instruction.
  1887  // If it can, we return true followed by the total size of the fused jump. If it can't, we return false.
  1888  // Macro fusion rules are derived from the Intel Optimization Manual (April 2019) section 3.4.2.2.
  1889  func fusedJump(p *obj.Prog) (bool, uint8) {
  1890  	var fusedSize uint8
  1891  
  1892  	// The first instruction in a macro fused pair may be preceded by the LOCK prefix,
  1893  	// or possibly an XACQUIRE/XRELEASE prefix followed by a LOCK prefix. If it is, we
  1894  	// need to be careful to insert any padding before the locks rather than directly after them.
  1895  
  1896  	if p.As == AXRELEASE || p.As == AXACQUIRE {
  1897  		fusedSize += p.Isize
  1898  		for p = p.Link; p != nil && (p.As == obj.APCDATA || p.As == obj.AFUNCDATA); p = p.Link {
  1899  		}
  1900  		if p == nil {
  1901  			return false, 0
  1902  		}
  1903  	}
  1904  	if p.As == ALOCK {
  1905  		fusedSize += p.Isize
  1906  		for p = p.Link; p != nil && (p.As == obj.APCDATA || p.As == obj.AFUNCDATA); p = p.Link {
  1907  		}
  1908  		if p == nil {
  1909  			return false, 0
  1910  		}
  1911  	}
  1912  	cmp := p.As == ACMPB || p.As == ACMPL || p.As == ACMPQ || p.As == ACMPW
  1913  
  1914  	cmpAddSub := p.As == AADDB || p.As == AADDL || p.As == AADDW || p.As == AADDQ ||
  1915  		p.As == ASUBB || p.As == ASUBL || p.As == ASUBW || p.As == ASUBQ || cmp
  1916  
  1917  	testAnd := p.As == ATESTB || p.As == ATESTL || p.As == ATESTQ || p.As == ATESTW ||
  1918  		p.As == AANDB || p.As == AANDL || p.As == AANDQ || p.As == AANDW
  1919  
  1920  	incDec := p.As == AINCB || p.As == AINCL || p.As == AINCQ || p.As == AINCW ||
  1921  		p.As == ADECB || p.As == ADECL || p.As == ADECQ || p.As == ADECW
  1922  
  1923  	if !cmpAddSub && !testAnd && !incDec {
  1924  		return false, 0
  1925  	}
  1926  
  1927  	if !incDec {
  1928  		var argOne obj.AddrType
  1929  		var argTwo obj.AddrType
  1930  		if cmp {
  1931  			argOne = p.From.Type
  1932  			argTwo = p.To.Type
  1933  		} else {
  1934  			argOne = p.To.Type
  1935  			argTwo = p.From.Type
  1936  		}
  1937  		if argOne == obj.TYPE_REG {
  1938  			if argTwo != obj.TYPE_REG && argTwo != obj.TYPE_CONST && argTwo != obj.TYPE_MEM {
  1939  				return false, 0
  1940  			}
  1941  		} else if argOne == obj.TYPE_MEM {
  1942  			if argTwo != obj.TYPE_REG {
  1943  				return false, 0
  1944  			}
  1945  		} else {
  1946  			return false, 0
  1947  		}
  1948  	}
  1949  
  1950  	fusedSize += p.Isize
  1951  	jmp := lookForJCC(p)
  1952  	if jmp == nil {
  1953  		return false, 0
  1954  	}
  1955  
  1956  	fusedSize += jmp.Isize
  1957  
  1958  	if testAnd {
  1959  		return true, fusedSize
  1960  	}
  1961  
  1962  	if jmp.As == AJOC || jmp.As == AJOS || jmp.As == AJMI ||
  1963  		jmp.As == AJPL || jmp.As == AJPS || jmp.As == AJPC {
  1964  		return false, 0
  1965  	}
  1966  
  1967  	if cmpAddSub {
  1968  		return true, fusedSize
  1969  	}
  1970  
  1971  	if jmp.As == AJCS || jmp.As == AJCC || jmp.As == AJHI || jmp.As == AJLS {
  1972  		return false, 0
  1973  	}
  1974  
  1975  	return true, fusedSize
  1976  }
  1977  
  1978  type padJumpsCtx int32
  1979  
  1980  func makePjcCtx(ctxt *obj.Link) padJumpsCtx {
  1981  	// Disable jump padding on 32 bit builds by setting
  1982  	// padJumps to 0.
  1983  	if ctxt.Arch.Family == sys.I386 {
  1984  		return padJumpsCtx(0)
  1985  	}
  1986  
  1987  	// Disable jump padding for hand written assembly code.
  1988  	if ctxt.IsAsm {
  1989  		return padJumpsCtx(0)
  1990  	}
  1991  
  1992  	return padJumpsCtx(32)
  1993  }
  1994  
  1995  // padJump detects whether the instruction being assembled is a standalone or a macro-fused
  1996  // jump that needs to be padded. If it is, NOPs are inserted to ensure that the jump does
  1997  // not cross or end on a 32 byte boundary.
  1998  func (pjc padJumpsCtx) padJump(ctxt *obj.Link, s *obj.LSym, p *obj.Prog, c int32) int32 {
  1999  	if pjc == 0 {
  2000  		return c
  2001  	}
  2002  
  2003  	var toPad int32
  2004  	fj, fjSize := fusedJump(p)
  2005  	mask := int32(pjc - 1)
  2006  	if fj {
  2007  		if (c&mask)+int32(fjSize) >= int32(pjc) {
  2008  			toPad = int32(pjc) - (c & mask)
  2009  		}
  2010  	} else if isJump(p) {
  2011  		if (c&mask)+int32(p.Isize) >= int32(pjc) {
  2012  			toPad = int32(pjc) - (c & mask)
  2013  		}
  2014  	}
  2015  	if toPad <= 0 {
  2016  		return c
  2017  	}
  2018  
  2019  	return noppad(ctxt, s, c, toPad)
  2020  }
  2021  
  2022  // reAssemble is called if an instruction's size changes during assembly. If
  2023  // it does and the instruction is a standalone or a macro-fused jump we need to
  2024  // reassemble.
  2025  func (pjc padJumpsCtx) reAssemble(p *obj.Prog) bool {
  2026  	if pjc == 0 {
  2027  		return false
  2028  	}
  2029  
  2030  	fj, _ := fusedJump(p)
  2031  	return fj || isJump(p)
  2032  }
  2033  
  2034  type nopPad struct {
  2035  	p *obj.Prog // Instruction before the pad
  2036  	n int32     // Size of the pad
  2037  }
  2038  
  2039  func span6(ctxt *obj.Link, s *obj.LSym, newprog obj.ProgAlloc) {
  2040  	if ctxt.Retpoline && ctxt.Arch.Family == sys.I386 {
  2041  		ctxt.Diag("-spectre=ret not supported on 386")
  2042  		ctxt.Retpoline = false // don't keep printing
  2043  	}
  2044  
  2045  	pjc := makePjcCtx(ctxt)
  2046  
  2047  	if s.P != nil {
  2048  		return
  2049  	}
  2050  
  2051  	if ycover[0] == 0 {
  2052  		ctxt.Diag("x86 tables not initialized, call x86.instinit first")
  2053  	}
  2054  
  2055  	for p := s.Func().Text; p != nil; p = p.Link {
  2056  		if p.To.Type == obj.TYPE_BRANCH && p.To.Target() == nil {
  2057  			p.To.SetTarget(p)
  2058  		}
  2059  		if p.As == AADJSP {
  2060  			p.To.Type = obj.TYPE_REG
  2061  			p.To.Reg = REG_SP
  2062  			// Generate 'ADDQ $x, SP' or 'SUBQ $x, SP', with x positive.
  2063  			// One exception: It is smaller to encode $-0x80 than $0x80.
  2064  			// For that case, flip the sign and the op:
  2065  			// Instead of 'ADDQ $0x80, SP', generate 'SUBQ $-0x80, SP'.
  2066  			switch v := p.From.Offset; {
  2067  			case v == 0:
  2068  				p.As = obj.ANOP
  2069  			case v == 0x80 || (v < 0 && v != -0x80):
  2070  				p.As = spadjop(ctxt, AADDL, AADDQ)
  2071  				p.From.Offset *= -1
  2072  			default:
  2073  				p.As = spadjop(ctxt, ASUBL, ASUBQ)
  2074  			}
  2075  		}
  2076  		if ctxt.Retpoline && (p.As == obj.ACALL || p.As == obj.AJMP) && (p.To.Type == obj.TYPE_REG || p.To.Type == obj.TYPE_MEM) {
  2077  			if p.To.Type != obj.TYPE_REG {
  2078  				ctxt.Diag("non-retpoline-compatible: %v", p)
  2079  				continue
  2080  			}
  2081  			p.To.Type = obj.TYPE_BRANCH
  2082  			p.To.Name = obj.NAME_EXTERN
  2083  			p.To.Sym = ctxt.Lookup("runtime.retpoline" + obj.Rconv(int(p.To.Reg)))
  2084  			p.To.Reg = 0
  2085  			p.To.Offset = 0
  2086  		}
  2087  	}
  2088  
  2089  	var count int64 // rough count of number of instructions
  2090  	for p := s.Func().Text; p != nil; p = p.Link {
  2091  		count++
  2092  		p.Back = branchShort // use short branches first time through
  2093  		if q := p.To.Target(); q != nil && (q.Back&branchShort != 0) {
  2094  			p.Back |= branchBackwards
  2095  			q.Back |= branchLoopHead
  2096  		}
  2097  	}
  2098  	s.GrowCap(count * 5) // preallocate roughly 5 bytes per instruction
  2099  
  2100  	var ab AsmBuf
  2101  	var n int
  2102  	var c int32
  2103  	errors := ctxt.Errors
  2104  	var nops []nopPad // Padding for a particular assembly (reuse slice storage if multiple assemblies)
  2105  	nrelocs0 := len(s.R)
  2106  	for {
  2107  		// This loop continues while there are reasons to re-assemble
  2108  		// whole block, like the presence of long forward jumps.
  2109  		reAssemble := false
  2110  		for i := range s.R[nrelocs0:] {
  2111  			s.R[nrelocs0+i] = obj.Reloc{}
  2112  		}
  2113  		s.R = s.R[:nrelocs0] // preserve marker relocations generated by the compiler
  2114  		s.P = s.P[:0]
  2115  		c = 0
  2116  		var pPrev *obj.Prog
  2117  		nops = nops[:0]
  2118  		for p := s.Func().Text; p != nil; p = p.Link {
  2119  			c0 := c
  2120  			c = pjc.padJump(ctxt, s, p, c)
  2121  
  2122  			if maxLoopPad > 0 && p.Back&branchLoopHead != 0 && c&(loopAlign-1) != 0 {
  2123  				// pad with NOPs
  2124  				v := -c & (loopAlign - 1)
  2125  
  2126  				if v <= maxLoopPad {
  2127  					s.Grow(int64(c) + int64(v))
  2128  					fillnop(s.P[c:], int(v))
  2129  					c += v
  2130  				}
  2131  			}
  2132  
  2133  			p.Pc = int64(c)
  2134  
  2135  			// process forward jumps to p
  2136  			for q := p.Rel; q != nil; q = q.Forwd {
  2137  				v := int32(p.Pc - (q.Pc + int64(q.Isize)))
  2138  				if q.Back&branchShort != 0 {
  2139  					if v > 127 {
  2140  						reAssemble = true
  2141  						q.Back ^= branchShort
  2142  					}
  2143  
  2144  					if q.As == AJCXZL || q.As == AXBEGIN {
  2145  						s.P[q.Pc+2] = byte(v)
  2146  					} else {
  2147  						s.P[q.Pc+1] = byte(v)
  2148  					}
  2149  				} else {
  2150  					binary.LittleEndian.PutUint32(s.P[q.Pc+int64(q.Isize)-4:], uint32(v))
  2151  				}
  2152  			}
  2153  
  2154  			if p.As == obj.APCALIGN || p.As == obj.APCALIGNMAX {
  2155  				v := obj.AlignmentPadding(c, p, ctxt, s)
  2156  				if v > 0 {
  2157  					s.Grow(int64(c) + int64(v))
  2158  					fillnop(s.P[c:], v)
  2159  				}
  2160  				p.Pc = int64(c)
  2161  				c += int32(v)
  2162  				pPrev = p
  2163  				continue
  2164  			}
  2165  
  2166  			p.Rel = nil
  2167  
  2168  			p.Pc = int64(c)
  2169  			ab.asmins(ctxt, s, p)
  2170  			m := ab.Len()
  2171  			if int(p.Isize) != m {
  2172  				p.Isize = uint8(m)
  2173  				if pjc.reAssemble(p) {
  2174  					// We need to re-assemble here to check for jumps and fused jumps
  2175  					// that span or end on 32 byte boundaries.
  2176  					reAssemble = true
  2177  				}
  2178  			}
  2179  
  2180  			s.Grow(p.Pc + int64(m))
  2181  			copy(s.P[p.Pc:], ab.Bytes())
  2182  			// If there was padding, remember it.
  2183  			if pPrev != nil && !ctxt.IsAsm && c > c0 {
  2184  				nops = append(nops, nopPad{p: pPrev, n: c - c0})
  2185  			}
  2186  			c += int32(m)
  2187  			pPrev = p
  2188  		}
  2189  
  2190  		n++
  2191  		if n > 1000 {
  2192  			ctxt.Diag("span must be looping")
  2193  			log.Fatalf("loop")
  2194  		}
  2195  		if !reAssemble {
  2196  			break
  2197  		}
  2198  		if ctxt.Errors > errors {
  2199  			return
  2200  		}
  2201  	}
  2202  	// splice padding nops into Progs
  2203  	for _, n := range nops {
  2204  		pp := n.p
  2205  		np := &obj.Prog{Link: pp.Link, Ctxt: pp.Ctxt, As: obj.ANOP, Pos: pp.Pos.WithNotStmt(), Pc: pp.Pc + int64(pp.Isize), Isize: uint8(n.n)}
  2206  		pp.Link = np
  2207  	}
  2208  
  2209  	s.Size = int64(c)
  2210  
  2211  	if false { /* debug['a'] > 1 */
  2212  		fmt.Printf("span1 %s %d (%d tries)\n %.6x", s.Name, s.Size, n, 0)
  2213  		var i int
  2214  		for i = 0; i < len(s.P); i++ {
  2215  			fmt.Printf(" %.2x", s.P[i])
  2216  			if i%16 == 15 {
  2217  				fmt.Printf("\n  %.6x", uint(i+1))
  2218  			}
  2219  		}
  2220  
  2221  		if i%16 != 0 {
  2222  			fmt.Printf("\n")
  2223  		}
  2224  
  2225  		for i := 0; i < len(s.R); i++ {
  2226  			r := &s.R[i]
  2227  			fmt.Printf(" rel %#.4x/%d %s%+d\n", uint32(r.Off), r.Siz, r.Sym.Name, r.Add)
  2228  		}
  2229  	}
  2230  
  2231  	// Mark nonpreemptible instruction sequences.
  2232  	// The 2-instruction TLS access sequence
  2233  	//	MOVQ TLS, BX
  2234  	//	MOVQ 0(BX)(TLS*1), BX
  2235  	// is not async preemptible, as if it is preempted and resumed on
  2236  	// a different thread, the TLS address may become invalid.
  2237  	if !CanUse1InsnTLS(ctxt) {
  2238  		useTLS := func(p *obj.Prog) bool {
  2239  			// Only need to mark the second instruction, which has
  2240  			// REG_TLS as Index. (It is okay to interrupt and restart
  2241  			// the first instruction.)
  2242  			return p.From.Index == REG_TLS
  2243  		}
  2244  		obj.MarkUnsafePoints(ctxt, s.Func().Text, newprog, useTLS, nil)
  2245  	}
  2246  
  2247  	// Now that we know byte offsets, we can generate jump table entries.
  2248  	// TODO: could this live in obj instead of obj/$ARCH?
  2249  	for _, jt := range s.Func().JumpTables {
  2250  		for i, p := range jt.Targets {
  2251  			// The ith jumptable entry points to the p.Pc'th
  2252  			// byte in the function symbol s.
  2253  			jt.Sym.WriteAddr(ctxt, int64(i)*8, 8, s, p.Pc)
  2254  		}
  2255  	}
  2256  }
  2257  
  2258  func instinit(ctxt *obj.Link) {
  2259  	if ycover[0] != 0 {
  2260  		// Already initialized; stop now.
  2261  		// This happens in the cmd/asm tests,
  2262  		// each of which re-initializes the arch.
  2263  		return
  2264  	}
  2265  
  2266  	switch ctxt.Headtype {
  2267  	case objabi.Hplan9:
  2268  		// _privates is a special symbol on Plan 9 that
  2269  		// points to per–process private data (like TLS area).
  2270  		// See https://9p.io/magic/man2html/2/exec .
  2271  		// The assembler inserts a reference to this symbol
  2272  		// for accessing the G. Mark it as linkname so it is
  2273  		// allowed to access from anywhere. (Would be nice to
  2274  		// mark it external, but we don't have a mechanism for
  2275  		// that.)
  2276  		plan9privates = ctxt.Lookup("_privates")
  2277  		plan9privates.Set(obj.AttrLinkname, true)
  2278  	}
  2279  
  2280  	for i := range avxOptab {
  2281  		c := avxOptab[i].as
  2282  		if opindex[c&obj.AMask] != nil {
  2283  			ctxt.Diag("phase error in avxOptab: %d (%v)", i, c)
  2284  		}
  2285  		opindex[c&obj.AMask] = &avxOptab[i]
  2286  	}
  2287  	for i := 1; optab[i].as != 0; i++ {
  2288  		c := optab[i].as
  2289  		if opindex[c&obj.AMask] != nil {
  2290  			ctxt.Diag("phase error in optab: %d (%v)", i, c)
  2291  		}
  2292  		opindex[c&obj.AMask] = &optab[i]
  2293  	}
  2294  
  2295  	for i := 0; i < Ymax; i++ {
  2296  		ycover[i*Ymax+i] = 1
  2297  	}
  2298  
  2299  	ycover[Yi0*Ymax+Yu2] = 1
  2300  	ycover[Yi1*Ymax+Yu2] = 1
  2301  
  2302  	ycover[Yi0*Ymax+Yi8] = 1
  2303  	ycover[Yi1*Ymax+Yi8] = 1
  2304  	ycover[Yu2*Ymax+Yi8] = 1
  2305  	ycover[Yu7*Ymax+Yi8] = 1
  2306  
  2307  	ycover[Yi0*Ymax+Yu7] = 1
  2308  	ycover[Yi1*Ymax+Yu7] = 1
  2309  	ycover[Yu2*Ymax+Yu7] = 1
  2310  
  2311  	ycover[Yi0*Ymax+Yu8] = 1
  2312  	ycover[Yi1*Ymax+Yu8] = 1
  2313  	ycover[Yu2*Ymax+Yu8] = 1
  2314  	ycover[Yu7*Ymax+Yu8] = 1
  2315  
  2316  	ycover[Yi0*Ymax+Ys32] = 1
  2317  	ycover[Yi1*Ymax+Ys32] = 1
  2318  	ycover[Yu2*Ymax+Ys32] = 1
  2319  	ycover[Yu7*Ymax+Ys32] = 1
  2320  	ycover[Yu8*Ymax+Ys32] = 1
  2321  	ycover[Yi8*Ymax+Ys32] = 1
  2322  
  2323  	ycover[Yi0*Ymax+Yi32] = 1
  2324  	ycover[Yi1*Ymax+Yi32] = 1
  2325  	ycover[Yu2*Ymax+Yi32] = 1
  2326  	ycover[Yu7*Ymax+Yi32] = 1
  2327  	ycover[Yu8*Ymax+Yi32] = 1
  2328  	ycover[Yi8*Ymax+Yi32] = 1
  2329  	ycover[Ys32*Ymax+Yi32] = 1
  2330  
  2331  	ycover[Yi0*Ymax+Yi64] = 1
  2332  	ycover[Yi1*Ymax+Yi64] = 1
  2333  	ycover[Yu7*Ymax+Yi64] = 1
  2334  	ycover[Yu2*Ymax+Yi64] = 1
  2335  	ycover[Yu8*Ymax+Yi64] = 1
  2336  	ycover[Yi8*Ymax+Yi64] = 1
  2337  	ycover[Ys32*Ymax+Yi64] = 1
  2338  	ycover[Yi32*Ymax+Yi64] = 1
  2339  
  2340  	ycover[Yal*Ymax+Yrb] = 1
  2341  	ycover[Ycl*Ymax+Yrb] = 1
  2342  	ycover[Yax*Ymax+Yrb] = 1
  2343  	ycover[Ycx*Ymax+Yrb] = 1
  2344  	ycover[Yrx*Ymax+Yrb] = 1
  2345  	ycover[Yrl*Ymax+Yrb] = 1 // but not Yrl32
  2346  
  2347  	ycover[Ycl*Ymax+Ycx] = 1
  2348  
  2349  	ycover[Yax*Ymax+Yrx] = 1
  2350  	ycover[Ycx*Ymax+Yrx] = 1
  2351  
  2352  	ycover[Yax*Ymax+Yrl] = 1
  2353  	ycover[Ycx*Ymax+Yrl] = 1
  2354  	ycover[Yrx*Ymax+Yrl] = 1
  2355  	ycover[Yrl32*Ymax+Yrl] = 1
  2356  
  2357  	ycover[Yf0*Ymax+Yrf] = 1
  2358  
  2359  	ycover[Yal*Ymax+Ymb] = 1
  2360  	ycover[Ycl*Ymax+Ymb] = 1
  2361  	ycover[Yax*Ymax+Ymb] = 1
  2362  	ycover[Ycx*Ymax+Ymb] = 1
  2363  	ycover[Yrx*Ymax+Ymb] = 1
  2364  	ycover[Yrb*Ymax+Ymb] = 1
  2365  	ycover[Yrl*Ymax+Ymb] = 1 // but not Yrl32
  2366  	ycover[Ym*Ymax+Ymb] = 1
  2367  
  2368  	ycover[Yax*Ymax+Yml] = 1
  2369  	ycover[Ycx*Ymax+Yml] = 1
  2370  	ycover[Yrx*Ymax+Yml] = 1
  2371  	ycover[Yrl*Ymax+Yml] = 1
  2372  	ycover[Yrl32*Ymax+Yml] = 1
  2373  	ycover[Ym*Ymax+Yml] = 1
  2374  
  2375  	ycover[Yax*Ymax+Ymm] = 1
  2376  	ycover[Ycx*Ymax+Ymm] = 1
  2377  	ycover[Yrx*Ymax+Ymm] = 1
  2378  	ycover[Yrl*Ymax+Ymm] = 1
  2379  	ycover[Yrl32*Ymax+Ymm] = 1
  2380  	ycover[Ym*Ymax+Ymm] = 1
  2381  	ycover[Ymr*Ymax+Ymm] = 1
  2382  
  2383  	ycover[Yxr0*Ymax+Yxr] = 1
  2384  
  2385  	ycover[Ym*Ymax+Yxm] = 1
  2386  	ycover[Yxr0*Ymax+Yxm] = 1
  2387  	ycover[Yxr*Ymax+Yxm] = 1
  2388  
  2389  	ycover[Ym*Ymax+Yym] = 1
  2390  	ycover[Yyr*Ymax+Yym] = 1
  2391  
  2392  	ycover[Yxr0*Ymax+YxrEvex] = 1
  2393  	ycover[Yxr*Ymax+YxrEvex] = 1
  2394  
  2395  	ycover[Ym*Ymax+YxmEvex] = 1
  2396  	ycover[Yxr0*Ymax+YxmEvex] = 1
  2397  	ycover[Yxr*Ymax+YxmEvex] = 1
  2398  	ycover[YxrEvex*Ymax+YxmEvex] = 1
  2399  
  2400  	ycover[Yyr*Ymax+YyrEvex] = 1
  2401  
  2402  	ycover[Ym*Ymax+YymEvex] = 1
  2403  	ycover[Yyr*Ymax+YymEvex] = 1
  2404  	ycover[YyrEvex*Ymax+YymEvex] = 1
  2405  
  2406  	ycover[Ym*Ymax+Yzm] = 1
  2407  	ycover[Yzr*Ymax+Yzm] = 1
  2408  
  2409  	ycover[Yk0*Ymax+Yk] = 1
  2410  	ycover[Yknot0*Ymax+Yk] = 1
  2411  
  2412  	ycover[Yk0*Ymax+Ykm] = 1
  2413  	ycover[Yknot0*Ymax+Ykm] = 1
  2414  	ycover[Yk*Ymax+Ykm] = 1
  2415  	ycover[Ym*Ymax+Ykm] = 1
  2416  
  2417  	ycover[Yxvm*Ymax+YxvmEvex] = 1
  2418  
  2419  	ycover[Yyvm*Ymax+YyvmEvex] = 1
  2420  
  2421  	for i := 0; i < MAXREG; i++ {
  2422  		reg[i] = -1
  2423  		if i >= REG_AL && i <= REG_R15B {
  2424  			reg[i] = (i - REG_AL) & 7
  2425  			if i >= REG_SPB && i <= REG_DIB {
  2426  				regrex[i] = 0x40
  2427  			}
  2428  			if i >= REG_R8B && i <= REG_R15B {
  2429  				regrex[i] = Rxr | Rxx | Rxb
  2430  			}
  2431  		}
  2432  
  2433  		if i >= REG_AH && i <= REG_BH {
  2434  			reg[i] = 4 + ((i - REG_AH) & 7)
  2435  		}
  2436  		if i >= REG_AX && i <= REG_R15 {
  2437  			reg[i] = (i - REG_AX) & 7
  2438  			if i >= REG_R8 {
  2439  				regrex[i] = Rxr | Rxx | Rxb
  2440  			}
  2441  		}
  2442  
  2443  		if i >= REG_F0 && i <= REG_F0+7 {
  2444  			reg[i] = (i - REG_F0) & 7
  2445  		}
  2446  		if i >= REG_M0 && i <= REG_M0+7 {
  2447  			reg[i] = (i - REG_M0) & 7
  2448  		}
  2449  		if i >= REG_K0 && i <= REG_K0+7 {
  2450  			reg[i] = (i - REG_K0) & 7
  2451  		}
  2452  		if i >= REG_X0 && i <= REG_X0+15 {
  2453  			reg[i] = (i - REG_X0) & 7
  2454  			if i >= REG_X0+8 {
  2455  				regrex[i] = Rxr | Rxx | Rxb
  2456  			}
  2457  		}
  2458  		if i >= REG_X16 && i <= REG_X16+15 {
  2459  			reg[i] = (i - REG_X16) & 7
  2460  			if i >= REG_X16+8 {
  2461  				regrex[i] = Rxr | Rxx | Rxb | RxrEvex
  2462  			} else {
  2463  				regrex[i] = RxrEvex
  2464  			}
  2465  		}
  2466  		if i >= REG_Y0 && i <= REG_Y0+15 {
  2467  			reg[i] = (i - REG_Y0) & 7
  2468  			if i >= REG_Y0+8 {
  2469  				regrex[i] = Rxr | Rxx | Rxb
  2470  			}
  2471  		}
  2472  		if i >= REG_Y16 && i <= REG_Y16+15 {
  2473  			reg[i] = (i - REG_Y16) & 7
  2474  			if i >= REG_Y16+8 {
  2475  				regrex[i] = Rxr | Rxx | Rxb | RxrEvex
  2476  			} else {
  2477  				regrex[i] = RxrEvex
  2478  			}
  2479  		}
  2480  		if i >= REG_Z0 && i <= REG_Z0+15 {
  2481  			reg[i] = (i - REG_Z0) & 7
  2482  			if i > REG_Z0+7 {
  2483  				regrex[i] = Rxr | Rxx | Rxb
  2484  			}
  2485  		}
  2486  		if i >= REG_Z16 && i <= REG_Z16+15 {
  2487  			reg[i] = (i - REG_Z16) & 7
  2488  			if i >= REG_Z16+8 {
  2489  				regrex[i] = Rxr | Rxx | Rxb | RxrEvex
  2490  			} else {
  2491  				regrex[i] = RxrEvex
  2492  			}
  2493  		}
  2494  
  2495  		if i >= REG_CR+8 && i <= REG_CR+15 {
  2496  			regrex[i] = Rxr
  2497  		}
  2498  	}
  2499  }
  2500  
  2501  var isAndroid = buildcfg.GOOS == "android"
  2502  
  2503  func prefixof(ctxt *obj.Link, a *obj.Addr) int {
  2504  	if a.Reg < REG_CS && a.Index < REG_CS { // fast path
  2505  		return 0
  2506  	}
  2507  	if a.Type == obj.TYPE_MEM && a.Name == obj.NAME_NONE {
  2508  		switch a.Reg {
  2509  		case REG_CS:
  2510  			return 0x2e
  2511  
  2512  		case REG_DS:
  2513  			return 0x3e
  2514  
  2515  		case REG_ES:
  2516  			return 0x26
  2517  
  2518  		case REG_FS:
  2519  			return 0x64
  2520  
  2521  		case REG_GS:
  2522  			return 0x65
  2523  
  2524  		case REG_TLS:
  2525  			// NOTE: Systems listed here should be only systems that
  2526  			// support direct TLS references like 8(TLS) implemented as
  2527  			// direct references from FS or GS. Systems that require
  2528  			// the initial-exec model, where you load the TLS base into
  2529  			// a register and then index from that register, do not reach
  2530  			// this code and should not be listed.
  2531  			if ctxt.Arch.Family == sys.I386 {
  2532  				switch ctxt.Headtype {
  2533  				default:
  2534  					if isAndroid {
  2535  						return 0x65 // GS
  2536  					}
  2537  					log.Fatalf("unknown TLS base register for %v", ctxt.Headtype)
  2538  
  2539  				case objabi.Hdarwin,
  2540  					objabi.Hdragonfly,
  2541  					objabi.Hfreebsd,
  2542  					objabi.Hnetbsd,
  2543  					objabi.Hopenbsd:
  2544  					return 0x65 // GS
  2545  				}
  2546  			}
  2547  
  2548  			switch ctxt.Headtype {
  2549  			default:
  2550  				log.Fatalf("unknown TLS base register for %v", ctxt.Headtype)
  2551  
  2552  			case objabi.Hlinux:
  2553  				if isAndroid {
  2554  					return 0x64 // FS
  2555  				}
  2556  
  2557  				if ctxt.Flag_shared {
  2558  					log.Fatalf("unknown TLS base register for linux with -shared")
  2559  				} else {
  2560  					return 0x64 // FS
  2561  				}
  2562  
  2563  			case objabi.Hdragonfly,
  2564  				objabi.Hfreebsd,
  2565  				objabi.Hnetbsd,
  2566  				objabi.Hopenbsd,
  2567  				objabi.Hsolaris:
  2568  				return 0x64 // FS
  2569  
  2570  			case objabi.Hdarwin:
  2571  				return 0x65 // GS
  2572  			}
  2573  		}
  2574  	}
  2575  
  2576  	switch a.Index {
  2577  	case REG_CS:
  2578  		return 0x2e
  2579  
  2580  	case REG_DS:
  2581  		return 0x3e
  2582  
  2583  	case REG_ES:
  2584  		return 0x26
  2585  
  2586  	case REG_TLS:
  2587  		if ctxt.Flag_shared && ctxt.Headtype != objabi.Hwindows {
  2588  			// When building for inclusion into a shared library, an instruction of the form
  2589  			//     MOV off(CX)(TLS*1), AX
  2590  			// becomes
  2591  			//     mov %gs:off(%ecx), %eax // on i386
  2592  			//     mov %fs:off(%rcx), %rax // on amd64
  2593  			// which assumes that the correct TLS offset has been loaded into CX (today
  2594  			// there is only one TLS variable -- g -- so this is OK). When not building for
  2595  			// a shared library the instruction it becomes
  2596  			//     mov 0x0(%ecx), %eax // on i386
  2597  			//     mov 0x0(%rcx), %rax // on amd64
  2598  			// and a R_TLS_LE relocation, and so does not require a prefix.
  2599  			if ctxt.Arch.Family == sys.I386 {
  2600  				return 0x65 // GS
  2601  			}
  2602  			return 0x64 // FS
  2603  		}
  2604  
  2605  	case REG_FS:
  2606  		return 0x64
  2607  
  2608  	case REG_GS:
  2609  		return 0x65
  2610  	}
  2611  
  2612  	return 0
  2613  }
  2614  
  2615  // oclassRegList returns multisource operand class for addr.
  2616  func oclassRegList(ctxt *obj.Link, addr *obj.Addr) int {
  2617  	// TODO(quasilyte): when oclass register case is refactored into
  2618  	// lookup table, use it here to get register kind more easily.
  2619  	// Helper functions like regIsXmm should go away too (they will become redundant).
  2620  
  2621  	regIsXmm := func(r int) bool { return r >= REG_X0 && r <= REG_X31 }
  2622  	regIsYmm := func(r int) bool { return r >= REG_Y0 && r <= REG_Y31 }
  2623  	regIsZmm := func(r int) bool { return r >= REG_Z0 && r <= REG_Z31 }
  2624  
  2625  	reg0, reg1 := decodeRegisterRange(addr.Offset)
  2626  	low := regIndex(int16(reg0))
  2627  	high := regIndex(int16(reg1))
  2628  
  2629  	if ctxt.Arch.Family == sys.I386 {
  2630  		if low >= 8 || high >= 8 {
  2631  			return Yxxx
  2632  		}
  2633  	}
  2634  
  2635  	switch high - low {
  2636  	case 3:
  2637  		switch {
  2638  		case regIsXmm(reg0) && regIsXmm(reg1):
  2639  			return YxrEvexMulti4
  2640  		case regIsYmm(reg0) && regIsYmm(reg1):
  2641  			return YyrEvexMulti4
  2642  		case regIsZmm(reg0) && regIsZmm(reg1):
  2643  			return YzrMulti4
  2644  		default:
  2645  			return Yxxx
  2646  		}
  2647  	default:
  2648  		return Yxxx
  2649  	}
  2650  }
  2651  
  2652  // oclassVMem returns V-mem (vector memory with VSIB) operand class.
  2653  // For addr that is not V-mem returns (Yxxx, false).
  2654  func oclassVMem(ctxt *obj.Link, addr *obj.Addr) (int, bool) {
  2655  	switch addr.Index {
  2656  	case REG_X0 + 0,
  2657  		REG_X0 + 1,
  2658  		REG_X0 + 2,
  2659  		REG_X0 + 3,
  2660  		REG_X0 + 4,
  2661  		REG_X0 + 5,
  2662  		REG_X0 + 6,
  2663  		REG_X0 + 7:
  2664  		return Yxvm, true
  2665  	case REG_X8 + 0,
  2666  		REG_X8 + 1,
  2667  		REG_X8 + 2,
  2668  		REG_X8 + 3,
  2669  		REG_X8 + 4,
  2670  		REG_X8 + 5,
  2671  		REG_X8 + 6,
  2672  		REG_X8 + 7:
  2673  		if ctxt.Arch.Family == sys.I386 {
  2674  			return Yxxx, true
  2675  		}
  2676  		return Yxvm, true
  2677  	case REG_X16 + 0,
  2678  		REG_X16 + 1,
  2679  		REG_X16 + 2,
  2680  		REG_X16 + 3,
  2681  		REG_X16 + 4,
  2682  		REG_X16 + 5,
  2683  		REG_X16 + 6,
  2684  		REG_X16 + 7,
  2685  		REG_X16 + 8,
  2686  		REG_X16 + 9,
  2687  		REG_X16 + 10,
  2688  		REG_X16 + 11,
  2689  		REG_X16 + 12,
  2690  		REG_X16 + 13,
  2691  		REG_X16 + 14,
  2692  		REG_X16 + 15:
  2693  		if ctxt.Arch.Family == sys.I386 {
  2694  			return Yxxx, true
  2695  		}
  2696  		return YxvmEvex, true
  2697  
  2698  	case REG_Y0 + 0,
  2699  		REG_Y0 + 1,
  2700  		REG_Y0 + 2,
  2701  		REG_Y0 + 3,
  2702  		REG_Y0 + 4,
  2703  		REG_Y0 + 5,
  2704  		REG_Y0 + 6,
  2705  		REG_Y0 + 7:
  2706  		return Yyvm, true
  2707  	case REG_Y8 + 0,
  2708  		REG_Y8 + 1,
  2709  		REG_Y8 + 2,
  2710  		REG_Y8 + 3,
  2711  		REG_Y8 + 4,
  2712  		REG_Y8 + 5,
  2713  		REG_Y8 + 6,
  2714  		REG_Y8 + 7:
  2715  		if ctxt.Arch.Family == sys.I386 {
  2716  			return Yxxx, true
  2717  		}
  2718  		return Yyvm, true
  2719  	case REG_Y16 + 0,
  2720  		REG_Y16 + 1,
  2721  		REG_Y16 + 2,
  2722  		REG_Y16 + 3,
  2723  		REG_Y16 + 4,
  2724  		REG_Y16 + 5,
  2725  		REG_Y16 + 6,
  2726  		REG_Y16 + 7,
  2727  		REG_Y16 + 8,
  2728  		REG_Y16 + 9,
  2729  		REG_Y16 + 10,
  2730  		REG_Y16 + 11,
  2731  		REG_Y16 + 12,
  2732  		REG_Y16 + 13,
  2733  		REG_Y16 + 14,
  2734  		REG_Y16 + 15:
  2735  		if ctxt.Arch.Family == sys.I386 {
  2736  			return Yxxx, true
  2737  		}
  2738  		return YyvmEvex, true
  2739  
  2740  	case REG_Z0 + 0,
  2741  		REG_Z0 + 1,
  2742  		REG_Z0 + 2,
  2743  		REG_Z0 + 3,
  2744  		REG_Z0 + 4,
  2745  		REG_Z0 + 5,
  2746  		REG_Z0 + 6,
  2747  		REG_Z0 + 7:
  2748  		return Yzvm, true
  2749  	case REG_Z8 + 0,
  2750  		REG_Z8 + 1,
  2751  		REG_Z8 + 2,
  2752  		REG_Z8 + 3,
  2753  		REG_Z8 + 4,
  2754  		REG_Z8 + 5,
  2755  		REG_Z8 + 6,
  2756  		REG_Z8 + 7,
  2757  		REG_Z8 + 8,
  2758  		REG_Z8 + 9,
  2759  		REG_Z8 + 10,
  2760  		REG_Z8 + 11,
  2761  		REG_Z8 + 12,
  2762  		REG_Z8 + 13,
  2763  		REG_Z8 + 14,
  2764  		REG_Z8 + 15,
  2765  		REG_Z8 + 16,
  2766  		REG_Z8 + 17,
  2767  		REG_Z8 + 18,
  2768  		REG_Z8 + 19,
  2769  		REG_Z8 + 20,
  2770  		REG_Z8 + 21,
  2771  		REG_Z8 + 22,
  2772  		REG_Z8 + 23:
  2773  		if ctxt.Arch.Family == sys.I386 {
  2774  			return Yxxx, true
  2775  		}
  2776  		return Yzvm, true
  2777  	}
  2778  
  2779  	return Yxxx, false
  2780  }
  2781  
  2782  func oclass(ctxt *obj.Link, p *obj.Prog, a *obj.Addr) int {
  2783  	switch a.Type {
  2784  	case obj.TYPE_REGLIST:
  2785  		return oclassRegList(ctxt, a)
  2786  
  2787  	case obj.TYPE_NONE:
  2788  		return Ynone
  2789  
  2790  	case obj.TYPE_BRANCH:
  2791  		return Ybr
  2792  
  2793  	case obj.TYPE_INDIR:
  2794  		if a.Name != obj.NAME_NONE && a.Reg == REG_NONE && a.Index == REG_NONE && a.Scale == 0 {
  2795  			return Yindir
  2796  		}
  2797  		return Yxxx
  2798  
  2799  	case obj.TYPE_MEM:
  2800  		// Pseudo registers have negative index, but SP is
  2801  		// not pseudo on x86, hence REG_SP check is not redundant.
  2802  		if a.Index == REG_SP || a.Index < 0 {
  2803  			// Can't use FP/SB/PC/SP as the index register.
  2804  			return Yxxx
  2805  		}
  2806  
  2807  		if vmem, ok := oclassVMem(ctxt, a); ok {
  2808  			return vmem
  2809  		}
  2810  
  2811  		if ctxt.Arch.Family == sys.AMD64 {
  2812  			switch a.Name {
  2813  			case obj.NAME_EXTERN, obj.NAME_STATIC, obj.NAME_GOTREF:
  2814  				// Global variables can't use index registers and their
  2815  				// base register is %rip (%rip is encoded as REG_NONE).
  2816  				if a.Reg != REG_NONE || a.Index != REG_NONE || a.Scale != 0 {
  2817  					return Yxxx
  2818  				}
  2819  			case obj.NAME_AUTO, obj.NAME_PARAM:
  2820  				// These names must have a base of SP.  The old compiler
  2821  				// uses 0 for the base register. SSA uses REG_SP.
  2822  				if a.Reg != REG_SP && a.Reg != 0 {
  2823  					return Yxxx
  2824  				}
  2825  			case obj.NAME_NONE:
  2826  				// everything is ok
  2827  			default:
  2828  				// unknown name
  2829  				return Yxxx
  2830  			}
  2831  		}
  2832  		return Ym
  2833  
  2834  	case obj.TYPE_ADDR:
  2835  		switch a.Name {
  2836  		case obj.NAME_GOTREF:
  2837  			ctxt.Diag("unexpected TYPE_ADDR with NAME_GOTREF")
  2838  			return Yxxx
  2839  
  2840  		case obj.NAME_EXTERN,
  2841  			obj.NAME_STATIC:
  2842  			if a.Sym != nil && useAbs(ctxt, a.Sym) {
  2843  				return Yi32
  2844  			}
  2845  			return Yiauto // use pc-relative addressing
  2846  
  2847  		case obj.NAME_AUTO,
  2848  			obj.NAME_PARAM:
  2849  			return Yiauto
  2850  		}
  2851  
  2852  		// TODO(rsc): DUFFZERO/DUFFCOPY encoding forgot to set a->index
  2853  		// and got Yi32 in an earlier version of this code.
  2854  		// Keep doing that until we fix yduff etc.
  2855  		if a.Sym != nil && strings.HasPrefix(a.Sym.Name, "runtime.duff") {
  2856  			return Yi32
  2857  		}
  2858  
  2859  		if a.Sym != nil || a.Name != obj.NAME_NONE {
  2860  			ctxt.Diag("unexpected addr: %v", obj.Dconv(p, a))
  2861  		}
  2862  		fallthrough
  2863  
  2864  	case obj.TYPE_CONST:
  2865  		if a.Sym != nil {
  2866  			ctxt.Diag("TYPE_CONST with symbol: %v", obj.Dconv(p, a))
  2867  		}
  2868  
  2869  		v := a.Offset
  2870  		if ctxt.Arch.Family == sys.I386 {
  2871  			v = int64(int32(v))
  2872  		}
  2873  		switch {
  2874  		case v == 0:
  2875  			return Yi0
  2876  		case v == 1:
  2877  			return Yi1
  2878  		case v >= 0 && v <= 3:
  2879  			return Yu2
  2880  		case v >= 0 && v <= 127:
  2881  			return Yu7
  2882  		case v >= 0 && v <= 255:
  2883  			return Yu8
  2884  		case v >= -128 && v <= 127:
  2885  			return Yi8
  2886  		}
  2887  		if ctxt.Arch.Family == sys.I386 {
  2888  			return Yi32
  2889  		}
  2890  		l := int32(v)
  2891  		if int64(l) == v {
  2892  			return Ys32 // can sign extend
  2893  		}
  2894  		if v>>32 == 0 {
  2895  			return Yi32 // unsigned
  2896  		}
  2897  		return Yi64
  2898  
  2899  	case obj.TYPE_TEXTSIZE:
  2900  		return Ytextsize
  2901  	}
  2902  
  2903  	if a.Type != obj.TYPE_REG {
  2904  		ctxt.Diag("unexpected addr1: type=%d %v", a.Type, obj.Dconv(p, a))
  2905  		return Yxxx
  2906  	}
  2907  
  2908  	switch a.Reg {
  2909  	case REG_AL:
  2910  		return Yal
  2911  
  2912  	case REG_AX:
  2913  		return Yax
  2914  
  2915  		/*
  2916  			case REG_SPB:
  2917  		*/
  2918  	case REG_BPB,
  2919  		REG_SIB,
  2920  		REG_DIB,
  2921  		REG_R8B,
  2922  		REG_R9B,
  2923  		REG_R10B,
  2924  		REG_R11B,
  2925  		REG_R12B,
  2926  		REG_R13B,
  2927  		REG_R14B,
  2928  		REG_R15B:
  2929  		if ctxt.Arch.Family == sys.I386 {
  2930  			return Yxxx
  2931  		}
  2932  		fallthrough
  2933  
  2934  	case REG_DL,
  2935  		REG_BL,
  2936  		REG_AH,
  2937  		REG_CH,
  2938  		REG_DH,
  2939  		REG_BH:
  2940  		return Yrb
  2941  
  2942  	case REG_CL:
  2943  		return Ycl
  2944  
  2945  	case REG_CX:
  2946  		return Ycx
  2947  
  2948  	case REG_DX, REG_BX:
  2949  		return Yrx
  2950  
  2951  	case REG_R8, // not really Yrl
  2952  		REG_R9,
  2953  		REG_R10,
  2954  		REG_R11,
  2955  		REG_R12,
  2956  		REG_R13,
  2957  		REG_R14,
  2958  		REG_R15:
  2959  		if ctxt.Arch.Family == sys.I386 {
  2960  			return Yxxx
  2961  		}
  2962  		fallthrough
  2963  
  2964  	case REG_SP, REG_BP, REG_SI, REG_DI:
  2965  		if ctxt.Arch.Family == sys.I386 {
  2966  			return Yrl32
  2967  		}
  2968  		return Yrl
  2969  
  2970  	case REG_F0 + 0:
  2971  		return Yf0
  2972  
  2973  	case REG_F0 + 1,
  2974  		REG_F0 + 2,
  2975  		REG_F0 + 3,
  2976  		REG_F0 + 4,
  2977  		REG_F0 + 5,
  2978  		REG_F0 + 6,
  2979  		REG_F0 + 7:
  2980  		return Yrf
  2981  
  2982  	case REG_M0 + 0,
  2983  		REG_M0 + 1,
  2984  		REG_M0 + 2,
  2985  		REG_M0 + 3,
  2986  		REG_M0 + 4,
  2987  		REG_M0 + 5,
  2988  		REG_M0 + 6,
  2989  		REG_M0 + 7:
  2990  		return Ymr
  2991  
  2992  	case REG_X0:
  2993  		return Yxr0
  2994  
  2995  	case REG_X0 + 1,
  2996  		REG_X0 + 2,
  2997  		REG_X0 + 3,
  2998  		REG_X0 + 4,
  2999  		REG_X0 + 5,
  3000  		REG_X0 + 6,
  3001  		REG_X0 + 7,
  3002  		REG_X0 + 8,
  3003  		REG_X0 + 9,
  3004  		REG_X0 + 10,
  3005  		REG_X0 + 11,
  3006  		REG_X0 + 12,
  3007  		REG_X0 + 13,
  3008  		REG_X0 + 14,
  3009  		REG_X0 + 15:
  3010  		return Yxr
  3011  
  3012  	case REG_X0 + 16,
  3013  		REG_X0 + 17,
  3014  		REG_X0 + 18,
  3015  		REG_X0 + 19,
  3016  		REG_X0 + 20,
  3017  		REG_X0 + 21,
  3018  		REG_X0 + 22,
  3019  		REG_X0 + 23,
  3020  		REG_X0 + 24,
  3021  		REG_X0 + 25,
  3022  		REG_X0 + 26,
  3023  		REG_X0 + 27,
  3024  		REG_X0 + 28,
  3025  		REG_X0 + 29,
  3026  		REG_X0 + 30,
  3027  		REG_X0 + 31:
  3028  		return YxrEvex
  3029  
  3030  	case REG_Y0 + 0,
  3031  		REG_Y0 + 1,
  3032  		REG_Y0 + 2,
  3033  		REG_Y0 + 3,
  3034  		REG_Y0 + 4,
  3035  		REG_Y0 + 5,
  3036  		REG_Y0 + 6,
  3037  		REG_Y0 + 7,
  3038  		REG_Y0 + 8,
  3039  		REG_Y0 + 9,
  3040  		REG_Y0 + 10,
  3041  		REG_Y0 + 11,
  3042  		REG_Y0 + 12,
  3043  		REG_Y0 + 13,
  3044  		REG_Y0 + 14,
  3045  		REG_Y0 + 15:
  3046  		return Yyr
  3047  
  3048  	case REG_Y0 + 16,
  3049  		REG_Y0 + 17,
  3050  		REG_Y0 + 18,
  3051  		REG_Y0 + 19,
  3052  		REG_Y0 + 20,
  3053  		REG_Y0 + 21,
  3054  		REG_Y0 + 22,
  3055  		REG_Y0 + 23,
  3056  		REG_Y0 + 24,
  3057  		REG_Y0 + 25,
  3058  		REG_Y0 + 26,
  3059  		REG_Y0 + 27,
  3060  		REG_Y0 + 28,
  3061  		REG_Y0 + 29,
  3062  		REG_Y0 + 30,
  3063  		REG_Y0 + 31:
  3064  		return YyrEvex
  3065  
  3066  	case REG_Z0 + 0,
  3067  		REG_Z0 + 1,
  3068  		REG_Z0 + 2,
  3069  		REG_Z0 + 3,
  3070  		REG_Z0 + 4,
  3071  		REG_Z0 + 5,
  3072  		REG_Z0 + 6,
  3073  		REG_Z0 + 7:
  3074  		return Yzr
  3075  
  3076  	case REG_Z0 + 8,
  3077  		REG_Z0 + 9,
  3078  		REG_Z0 + 10,
  3079  		REG_Z0 + 11,
  3080  		REG_Z0 + 12,
  3081  		REG_Z0 + 13,
  3082  		REG_Z0 + 14,
  3083  		REG_Z0 + 15,
  3084  		REG_Z0 + 16,
  3085  		REG_Z0 + 17,
  3086  		REG_Z0 + 18,
  3087  		REG_Z0 + 19,
  3088  		REG_Z0 + 20,
  3089  		REG_Z0 + 21,
  3090  		REG_Z0 + 22,
  3091  		REG_Z0 + 23,
  3092  		REG_Z0 + 24,
  3093  		REG_Z0 + 25,
  3094  		REG_Z0 + 26,
  3095  		REG_Z0 + 27,
  3096  		REG_Z0 + 28,
  3097  		REG_Z0 + 29,
  3098  		REG_Z0 + 30,
  3099  		REG_Z0 + 31:
  3100  		if ctxt.Arch.Family == sys.I386 {
  3101  			return Yxxx
  3102  		}
  3103  		return Yzr
  3104  
  3105  	case REG_K0:
  3106  		return Yk0
  3107  
  3108  	case REG_K0 + 1,
  3109  		REG_K0 + 2,
  3110  		REG_K0 + 3,
  3111  		REG_K0 + 4,
  3112  		REG_K0 + 5,
  3113  		REG_K0 + 6,
  3114  		REG_K0 + 7:
  3115  		return Yknot0
  3116  
  3117  	case REG_CS:
  3118  		return Ycs
  3119  	case REG_SS:
  3120  		return Yss
  3121  	case REG_DS:
  3122  		return Yds
  3123  	case REG_ES:
  3124  		return Yes
  3125  	case REG_FS:
  3126  		return Yfs
  3127  	case REG_GS:
  3128  		return Ygs
  3129  	case REG_TLS:
  3130  		return Ytls
  3131  
  3132  	case REG_GDTR:
  3133  		return Ygdtr
  3134  	case REG_IDTR:
  3135  		return Yidtr
  3136  	case REG_LDTR:
  3137  		return Yldtr
  3138  	case REG_MSW:
  3139  		return Ymsw
  3140  	case REG_TASK:
  3141  		return Ytask
  3142  
  3143  	case REG_CR + 0:
  3144  		return Ycr0
  3145  	case REG_CR + 1:
  3146  		return Ycr1
  3147  	case REG_CR + 2:
  3148  		return Ycr2
  3149  	case REG_CR + 3:
  3150  		return Ycr3
  3151  	case REG_CR + 4:
  3152  		return Ycr4
  3153  	case REG_CR + 5:
  3154  		return Ycr5
  3155  	case REG_CR + 6:
  3156  		return Ycr6
  3157  	case REG_CR + 7:
  3158  		return Ycr7
  3159  	case REG_CR + 8:
  3160  		return Ycr8
  3161  
  3162  	case REG_DR + 0:
  3163  		return Ydr0
  3164  	case REG_DR + 1:
  3165  		return Ydr1
  3166  	case REG_DR + 2:
  3167  		return Ydr2
  3168  	case REG_DR + 3:
  3169  		return Ydr3
  3170  	case REG_DR + 4:
  3171  		return Ydr4
  3172  	case REG_DR + 5:
  3173  		return Ydr5
  3174  	case REG_DR + 6:
  3175  		return Ydr6
  3176  	case REG_DR + 7:
  3177  		return Ydr7
  3178  
  3179  	case REG_TR + 0:
  3180  		return Ytr0
  3181  	case REG_TR + 1:
  3182  		return Ytr1
  3183  	case REG_TR + 2:
  3184  		return Ytr2
  3185  	case REG_TR + 3:
  3186  		return Ytr3
  3187  	case REG_TR + 4:
  3188  		return Ytr4
  3189  	case REG_TR + 5:
  3190  		return Ytr5
  3191  	case REG_TR + 6:
  3192  		return Ytr6
  3193  	case REG_TR + 7:
  3194  		return Ytr7
  3195  	}
  3196  
  3197  	return Yxxx
  3198  }
  3199  
  3200  // AsmBuf is a simple buffer to assemble variable-length x86 instructions into
  3201  // and hold assembly state.
  3202  type AsmBuf struct {
  3203  	buf      [100]byte
  3204  	off      int
  3205  	rexflag  int
  3206  	vexflag  bool // Per inst: true for VEX-encoded
  3207  	evexflag bool // Per inst: true for EVEX-encoded
  3208  	rep      bool
  3209  	repn     bool
  3210  	lock     bool
  3211  
  3212  	evex evexBits // Initialized when evexflag is true
  3213  }
  3214  
  3215  // Put1 appends one byte to the end of the buffer.
  3216  func (ab *AsmBuf) Put1(x byte) {
  3217  	ab.buf[ab.off] = x
  3218  	ab.off++
  3219  }
  3220  
  3221  // Put2 appends two bytes to the end of the buffer.
  3222  func (ab *AsmBuf) Put2(x, y byte) {
  3223  	ab.buf[ab.off+0] = x
  3224  	ab.buf[ab.off+1] = y
  3225  	ab.off += 2
  3226  }
  3227  
  3228  // Put3 appends three bytes to the end of the buffer.
  3229  func (ab *AsmBuf) Put3(x, y, z byte) {
  3230  	ab.buf[ab.off+0] = x
  3231  	ab.buf[ab.off+1] = y
  3232  	ab.buf[ab.off+2] = z
  3233  	ab.off += 3
  3234  }
  3235  
  3236  // Put4 appends four bytes to the end of the buffer.
  3237  func (ab *AsmBuf) Put4(x, y, z, w byte) {
  3238  	ab.buf[ab.off+0] = x
  3239  	ab.buf[ab.off+1] = y
  3240  	ab.buf[ab.off+2] = z
  3241  	ab.buf[ab.off+3] = w
  3242  	ab.off += 4
  3243  }
  3244  
  3245  // PutInt16 writes v into the buffer using little-endian encoding.
  3246  func (ab *AsmBuf) PutInt16(v int16) {
  3247  	ab.buf[ab.off+0] = byte(v)
  3248  	ab.buf[ab.off+1] = byte(v >> 8)
  3249  	ab.off += 2
  3250  }
  3251  
  3252  // PutInt32 writes v into the buffer using little-endian encoding.
  3253  func (ab *AsmBuf) PutInt32(v int32) {
  3254  	ab.buf[ab.off+0] = byte(v)
  3255  	ab.buf[ab.off+1] = byte(v >> 8)
  3256  	ab.buf[ab.off+2] = byte(v >> 16)
  3257  	ab.buf[ab.off+3] = byte(v >> 24)
  3258  	ab.off += 4
  3259  }
  3260  
  3261  // PutInt64 writes v into the buffer using little-endian encoding.
  3262  func (ab *AsmBuf) PutInt64(v int64) {
  3263  	ab.buf[ab.off+0] = byte(v)
  3264  	ab.buf[ab.off+1] = byte(v >> 8)
  3265  	ab.buf[ab.off+2] = byte(v >> 16)
  3266  	ab.buf[ab.off+3] = byte(v >> 24)
  3267  	ab.buf[ab.off+4] = byte(v >> 32)
  3268  	ab.buf[ab.off+5] = byte(v >> 40)
  3269  	ab.buf[ab.off+6] = byte(v >> 48)
  3270  	ab.buf[ab.off+7] = byte(v >> 56)
  3271  	ab.off += 8
  3272  }
  3273  
  3274  // Put copies b into the buffer.
  3275  func (ab *AsmBuf) Put(b []byte) {
  3276  	copy(ab.buf[ab.off:], b)
  3277  	ab.off += len(b)
  3278  }
  3279  
  3280  // PutOpBytesLit writes zero terminated sequence of bytes from op,
  3281  // starting at specified offset (e.g. z counter value).
  3282  // Trailing 0 is not written.
  3283  //
  3284  // Intended to be used for literal Z cases.
  3285  // Literal Z cases usually have "Zlit" in their name (Zlit, Zlitr_m, Zlitm_r).
  3286  func (ab *AsmBuf) PutOpBytesLit(offset int, op *opBytes) {
  3287  	for int(op[offset]) != 0 {
  3288  		ab.Put1(op[offset])
  3289  		offset++
  3290  	}
  3291  }
  3292  
  3293  // Insert inserts b at offset i.
  3294  func (ab *AsmBuf) Insert(i int, b byte) {
  3295  	ab.off++
  3296  	copy(ab.buf[i+1:ab.off], ab.buf[i:ab.off-1])
  3297  	ab.buf[i] = b
  3298  }
  3299  
  3300  // Last returns the byte at the end of the buffer.
  3301  func (ab *AsmBuf) Last() byte { return ab.buf[ab.off-1] }
  3302  
  3303  // Len returns the length of the buffer.
  3304  func (ab *AsmBuf) Len() int { return ab.off }
  3305  
  3306  // Bytes returns the contents of the buffer.
  3307  func (ab *AsmBuf) Bytes() []byte { return ab.buf[:ab.off] }
  3308  
  3309  // Reset empties the buffer.
  3310  func (ab *AsmBuf) Reset() { ab.off = 0 }
  3311  
  3312  // At returns the byte at offset i.
  3313  func (ab *AsmBuf) At(i int) byte { return ab.buf[i] }
  3314  
  3315  // asmidx emits SIB byte.
  3316  func (ab *AsmBuf) asmidx(ctxt *obj.Link, scale int, index int, base int) {
  3317  	var i int
  3318  
  3319  	// X/Y index register is used in VSIB.
  3320  	switch index {
  3321  	default:
  3322  		goto bad
  3323  
  3324  	case REG_NONE:
  3325  		i = 4 << 3
  3326  		goto bas
  3327  
  3328  	case REG_R8,
  3329  		REG_R9,
  3330  		REG_R10,
  3331  		REG_R11,
  3332  		REG_R12,
  3333  		REG_R13,
  3334  		REG_R14,
  3335  		REG_R15,
  3336  		REG_X8,
  3337  		REG_X9,
  3338  		REG_X10,
  3339  		REG_X11,
  3340  		REG_X12,
  3341  		REG_X13,
  3342  		REG_X14,
  3343  		REG_X15,
  3344  		REG_X16,
  3345  		REG_X17,
  3346  		REG_X18,
  3347  		REG_X19,
  3348  		REG_X20,
  3349  		REG_X21,
  3350  		REG_X22,
  3351  		REG_X23,
  3352  		REG_X24,
  3353  		REG_X25,
  3354  		REG_X26,
  3355  		REG_X27,
  3356  		REG_X28,
  3357  		REG_X29,
  3358  		REG_X30,
  3359  		REG_X31,
  3360  		REG_Y8,
  3361  		REG_Y9,
  3362  		REG_Y10,
  3363  		REG_Y11,
  3364  		REG_Y12,
  3365  		REG_Y13,
  3366  		REG_Y14,
  3367  		REG_Y15,
  3368  		REG_Y16,
  3369  		REG_Y17,
  3370  		REG_Y18,
  3371  		REG_Y19,
  3372  		REG_Y20,
  3373  		REG_Y21,
  3374  		REG_Y22,
  3375  		REG_Y23,
  3376  		REG_Y24,
  3377  		REG_Y25,
  3378  		REG_Y26,
  3379  		REG_Y27,
  3380  		REG_Y28,
  3381  		REG_Y29,
  3382  		REG_Y30,
  3383  		REG_Y31,
  3384  		REG_Z8,
  3385  		REG_Z9,
  3386  		REG_Z10,
  3387  		REG_Z11,
  3388  		REG_Z12,
  3389  		REG_Z13,
  3390  		REG_Z14,
  3391  		REG_Z15,
  3392  		REG_Z16,
  3393  		REG_Z17,
  3394  		REG_Z18,
  3395  		REG_Z19,
  3396  		REG_Z20,
  3397  		REG_Z21,
  3398  		REG_Z22,
  3399  		REG_Z23,
  3400  		REG_Z24,
  3401  		REG_Z25,
  3402  		REG_Z26,
  3403  		REG_Z27,
  3404  		REG_Z28,
  3405  		REG_Z29,
  3406  		REG_Z30,
  3407  		REG_Z31:
  3408  		if ctxt.Arch.Family == sys.I386 {
  3409  			goto bad
  3410  		}
  3411  		fallthrough
  3412  
  3413  	case REG_AX,
  3414  		REG_CX,
  3415  		REG_DX,
  3416  		REG_BX,
  3417  		REG_BP,
  3418  		REG_SI,
  3419  		REG_DI,
  3420  		REG_X0,
  3421  		REG_X1,
  3422  		REG_X2,
  3423  		REG_X3,
  3424  		REG_X4,
  3425  		REG_X5,
  3426  		REG_X6,
  3427  		REG_X7,
  3428  		REG_Y0,
  3429  		REG_Y1,
  3430  		REG_Y2,
  3431  		REG_Y3,
  3432  		REG_Y4,
  3433  		REG_Y5,
  3434  		REG_Y6,
  3435  		REG_Y7,
  3436  		REG_Z0,
  3437  		REG_Z1,
  3438  		REG_Z2,
  3439  		REG_Z3,
  3440  		REG_Z4,
  3441  		REG_Z5,
  3442  		REG_Z6,
  3443  		REG_Z7:
  3444  		i = reg[index] << 3
  3445  	}
  3446  
  3447  	switch scale {
  3448  	default:
  3449  		goto bad
  3450  
  3451  	case 1:
  3452  		break
  3453  
  3454  	case 2:
  3455  		i |= 1 << 6
  3456  
  3457  	case 4:
  3458  		i |= 2 << 6
  3459  
  3460  	case 8:
  3461  		i |= 3 << 6
  3462  	}
  3463  
  3464  bas:
  3465  	switch base {
  3466  	default:
  3467  		goto bad
  3468  
  3469  	case REG_NONE: // must be mod=00
  3470  		i |= 5
  3471  
  3472  	case REG_R8,
  3473  		REG_R9,
  3474  		REG_R10,
  3475  		REG_R11,
  3476  		REG_R12,
  3477  		REG_R13,
  3478  		REG_R14,
  3479  		REG_R15:
  3480  		if ctxt.Arch.Family == sys.I386 {
  3481  			goto bad
  3482  		}
  3483  		fallthrough
  3484  
  3485  	case REG_AX,
  3486  		REG_CX,
  3487  		REG_DX,
  3488  		REG_BX,
  3489  		REG_SP,
  3490  		REG_BP,
  3491  		REG_SI,
  3492  		REG_DI:
  3493  		i |= reg[base]
  3494  	}
  3495  
  3496  	ab.Put1(byte(i))
  3497  	return
  3498  
  3499  bad:
  3500  	ctxt.Diag("asmidx: bad address %d/%s/%s", scale, rconv(index), rconv(base))
  3501  	ab.Put1(0)
  3502  }
  3503  
  3504  func (ab *AsmBuf) relput4(ctxt *obj.Link, cursym *obj.LSym, p *obj.Prog, a *obj.Addr) {
  3505  	var rel obj.Reloc
  3506  
  3507  	v := vaddr(ctxt, p, a, &rel)
  3508  	if rel.Siz != 0 {
  3509  		if rel.Siz != 4 {
  3510  			ctxt.Diag("bad reloc")
  3511  		}
  3512  		rel.Off = int32(p.Pc + int64(ab.Len()))
  3513  		cursym.AddRel(ctxt, rel)
  3514  	}
  3515  
  3516  	ab.PutInt32(int32(v))
  3517  }
  3518  
  3519  func vaddr(ctxt *obj.Link, p *obj.Prog, a *obj.Addr, r *obj.Reloc) int64 {
  3520  	if r != nil {
  3521  		*r = obj.Reloc{}
  3522  	}
  3523  
  3524  	switch a.Name {
  3525  	case obj.NAME_STATIC,
  3526  		obj.NAME_GOTREF,
  3527  		obj.NAME_EXTERN:
  3528  		s := a.Sym
  3529  		if r == nil {
  3530  			ctxt.Diag("need reloc for %v", obj.Dconv(p, a))
  3531  			log.Fatalf("reloc")
  3532  		}
  3533  
  3534  		if a.Name == obj.NAME_GOTREF {
  3535  			r.Siz = 4
  3536  			r.Type = objabi.R_GOTPCREL
  3537  		} else if useAbs(ctxt, s) {
  3538  			r.Siz = 4
  3539  			r.Type = objabi.R_ADDR
  3540  		} else {
  3541  			r.Siz = 4
  3542  			r.Type = objabi.R_PCREL
  3543  		}
  3544  
  3545  		r.Off = -1 // caller must fill in
  3546  		r.Sym = s
  3547  		r.Add = a.Offset
  3548  
  3549  		return 0
  3550  	}
  3551  
  3552  	if (a.Type == obj.TYPE_MEM || a.Type == obj.TYPE_ADDR) && a.Reg == REG_TLS {
  3553  		if r == nil {
  3554  			ctxt.Diag("need reloc for %v", obj.Dconv(p, a))
  3555  			log.Fatalf("reloc")
  3556  		}
  3557  
  3558  		if !ctxt.Flag_shared || isAndroid || ctxt.Headtype == objabi.Hdarwin {
  3559  			r.Type = objabi.R_TLS_LE
  3560  			r.Siz = 4
  3561  			r.Off = -1 // caller must fill in
  3562  			r.Add = a.Offset
  3563  		}
  3564  		return 0
  3565  	}
  3566  
  3567  	return a.Offset
  3568  }
  3569  
  3570  func (ab *AsmBuf) asmandsz(ctxt *obj.Link, cursym *obj.LSym, p *obj.Prog, a *obj.Addr, r int, rex int, m64 int) {
  3571  	var base int
  3572  	var rel obj.Reloc
  3573  
  3574  	rex &= 0x40 | Rxr
  3575  	if a.Offset != int64(int32(a.Offset)) {
  3576  		// The rules are slightly different for 386 and AMD64,
  3577  		// mostly for historical reasons. We may unify them later,
  3578  		// but it must be discussed beforehand.
  3579  		//
  3580  		// For 64bit mode only LEAL is allowed to overflow.
  3581  		// It's how https://golang.org/cl/59630 made it.
  3582  		// crypto/sha1/sha1block_amd64.s depends on this feature.
  3583  		//
  3584  		// For 32bit mode rules are more permissive.
  3585  		// If offset fits uint32, it's permitted.
  3586  		// This is allowed for assembly that wants to use 32-bit hex
  3587  		// constants, e.g. LEAL 0x99999999(AX), AX.
  3588  		overflowOK := (ctxt.Arch.Family == sys.AMD64 && p.As == ALEAL) ||
  3589  			(ctxt.Arch.Family != sys.AMD64 &&
  3590  				int64(uint32(a.Offset)) == a.Offset &&
  3591  				ab.rexflag&Rxw == 0)
  3592  		if !overflowOK {
  3593  			ctxt.Diag("offset too large in %s", p)
  3594  		}
  3595  	}
  3596  	v := int32(a.Offset)
  3597  	rel.Siz = 0
  3598  
  3599  	switch a.Type {
  3600  	case obj.TYPE_ADDR:
  3601  		if a.Name == obj.NAME_NONE {
  3602  			ctxt.Diag("unexpected TYPE_ADDR with NAME_NONE")
  3603  		}
  3604  		if a.Index == REG_TLS {
  3605  			ctxt.Diag("unexpected TYPE_ADDR with index==REG_TLS")
  3606  		}
  3607  		goto bad
  3608  
  3609  	case obj.TYPE_REG:
  3610  		const regFirst = REG_AL
  3611  		const regLast = REG_Z31
  3612  		if a.Reg < regFirst || regLast < a.Reg {
  3613  			goto bad
  3614  		}
  3615  		if v != 0 {
  3616  			goto bad
  3617  		}
  3618  		ab.Put1(byte(3<<6 | reg[a.Reg]<<0 | r<<3))
  3619  		ab.rexflag |= regrex[a.Reg]&(0x40|Rxb) | rex
  3620  		return
  3621  	}
  3622  
  3623  	if a.Type != obj.TYPE_MEM {
  3624  		goto bad
  3625  	}
  3626  
  3627  	if a.Index != REG_NONE && a.Index != REG_TLS && !(REG_CS <= a.Index && a.Index <= REG_GS) {
  3628  		base := int(a.Reg)
  3629  		switch a.Name {
  3630  		case obj.NAME_EXTERN,
  3631  			obj.NAME_GOTREF,
  3632  			obj.NAME_STATIC:
  3633  			if !useAbs(ctxt, a.Sym) && ctxt.Arch.Family == sys.AMD64 {
  3634  				goto bad
  3635  			}
  3636  			if ctxt.Arch.Family == sys.I386 && ctxt.Flag_shared {
  3637  				// The base register has already been set. It holds the PC
  3638  				// of this instruction returned by a PC-reading thunk.
  3639  				// See obj6.go:rewriteToPcrel.
  3640  			} else {
  3641  				base = REG_NONE
  3642  			}
  3643  			v = int32(vaddr(ctxt, p, a, &rel))
  3644  
  3645  		case obj.NAME_AUTO,
  3646  			obj.NAME_PARAM:
  3647  			base = REG_SP
  3648  		}
  3649  
  3650  		ab.rexflag |= regrex[int(a.Index)]&Rxx | regrex[base]&Rxb | rex
  3651  		if base == REG_NONE {
  3652  			ab.Put1(byte(0<<6 | 4<<0 | r<<3))
  3653  			ab.asmidx(ctxt, int(a.Scale), int(a.Index), base)
  3654  			goto putrelv
  3655  		}
  3656  
  3657  		if v == 0 && rel.Siz == 0 && base != REG_BP && base != REG_R13 {
  3658  			ab.Put1(byte(0<<6 | 4<<0 | r<<3))
  3659  			ab.asmidx(ctxt, int(a.Scale), int(a.Index), base)
  3660  			return
  3661  		}
  3662  
  3663  		if disp8, ok := toDisp8(v, p, ab); ok && rel.Siz == 0 {
  3664  			ab.Put1(byte(1<<6 | 4<<0 | r<<3))
  3665  			ab.asmidx(ctxt, int(a.Scale), int(a.Index), base)
  3666  			ab.Put1(disp8)
  3667  			return
  3668  		}
  3669  
  3670  		ab.Put1(byte(2<<6 | 4<<0 | r<<3))
  3671  		ab.asmidx(ctxt, int(a.Scale), int(a.Index), base)
  3672  		goto putrelv
  3673  	}
  3674  
  3675  	base = int(a.Reg)
  3676  	switch a.Name {
  3677  	case obj.NAME_STATIC,
  3678  		obj.NAME_GOTREF,
  3679  		obj.NAME_EXTERN:
  3680  		if a.Sym == nil {
  3681  			ctxt.Diag("bad addr: %v", p)
  3682  		}
  3683  		if ctxt.Arch.Family == sys.I386 && ctxt.Flag_shared {
  3684  			// The base register has already been set. It holds the PC
  3685  			// of this instruction returned by a PC-reading thunk.
  3686  			// See obj6.go:rewriteToPcrel.
  3687  		} else {
  3688  			base = REG_NONE
  3689  		}
  3690  		v = int32(vaddr(ctxt, p, a, &rel))
  3691  
  3692  	case obj.NAME_AUTO,
  3693  		obj.NAME_PARAM:
  3694  		base = REG_SP
  3695  	}
  3696  
  3697  	if base == REG_TLS {
  3698  		v = int32(vaddr(ctxt, p, a, &rel))
  3699  	}
  3700  
  3701  	ab.rexflag |= regrex[base]&Rxb | rex
  3702  	if base == REG_NONE || (REG_CS <= base && base <= REG_GS) || base == REG_TLS {
  3703  		if (a.Sym == nil || !useAbs(ctxt, a.Sym)) && base == REG_NONE && (a.Name == obj.NAME_STATIC || a.Name == obj.NAME_EXTERN || a.Name == obj.NAME_GOTREF) || ctxt.Arch.Family != sys.AMD64 {
  3704  			if a.Name == obj.NAME_GOTREF && (a.Offset != 0 || a.Index != 0 || a.Scale != 0) {
  3705  				ctxt.Diag("%v has offset against gotref", p)
  3706  			}
  3707  			ab.Put1(byte(0<<6 | 5<<0 | r<<3))
  3708  			goto putrelv
  3709  		}
  3710  
  3711  		// temporary
  3712  		ab.Put2(
  3713  			byte(0<<6|4<<0|r<<3), // sib present
  3714  			0<<6|4<<3|5<<0,       // DS:d32
  3715  		)
  3716  		goto putrelv
  3717  	}
  3718  
  3719  	if base == REG_SP || base == REG_R12 {
  3720  		if v == 0 {
  3721  			ab.Put1(byte(0<<6 | reg[base]<<0 | r<<3))
  3722  			ab.asmidx(ctxt, int(a.Scale), REG_NONE, base)
  3723  			return
  3724  		}
  3725  
  3726  		if disp8, ok := toDisp8(v, p, ab); ok {
  3727  			ab.Put1(byte(1<<6 | reg[base]<<0 | r<<3))
  3728  			ab.asmidx(ctxt, int(a.Scale), REG_NONE, base)
  3729  			ab.Put1(disp8)
  3730  			return
  3731  		}
  3732  
  3733  		ab.Put1(byte(2<<6 | reg[base]<<0 | r<<3))
  3734  		ab.asmidx(ctxt, int(a.Scale), REG_NONE, base)
  3735  		goto putrelv
  3736  	}
  3737  
  3738  	if REG_AX <= base && base <= REG_R15 {
  3739  		if a.Index == REG_TLS && !ctxt.Flag_shared && !isAndroid &&
  3740  			ctxt.Headtype != objabi.Hwindows {
  3741  			rel = obj.Reloc{}
  3742  			rel.Type = objabi.R_TLS_LE
  3743  			rel.Siz = 4
  3744  			rel.Sym = nil
  3745  			rel.Add = int64(v)
  3746  			v = 0
  3747  		}
  3748  
  3749  		if v == 0 && rel.Siz == 0 && base != REG_BP && base != REG_R13 {
  3750  			ab.Put1(byte(0<<6 | reg[base]<<0 | r<<3))
  3751  			return
  3752  		}
  3753  
  3754  		if disp8, ok := toDisp8(v, p, ab); ok && rel.Siz == 0 {
  3755  			ab.Put2(byte(1<<6|reg[base]<<0|r<<3), disp8)
  3756  			return
  3757  		}
  3758  
  3759  		ab.Put1(byte(2<<6 | reg[base]<<0 | r<<3))
  3760  		goto putrelv
  3761  	}
  3762  
  3763  	goto bad
  3764  
  3765  putrelv:
  3766  	if rel.Siz != 0 {
  3767  		if rel.Siz != 4 {
  3768  			ctxt.Diag("bad rel")
  3769  			goto bad
  3770  		}
  3771  
  3772  		rel.Off = int32(p.Pc + int64(ab.Len()))
  3773  		cursym.AddRel(ctxt, rel)
  3774  	}
  3775  
  3776  	ab.PutInt32(v)
  3777  	return
  3778  
  3779  bad:
  3780  	ctxt.Diag("asmand: bad address %v", obj.Dconv(p, a))
  3781  }
  3782  
  3783  func (ab *AsmBuf) asmand(ctxt *obj.Link, cursym *obj.LSym, p *obj.Prog, a *obj.Addr, ra *obj.Addr) {
  3784  	ab.asmandsz(ctxt, cursym, p, a, reg[ra.Reg], regrex[ra.Reg], 0)
  3785  }
  3786  
  3787  func (ab *AsmBuf) asmando(ctxt *obj.Link, cursym *obj.LSym, p *obj.Prog, a *obj.Addr, o int) {
  3788  	ab.asmandsz(ctxt, cursym, p, a, o, 0, 0)
  3789  }
  3790  
  3791  func bytereg(a *obj.Addr, t *uint8) {
  3792  	if a.Type == obj.TYPE_REG && a.Index == REG_NONE && (REG_AX <= a.Reg && a.Reg <= REG_R15) {
  3793  		a.Reg += REG_AL - REG_AX
  3794  		*t = 0
  3795  	}
  3796  }
  3797  
  3798  func unbytereg(a *obj.Addr, t *uint8) {
  3799  	if a.Type == obj.TYPE_REG && a.Index == REG_NONE && (REG_AL <= a.Reg && a.Reg <= REG_R15B) {
  3800  		a.Reg += REG_AX - REG_AL
  3801  		*t = 0
  3802  	}
  3803  }
  3804  
  3805  const (
  3806  	movLit uint8 = iota // Like Zlit
  3807  	movRegMem
  3808  	movMemReg
  3809  	movRegMem2op
  3810  	movMemReg2op
  3811  	movFullPtr // Load full pointer, trash heap (unsupported)
  3812  	movDoubleShift
  3813  	movTLSReg
  3814  )
  3815  
  3816  var ymovtab = []movtab{
  3817  	// push
  3818  	{APUSHL, Ycs, Ynone, Ynone, movLit, [4]uint8{0x0e, 0}},
  3819  	{APUSHL, Yss, Ynone, Ynone, movLit, [4]uint8{0x16, 0}},
  3820  	{APUSHL, Yds, Ynone, Ynone, movLit, [4]uint8{0x1e, 0}},
  3821  	{APUSHL, Yes, Ynone, Ynone, movLit, [4]uint8{0x06, 0}},
  3822  	{APUSHL, Yfs, Ynone, Ynone, movLit, [4]uint8{0x0f, 0xa0, 0}},
  3823  	{APUSHL, Ygs, Ynone, Ynone, movLit, [4]uint8{0x0f, 0xa8, 0}},
  3824  	{APUSHQ, Yfs, Ynone, Ynone, movLit, [4]uint8{0x0f, 0xa0, 0}},
  3825  	{APUSHQ, Ygs, Ynone, Ynone, movLit, [4]uint8{0x0f, 0xa8, 0}},
  3826  	{APUSHW, Ycs, Ynone, Ynone, movLit, [4]uint8{Pe, 0x0e, 0}},
  3827  	{APUSHW, Yss, Ynone, Ynone, movLit, [4]uint8{Pe, 0x16, 0}},
  3828  	{APUSHW, Yds, Ynone, Ynone, movLit, [4]uint8{Pe, 0x1e, 0}},
  3829  	{APUSHW, Yes, Ynone, Ynone, movLit, [4]uint8{Pe, 0x06, 0}},
  3830  	{APUSHW, Yfs, Ynone, Ynone, movLit, [4]uint8{Pe, 0x0f, 0xa0, 0}},
  3831  	{APUSHW, Ygs, Ynone, Ynone, movLit, [4]uint8{Pe, 0x0f, 0xa8, 0}},
  3832  
  3833  	// pop
  3834  	{APOPL, Ynone, Ynone, Yds, movLit, [4]uint8{0x1f, 0}},
  3835  	{APOPL, Ynone, Ynone, Yes, movLit, [4]uint8{0x07, 0}},
  3836  	{APOPL, Ynone, Ynone, Yss, movLit, [4]uint8{0x17, 0}},
  3837  	{APOPL, Ynone, Ynone, Yfs, movLit, [4]uint8{0x0f, 0xa1, 0}},
  3838  	{APOPL, Ynone, Ynone, Ygs, movLit, [4]uint8{0x0f, 0xa9, 0}},
  3839  	{APOPQ, Ynone, Ynone, Yfs, movLit, [4]uint8{0x0f, 0xa1, 0}},
  3840  	{APOPQ, Ynone, Ynone, Ygs, movLit, [4]uint8{0x0f, 0xa9, 0}},
  3841  	{APOPW, Ynone, Ynone, Yds, movLit, [4]uint8{Pe, 0x1f, 0}},
  3842  	{APOPW, Ynone, Ynone, Yes, movLit, [4]uint8{Pe, 0x07, 0}},
  3843  	{APOPW, Ynone, Ynone, Yss, movLit, [4]uint8{Pe, 0x17, 0}},
  3844  	{APOPW, Ynone, Ynone, Yfs, movLit, [4]uint8{Pe, 0x0f, 0xa1, 0}},
  3845  	{APOPW, Ynone, Ynone, Ygs, movLit, [4]uint8{Pe, 0x0f, 0xa9, 0}},
  3846  
  3847  	// mov seg
  3848  	{AMOVW, Yes, Ynone, Yml, movRegMem, [4]uint8{0x8c, 0, 0, 0}},
  3849  	{AMOVW, Ycs, Ynone, Yml, movRegMem, [4]uint8{0x8c, 1, 0, 0}},
  3850  	{AMOVW, Yss, Ynone, Yml, movRegMem, [4]uint8{0x8c, 2, 0, 0}},
  3851  	{AMOVW, Yds, Ynone, Yml, movRegMem, [4]uint8{0x8c, 3, 0, 0}},
  3852  	{AMOVW, Yfs, Ynone, Yml, movRegMem, [4]uint8{0x8c, 4, 0, 0}},
  3853  	{AMOVW, Ygs, Ynone, Yml, movRegMem, [4]uint8{0x8c, 5, 0, 0}},
  3854  	{AMOVW, Yml, Ynone, Yes, movMemReg, [4]uint8{0x8e, 0, 0, 0}},
  3855  	{AMOVW, Yml, Ynone, Ycs, movMemReg, [4]uint8{0x8e, 1, 0, 0}},
  3856  	{AMOVW, Yml, Ynone, Yss, movMemReg, [4]uint8{0x8e, 2, 0, 0}},
  3857  	{AMOVW, Yml, Ynone, Yds, movMemReg, [4]uint8{0x8e, 3, 0, 0}},
  3858  	{AMOVW, Yml, Ynone, Yfs, movMemReg, [4]uint8{0x8e, 4, 0, 0}},
  3859  	{AMOVW, Yml, Ynone, Ygs, movMemReg, [4]uint8{0x8e, 5, 0, 0}},
  3860  
  3861  	// mov cr
  3862  	{AMOVL, Ycr0, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 0, 0}},
  3863  	{AMOVL, Ycr2, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 2, 0}},
  3864  	{AMOVL, Ycr3, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 3, 0}},
  3865  	{AMOVL, Ycr4, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 4, 0}},
  3866  	{AMOVL, Ycr8, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 8, 0}},
  3867  	{AMOVQ, Ycr0, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 0, 0}},
  3868  	{AMOVQ, Ycr2, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 2, 0}},
  3869  	{AMOVQ, Ycr3, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 3, 0}},
  3870  	{AMOVQ, Ycr4, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 4, 0}},
  3871  	{AMOVQ, Ycr8, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x20, 8, 0}},
  3872  	{AMOVL, Yrl, Ynone, Ycr0, movMemReg2op, [4]uint8{0x0f, 0x22, 0, 0}},
  3873  	{AMOVL, Yrl, Ynone, Ycr2, movMemReg2op, [4]uint8{0x0f, 0x22, 2, 0}},
  3874  	{AMOVL, Yrl, Ynone, Ycr3, movMemReg2op, [4]uint8{0x0f, 0x22, 3, 0}},
  3875  	{AMOVL, Yrl, Ynone, Ycr4, movMemReg2op, [4]uint8{0x0f, 0x22, 4, 0}},
  3876  	{AMOVL, Yrl, Ynone, Ycr8, movMemReg2op, [4]uint8{0x0f, 0x22, 8, 0}},
  3877  	{AMOVQ, Yrl, Ynone, Ycr0, movMemReg2op, [4]uint8{0x0f, 0x22, 0, 0}},
  3878  	{AMOVQ, Yrl, Ynone, Ycr2, movMemReg2op, [4]uint8{0x0f, 0x22, 2, 0}},
  3879  	{AMOVQ, Yrl, Ynone, Ycr3, movMemReg2op, [4]uint8{0x0f, 0x22, 3, 0}},
  3880  	{AMOVQ, Yrl, Ynone, Ycr4, movMemReg2op, [4]uint8{0x0f, 0x22, 4, 0}},
  3881  	{AMOVQ, Yrl, Ynone, Ycr8, movMemReg2op, [4]uint8{0x0f, 0x22, 8, 0}},
  3882  
  3883  	// mov dr
  3884  	{AMOVL, Ydr0, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 0, 0}},
  3885  	{AMOVL, Ydr6, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 6, 0}},
  3886  	{AMOVL, Ydr7, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 7, 0}},
  3887  	{AMOVQ, Ydr0, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 0, 0}},
  3888  	{AMOVQ, Ydr2, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 2, 0}},
  3889  	{AMOVQ, Ydr3, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 3, 0}},
  3890  	{AMOVQ, Ydr6, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 6, 0}},
  3891  	{AMOVQ, Ydr7, Ynone, Yrl, movRegMem2op, [4]uint8{0x0f, 0x21, 7, 0}},
  3892  	{AMOVL, Yrl, Ynone, Ydr0, movMemReg2op, [4]uint8{0x0f, 0x23, 0, 0}},
  3893  	{AMOVL, Yrl, Ynone, Ydr6, movMemReg2op, [4]uint8{0x0f, 0x23, 6, 0}},
  3894  	{AMOVL, Yrl, Ynone, Ydr7, movMemReg2op, [4]uint8{0x0f, 0x23, 7, 0}},
  3895  	{AMOVQ, Yrl, Ynone, Ydr0, movMemReg2op, [4]uint8{0x0f, 0x23, 0, 0}},
  3896  	{AMOVQ, Yrl, Ynone, Ydr2, movMemReg2op, [4]uint8{0x0f, 0x23, 2, 0}},
  3897  	{AMOVQ, Yrl, Ynone, Ydr3, movMemReg2op, [4]uint8{0x0f, 0x23, 3, 0}},
  3898  	{AMOVQ, Yrl, Ynone, Ydr6, movMemReg2op, [4]uint8{0x0f, 0x23, 6, 0}},
  3899  	{AMOVQ, Yrl, Ynone, Ydr7, movMemReg2op, [4]uint8{0x0f, 0x23, 7, 0}},
  3900  
  3901  	// mov tr
  3902  	{AMOVL, Ytr6, Ynone, Yml, movRegMem2op, [4]uint8{0x0f, 0x24, 6, 0}},
  3903  	{AMOVL, Ytr7, Ynone, Yml, movRegMem2op, [4]uint8{0x0f, 0x24, 7, 0}},
  3904  	{AMOVL, Yml, Ynone, Ytr6, movMemReg2op, [4]uint8{0x0f, 0x26, 6, 0xff}},
  3905  	{AMOVL, Yml, Ynone, Ytr7, movMemReg2op, [4]uint8{0x0f, 0x26, 7, 0xff}},
  3906  
  3907  	// lgdt, sgdt, lidt, sidt
  3908  	{AMOVL, Ym, Ynone, Ygdtr, movMemReg2op, [4]uint8{0x0f, 0x01, 2, 0}},
  3909  	{AMOVL, Ygdtr, Ynone, Ym, movRegMem2op, [4]uint8{0x0f, 0x01, 0, 0}},
  3910  	{AMOVL, Ym, Ynone, Yidtr, movMemReg2op, [4]uint8{0x0f, 0x01, 3, 0}},
  3911  	{AMOVL, Yidtr, Ynone, Ym, movRegMem2op, [4]uint8{0x0f, 0x01, 1, 0}},
  3912  	{AMOVQ, Ym, Ynone, Ygdtr, movMemReg2op, [4]uint8{0x0f, 0x01, 2, 0}},
  3913  	{AMOVQ, Ygdtr, Ynone, Ym, movRegMem2op, [4]uint8{0x0f, 0x01, 0, 0}},
  3914  	{AMOVQ, Ym, Ynone, Yidtr, movMemReg2op, [4]uint8{0x0f, 0x01, 3, 0}},
  3915  	{AMOVQ, Yidtr, Ynone, Ym, movRegMem2op, [4]uint8{0x0f, 0x01, 1, 0}},
  3916  
  3917  	// lldt, sldt
  3918  	{AMOVW, Yml, Ynone, Yldtr, movMemReg2op, [4]uint8{0x0f, 0x00, 2, 0}},
  3919  	{AMOVW, Yldtr, Ynone, Yml, movRegMem2op, [4]uint8{0x0f, 0x00, 0, 0}},
  3920  
  3921  	// lmsw, smsw
  3922  	{AMOVW, Yml, Ynone, Ymsw, movMemReg2op, [4]uint8{0x0f, 0x01, 6, 0}},
  3923  	{AMOVW, Ymsw, Ynone, Yml, movRegMem2op, [4]uint8{0x0f, 0x01, 4, 0}},
  3924  
  3925  	// ltr, str
  3926  	{AMOVW, Yml, Ynone, Ytask, movMemReg2op, [4]uint8{0x0f, 0x00, 3, 0}},
  3927  	{AMOVW, Ytask, Ynone, Yml, movRegMem2op, [4]uint8{0x0f, 0x00, 1, 0}},
  3928  
  3929  	/* load full pointer - unsupported
  3930  	{AMOVL, Yml, Ycol, movFullPtr, [4]uint8{0, 0, 0, 0}},
  3931  	{AMOVW, Yml, Ycol, movFullPtr, [4]uint8{Pe, 0, 0, 0}},
  3932  	*/
  3933  
  3934  	// double shift
  3935  	{ASHLL, Yi8, Yrl, Yml, movDoubleShift, [4]uint8{0xa4, 0xa5, 0, 0}},
  3936  	{ASHLL, Ycl, Yrl, Yml, movDoubleShift, [4]uint8{0xa4, 0xa5, 0, 0}},
  3937  	{ASHLL, Ycx, Yrl, Yml, movDoubleShift, [4]uint8{0xa4, 0xa5, 0, 0}},
  3938  	{ASHRL, Yi8, Yrl, Yml, movDoubleShift, [4]uint8{0xac, 0xad, 0, 0}},
  3939  	{ASHRL, Ycl, Yrl, Yml, movDoubleShift, [4]uint8{0xac, 0xad, 0, 0}},
  3940  	{ASHRL, Ycx, Yrl, Yml, movDoubleShift, [4]uint8{0xac, 0xad, 0, 0}},
  3941  	{ASHLQ, Yi8, Yrl, Yml, movDoubleShift, [4]uint8{Pw, 0xa4, 0xa5, 0}},
  3942  	{ASHLQ, Ycl, Yrl, Yml, movDoubleShift, [4]uint8{Pw, 0xa4, 0xa5, 0}},
  3943  	{ASHLQ, Ycx, Yrl, Yml, movDoubleShift, [4]uint8{Pw, 0xa4, 0xa5, 0}},
  3944  	{ASHRQ, Yi8, Yrl, Yml, movDoubleShift, [4]uint8{Pw, 0xac, 0xad, 0}},
  3945  	{ASHRQ, Ycl, Yrl, Yml, movDoubleShift, [4]uint8{Pw, 0xac, 0xad, 0}},
  3946  	{ASHRQ, Ycx, Yrl, Yml, movDoubleShift, [4]uint8{Pw, 0xac, 0xad, 0}},
  3947  	{ASHLW, Yi8, Yrl, Yml, movDoubleShift, [4]uint8{Pe, 0xa4, 0xa5, 0}},
  3948  	{ASHLW, Ycl, Yrl, Yml, movDoubleShift, [4]uint8{Pe, 0xa4, 0xa5, 0}},
  3949  	{ASHLW, Ycx, Yrl, Yml, movDoubleShift, [4]uint8{Pe, 0xa4, 0xa5, 0}},
  3950  	{ASHRW, Yi8, Yrl, Yml, movDoubleShift, [4]uint8{Pe, 0xac, 0xad, 0}},
  3951  	{ASHRW, Ycl, Yrl, Yml, movDoubleShift, [4]uint8{Pe, 0xac, 0xad, 0}},
  3952  	{ASHRW, Ycx, Yrl, Yml, movDoubleShift, [4]uint8{Pe, 0xac, 0xad, 0}},
  3953  
  3954  	// load TLS base
  3955  	{AMOVL, Ytls, Ynone, Yrl, movTLSReg, [4]uint8{0, 0, 0, 0}},
  3956  	{AMOVQ, Ytls, Ynone, Yrl, movTLSReg, [4]uint8{0, 0, 0, 0}},
  3957  	{0, 0, 0, 0, 0, [4]uint8{}},
  3958  }
  3959  
  3960  func isax(a *obj.Addr) bool {
  3961  	switch a.Reg {
  3962  	case REG_AX, REG_AL, REG_AH:
  3963  		return true
  3964  	}
  3965  
  3966  	return a.Index == REG_AX
  3967  }
  3968  
  3969  func subreg(p *obj.Prog, from int, to int) {
  3970  	if false { /* debug['Q'] */
  3971  		fmt.Printf("\n%v\ts/%v/%v/\n", p, rconv(from), rconv(to))
  3972  	}
  3973  
  3974  	if int(p.From.Reg) == from {
  3975  		p.From.Reg = int16(to)
  3976  		p.Ft = 0
  3977  	}
  3978  
  3979  	if int(p.To.Reg) == from {
  3980  		p.To.Reg = int16(to)
  3981  		p.Tt = 0
  3982  	}
  3983  
  3984  	if int(p.From.Index) == from {
  3985  		p.From.Index = int16(to)
  3986  		p.Ft = 0
  3987  	}
  3988  
  3989  	if int(p.To.Index) == from {
  3990  		p.To.Index = int16(to)
  3991  		p.Tt = 0
  3992  	}
  3993  
  3994  	if false { /* debug['Q'] */
  3995  		fmt.Printf("%v\n", p)
  3996  	}
  3997  }
  3998  
  3999  func (ab *AsmBuf) mediaop(ctxt *obj.Link, o *Optab, op int, osize int, z int) int {
  4000  	switch op {
  4001  	case Pm, Pe, Pf2, Pf3:
  4002  		if osize != 1 {
  4003  			if op != Pm {
  4004  				ab.Put1(byte(op))
  4005  			}
  4006  			ab.Put1(Pm)
  4007  			z++
  4008  			op = int(o.op[z])
  4009  			break
  4010  		}
  4011  		fallthrough
  4012  
  4013  	default:
  4014  		if ab.Len() == 0 || ab.Last() != Pm {
  4015  			ab.Put1(Pm)
  4016  		}
  4017  	}
  4018  
  4019  	ab.Put1(byte(op))
  4020  	return z
  4021  }
  4022  
  4023  // asmevex emits EVEX pregis and opcode byte.
  4024  // In addition to asmvex r/m, vvvv and reg fields also requires optional
  4025  // K-masking register.
  4026  //
  4027  // Expects asmbuf.evex to be properly initialized.
  4028  func (ab *AsmBuf) asmevex(ctxt *obj.Link, p *obj.Prog, rm, v, r, k *obj.Addr) {
  4029  	ab.evexflag = true
  4030  	evex := ab.evex
  4031  
  4032  	rexR := byte(1)
  4033  	evexR := byte(1)
  4034  	rexX := byte(1)
  4035  	rexB := byte(1)
  4036  	if r != nil {
  4037  		if regrex[r.Reg]&Rxr != 0 {
  4038  			rexR = 0 // "ModR/M.reg" selector 4th bit.
  4039  		}
  4040  		if regrex[r.Reg]&RxrEvex != 0 {
  4041  			evexR = 0 // "ModR/M.reg" selector 5th bit.
  4042  		}
  4043  	}
  4044  	if rm != nil {
  4045  		if rm.Index == REG_NONE && regrex[rm.Reg]&RxrEvex != 0 {
  4046  			rexX = 0
  4047  		} else if regrex[rm.Index]&Rxx != 0 {
  4048  			rexX = 0
  4049  		}
  4050  		if regrex[rm.Reg]&Rxb != 0 {
  4051  			rexB = 0
  4052  		}
  4053  	}
  4054  	// P0 = [R][X][B][R'][00][mm]
  4055  	p0 := (rexR << 7) |
  4056  		(rexX << 6) |
  4057  		(rexB << 5) |
  4058  		(evexR << 4) |
  4059  		(0 << 2) |
  4060  		(evex.M() << 0)
  4061  
  4062  	vexV := byte(0)
  4063  	if v != nil {
  4064  		// 4bit-wide reg index.
  4065  		vexV = byte(reg[v.Reg]|(regrex[v.Reg]&Rxr)<<1) & 0xF
  4066  	}
  4067  	vexV ^= 0x0F
  4068  	// P1 = [W][vvvv][1][pp]
  4069  	p1 := (evex.W() << 7) |
  4070  		(vexV << 3) |
  4071  		(1 << 2) |
  4072  		(evex.P() << 0)
  4073  
  4074  	suffix := evexSuffixMap[p.Scond]
  4075  	evexZ := byte(0)
  4076  	evexLL := evex.L()
  4077  	evexB := byte(0)
  4078  	evexV := byte(1)
  4079  	evexA := byte(0)
  4080  	if suffix.zeroing {
  4081  		if !evex.ZeroingEnabled() {
  4082  			ctxt.Diag("unsupported zeroing: %v", p)
  4083  		}
  4084  		if k == nil {
  4085  			// When you request zeroing you must specify a mask register.
  4086  			// See issue 57952.
  4087  			ctxt.Diag("mask register must be specified for .Z instructions: %v", p)
  4088  		} else if k.Reg == REG_K0 {
  4089  			// The mask register must not be K0. That restriction is already
  4090  			// handled by the Yknot0 restriction in the opcode tables, so we
  4091  			// won't ever reach here. But put something sensible here just in case.
  4092  			ctxt.Diag("mask register must not be K0 for .Z instructions: %v", p)
  4093  		}
  4094  		evexZ = 1
  4095  	}
  4096  	switch {
  4097  	case suffix.rounding != rcUnset:
  4098  		if rm != nil && rm.Type == obj.TYPE_MEM {
  4099  			ctxt.Diag("illegal rounding with memory argument: %v", p)
  4100  		} else if !evex.RoundingEnabled() {
  4101  			ctxt.Diag("unsupported rounding: %v", p)
  4102  		}
  4103  		evexB = 1
  4104  		evexLL = suffix.rounding
  4105  	case suffix.broadcast:
  4106  		if rm == nil || rm.Type != obj.TYPE_MEM {
  4107  			ctxt.Diag("illegal broadcast without memory argument: %v", p)
  4108  		} else if !evex.BroadcastEnabled() {
  4109  			ctxt.Diag("unsupported broadcast: %v", p)
  4110  		}
  4111  		evexB = 1
  4112  	case suffix.sae:
  4113  		if rm != nil && rm.Type == obj.TYPE_MEM {
  4114  			ctxt.Diag("illegal SAE with memory argument: %v", p)
  4115  		} else if !evex.SaeEnabled() {
  4116  			ctxt.Diag("unsupported SAE: %v", p)
  4117  		}
  4118  		evexB = 1
  4119  	}
  4120  	if rm != nil && regrex[rm.Index]&RxrEvex != 0 {
  4121  		evexV = 0
  4122  	} else if v != nil && regrex[v.Reg]&RxrEvex != 0 {
  4123  		evexV = 0 // VSR selector 5th bit.
  4124  	}
  4125  	if k != nil {
  4126  		evexA = byte(reg[k.Reg])
  4127  	}
  4128  	// P2 = [z][L'L][b][V'][aaa]
  4129  	p2 := (evexZ << 7) |
  4130  		(evexLL << 5) |
  4131  		(evexB << 4) |
  4132  		(evexV << 3) |
  4133  		(evexA << 0)
  4134  
  4135  	const evexEscapeByte = 0x62
  4136  	ab.Put4(evexEscapeByte, p0, p1, p2)
  4137  	ab.Put1(evex.opcode)
  4138  }
  4139  
  4140  // Emit VEX prefix and opcode byte.
  4141  // The three addresses are the r/m, vvvv, and reg fields.
  4142  // The reg and rm arguments appear in the same order as the
  4143  // arguments to asmand, which typically follows the call to asmvex.
  4144  // The final two arguments are the VEX prefix (see encoding above)
  4145  // and the opcode byte.
  4146  // For details about vex prefix see:
  4147  // https://en.wikipedia.org/wiki/VEX_prefix#Technical_description
  4148  func (ab *AsmBuf) asmvex(ctxt *obj.Link, rm, v, r *obj.Addr, vex, opcode uint8) {
  4149  	ab.vexflag = true
  4150  	rexR := 0
  4151  	if r != nil {
  4152  		rexR = regrex[r.Reg] & Rxr
  4153  	}
  4154  	rexB := 0
  4155  	rexX := 0
  4156  	if rm != nil {
  4157  		rexB = regrex[rm.Reg] & Rxb
  4158  		rexX = regrex[rm.Index] & Rxx
  4159  	}
  4160  	vexM := (vex >> 3) & 0x7
  4161  	vexWLP := vex & 0x87
  4162  	vexV := byte(0)
  4163  	if v != nil {
  4164  		vexV = byte(reg[v.Reg]|(regrex[v.Reg]&Rxr)<<1) & 0xF
  4165  	}
  4166  	vexV ^= 0xF
  4167  	if vexM == 1 && (rexX|rexB) == 0 && vex&vexW1 == 0 {
  4168  		// Can use 2-byte encoding.
  4169  		ab.Put2(0xc5, byte(rexR<<5)^0x80|vexV<<3|vexWLP)
  4170  	} else {
  4171  		// Must use 3-byte encoding.
  4172  		ab.Put3(0xc4,
  4173  			(byte(rexR|rexX|rexB)<<5)^0xE0|vexM,
  4174  			vexV<<3|vexWLP,
  4175  		)
  4176  	}
  4177  	ab.Put1(opcode)
  4178  }
  4179  
  4180  // regIndex returns register index that fits in 5 bits.
  4181  //
  4182  //	R         : 3 bit | legacy instructions     | N/A
  4183  //	[R/V]EX.R : 1 bit | REX / VEX extension bit | Rxr
  4184  //	EVEX.R    : 1 bit | EVEX extension bit      | RxrEvex
  4185  //
  4186  // Examples:
  4187  //
  4188  //	REG_Z30 => 30
  4189  //	REG_X15 => 15
  4190  //	REG_R9  => 9
  4191  //	REG_AX  => 0
  4192  func regIndex(r int16) int {
  4193  	lower3bits := reg[r]
  4194  	high4bit := regrex[r] & Rxr << 1
  4195  	high5bit := regrex[r] & RxrEvex << 0
  4196  	return lower3bits | high4bit | high5bit
  4197  }
  4198  
  4199  // avx2gatherValid reports whether p satisfies AVX2 gather constraints.
  4200  // Reports errors via ctxt.
  4201  func avx2gatherValid(ctxt *obj.Link, p *obj.Prog) bool {
  4202  	// If any pair of the index, mask, or destination registers
  4203  	// are the same, illegal instruction trap (#UD) is triggered.
  4204  	index := regIndex(p.GetFrom3().Index)
  4205  	mask := regIndex(p.From.Reg)
  4206  	dest := regIndex(p.To.Reg)
  4207  	if dest == mask || dest == index || mask == index {
  4208  		ctxt.Diag("mask, index, and destination registers should be distinct: %v", p)
  4209  		return false
  4210  	}
  4211  
  4212  	return true
  4213  }
  4214  
  4215  // avx512gatherValid reports whether p satisfies AVX512 gather constraints.
  4216  // Reports errors via ctxt.
  4217  func avx512gatherValid(ctxt *obj.Link, p *obj.Prog) bool {
  4218  	// Illegal instruction trap (#UD) is triggered if the destination vector
  4219  	// register is the same as index vector in VSIB.
  4220  	index := regIndex(p.From.Index)
  4221  	dest := regIndex(p.To.Reg)
  4222  	if dest == index {
  4223  		ctxt.Diag("index and destination registers should be distinct: %v", p)
  4224  		return false
  4225  	}
  4226  
  4227  	return true
  4228  }
  4229  
  4230  func (ab *AsmBuf) doasm(ctxt *obj.Link, cursym *obj.LSym, p *obj.Prog) {
  4231  	o := opindex[p.As&obj.AMask]
  4232  
  4233  	if o == nil {
  4234  		ctxt.Diag("asmins: missing op %v", p)
  4235  		return
  4236  	}
  4237  
  4238  	if pre := prefixof(ctxt, &p.From); pre != 0 {
  4239  		ab.Put1(byte(pre))
  4240  	}
  4241  	if pre := prefixof(ctxt, &p.To); pre != 0 {
  4242  		ab.Put1(byte(pre))
  4243  	}
  4244  
  4245  	// Checks to warn about instruction/arguments combinations that
  4246  	// will unconditionally trigger illegal instruction trap (#UD).
  4247  	switch p.As {
  4248  	case AVGATHERDPD,
  4249  		AVGATHERQPD,
  4250  		AVGATHERDPS,
  4251  		AVGATHERQPS,
  4252  		AVPGATHERDD,
  4253  		AVPGATHERQD,
  4254  		AVPGATHERDQ,
  4255  		AVPGATHERQQ:
  4256  		if p.GetFrom3() == nil {
  4257  			// gathers need a 3rd arg. See issue 58822.
  4258  			ctxt.Diag("need a third arg for gather instruction: %v", p)
  4259  			return
  4260  		}
  4261  		// AVX512 gather requires explicit K mask.
  4262  		if p.GetFrom3().Reg >= REG_K0 && p.GetFrom3().Reg <= REG_K7 {
  4263  			if !avx512gatherValid(ctxt, p) {
  4264  				return
  4265  			}
  4266  		} else {
  4267  			if !avx2gatherValid(ctxt, p) {
  4268  				return
  4269  			}
  4270  		}
  4271  	}
  4272  
  4273  	if p.Ft == 0 {
  4274  		p.Ft = uint8(oclass(ctxt, p, &p.From))
  4275  	}
  4276  	if p.Tt == 0 {
  4277  		p.Tt = uint8(oclass(ctxt, p, &p.To))
  4278  	}
  4279  
  4280  	ft := int(p.Ft) * Ymax
  4281  	tt := int(p.Tt) * Ymax
  4282  
  4283  	xo := obj.Bool2int(o.op[0] == 0x0f)
  4284  	z := 0
  4285  
  4286  	args := make([]int, 0, argListMax)
  4287  	if ft != Ynone*Ymax {
  4288  		args = append(args, ft)
  4289  	}
  4290  	for i := range p.RestArgs {
  4291  		args = append(args, oclass(ctxt, p, &p.RestArgs[i].Addr)*Ymax)
  4292  	}
  4293  	if tt != Ynone*Ymax {
  4294  		args = append(args, tt)
  4295  	}
  4296  
  4297  	var f3t int
  4298  	for _, yt := range o.ytab {
  4299  		// ytab matching is purely args-based,
  4300  		// but AVX512 suffixes like "Z" or "RU_SAE" will
  4301  		// add EVEX-only filter that will reject non-EVEX matches.
  4302  		//
  4303  		// Consider "VADDPD.BCST 2032(DX), X0, X0".
  4304  		// Without this rule, operands will lead to VEX-encoded form
  4305  		// and produce "c5b15813" encoding.
  4306  		if !yt.match(args) {
  4307  			// "xo" is always zero for VEX/EVEX encoded insts.
  4308  			z += int(yt.zoffset) + xo
  4309  		} else {
  4310  			if p.Scond != 0 && !evexZcase(yt.zcase) {
  4311  				// Do not signal error and continue to search
  4312  				// for matching EVEX-encoded form.
  4313  				z += int(yt.zoffset)
  4314  				continue
  4315  			}
  4316  
  4317  			switch o.prefix {
  4318  			case Px1: // first option valid only in 32-bit mode
  4319  				if ctxt.Arch.Family == sys.AMD64 && z == 0 {
  4320  					z += int(yt.zoffset) + xo
  4321  					continue
  4322  				}
  4323  			case Pq: // 16 bit escape and opcode escape
  4324  				ab.Put2(Pe, Pm)
  4325  
  4326  			case Pq3: // 16 bit escape and opcode escape + REX.W
  4327  				ab.rexflag |= Pw
  4328  				ab.Put2(Pe, Pm)
  4329  
  4330  			case Pq4: // 66 0F 38
  4331  				ab.Put3(0x66, 0x0F, 0x38)
  4332  
  4333  			case Pq4w: // 66 0F 38 + REX.W
  4334  				ab.rexflag |= Pw
  4335  				ab.Put3(0x66, 0x0F, 0x38)
  4336  
  4337  			case Pq5: // F3 0F 38
  4338  				ab.Put3(0xF3, 0x0F, 0x38)
  4339  
  4340  			case Pq5w: //  F3 0F 38 + REX.W
  4341  				ab.rexflag |= Pw
  4342  				ab.Put3(0xF3, 0x0F, 0x38)
  4343  
  4344  			case Pf2, // xmm opcode escape
  4345  				Pf3:
  4346  				ab.Put2(o.prefix, Pm)
  4347  
  4348  			case Pef3:
  4349  				ab.Put3(Pe, Pf3, Pm)
  4350  
  4351  			case Pfw: // xmm opcode escape + REX.W
  4352  				ab.rexflag |= Pw
  4353  				ab.Put2(Pf3, Pm)
  4354  
  4355  			case Pm: // opcode escape
  4356  				ab.Put1(Pm)
  4357  
  4358  			case Pe: // 16 bit escape
  4359  				ab.Put1(Pe)
  4360  
  4361  			case Pw: // 64-bit escape
  4362  				if ctxt.Arch.Family != sys.AMD64 {
  4363  					ctxt.Diag("asmins: illegal 64: %v", p)
  4364  				}
  4365  				ab.rexflag |= Pw
  4366  
  4367  			case Pw8: // 64-bit escape if z >= 8
  4368  				if z >= 8 {
  4369  					if ctxt.Arch.Family != sys.AMD64 {
  4370  						ctxt.Diag("asmins: illegal 64: %v", p)
  4371  					}
  4372  					ab.rexflag |= Pw
  4373  				}
  4374  
  4375  			case Pb: // botch
  4376  				if ctxt.Arch.Family != sys.AMD64 && (isbadbyte(&p.From) || isbadbyte(&p.To)) {
  4377  					goto bad
  4378  				}
  4379  				// NOTE(rsc): This is probably safe to do always,
  4380  				// but when enabled it chooses different encodings
  4381  				// than the old cmd/internal/obj/i386 code did,
  4382  				// which breaks our "same bits out" checks.
  4383  				// In particular, CMPB AX, $0 encodes as 80 f8 00
  4384  				// in the original obj/i386, and it would encode
  4385  				// (using a valid, shorter form) as 3c 00 if we enabled
  4386  				// the call to bytereg here.
  4387  				if ctxt.Arch.Family == sys.AMD64 {
  4388  					bytereg(&p.From, &p.Ft)
  4389  					bytereg(&p.To, &p.Tt)
  4390  				}
  4391  
  4392  			case P32: // 32 bit but illegal if 64-bit mode
  4393  				if ctxt.Arch.Family == sys.AMD64 {
  4394  					ctxt.Diag("asmins: illegal in 64-bit mode: %v", p)
  4395  				}
  4396  
  4397  			case Py: // 64-bit only, no prefix
  4398  				if ctxt.Arch.Family != sys.AMD64 {
  4399  					ctxt.Diag("asmins: illegal in %d-bit mode: %v", ctxt.Arch.RegSize*8, p)
  4400  				}
  4401  
  4402  			case Py1: // 64-bit only if z < 1, no prefix
  4403  				if z < 1 && ctxt.Arch.Family != sys.AMD64 {
  4404  					ctxt.Diag("asmins: illegal in %d-bit mode: %v", ctxt.Arch.RegSize*8, p)
  4405  				}
  4406  
  4407  			case Py3: // 64-bit only if z < 3, no prefix
  4408  				if z < 3 && ctxt.Arch.Family != sys.AMD64 {
  4409  					ctxt.Diag("asmins: illegal in %d-bit mode: %v", ctxt.Arch.RegSize*8, p)
  4410  				}
  4411  			}
  4412  
  4413  			if z >= len(o.op) {
  4414  				log.Fatalf("asmins bad table %v", p)
  4415  			}
  4416  			op := int(o.op[z])
  4417  			if op == 0x0f {
  4418  				ab.Put1(byte(op))
  4419  				z++
  4420  				op = int(o.op[z])
  4421  			}
  4422  
  4423  			switch yt.zcase {
  4424  			default:
  4425  				ctxt.Diag("asmins: unknown z %d %v", yt.zcase, p)
  4426  				return
  4427  
  4428  			case Zpseudo:
  4429  				break
  4430  
  4431  			case Zlit:
  4432  				ab.PutOpBytesLit(z, &o.op)
  4433  
  4434  			case Zlitr_m:
  4435  				ab.PutOpBytesLit(z, &o.op)
  4436  				ab.asmand(ctxt, cursym, p, &p.To, &p.From)
  4437  
  4438  			case Zlitm_r:
  4439  				ab.PutOpBytesLit(z, &o.op)
  4440  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4441  
  4442  			case Zlit_m_r:
  4443  				ab.PutOpBytesLit(z, &o.op)
  4444  				ab.asmand(ctxt, cursym, p, p.GetFrom3(), &p.To)
  4445  
  4446  			case Zmb_r:
  4447  				bytereg(&p.From, &p.Ft)
  4448  				fallthrough
  4449  
  4450  			case Zm_r:
  4451  				ab.Put1(byte(op))
  4452  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4453  
  4454  			case Z_m_r:
  4455  				ab.Put1(byte(op))
  4456  				ab.asmand(ctxt, cursym, p, p.GetFrom3(), &p.To)
  4457  
  4458  			case Zm2_r:
  4459  				ab.Put2(byte(op), o.op[z+1])
  4460  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4461  
  4462  			case Zm_r_xm:
  4463  				ab.mediaop(ctxt, o, op, int(yt.zoffset), z)
  4464  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4465  
  4466  			case Zm_r_xm_nr:
  4467  				ab.rexflag = 0
  4468  				ab.mediaop(ctxt, o, op, int(yt.zoffset), z)
  4469  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4470  
  4471  			case Zm_r_i_xm:
  4472  				ab.mediaop(ctxt, o, op, int(yt.zoffset), z)
  4473  				ab.asmand(ctxt, cursym, p, &p.From, p.GetFrom3())
  4474  				ab.Put1(byte(p.To.Offset))
  4475  
  4476  			case Zibm_r, Zibr_m:
  4477  				ab.PutOpBytesLit(z, &o.op)
  4478  				if yt.zcase == Zibr_m {
  4479  					ab.asmand(ctxt, cursym, p, &p.To, p.GetFrom3())
  4480  				} else {
  4481  					ab.asmand(ctxt, cursym, p, p.GetFrom3(), &p.To)
  4482  				}
  4483  				switch {
  4484  				default:
  4485  					ab.Put1(byte(p.From.Offset))
  4486  				case yt.args[0] == Yi32 && o.prefix == Pe:
  4487  					ab.PutInt16(int16(p.From.Offset))
  4488  				case yt.args[0] == Yi32:
  4489  					ab.PutInt32(int32(p.From.Offset))
  4490  				}
  4491  
  4492  			case Zaut_r:
  4493  				ab.Put1(0x8d) // leal
  4494  				if p.From.Type != obj.TYPE_ADDR {
  4495  					ctxt.Diag("asmins: Zaut sb type ADDR")
  4496  				}
  4497  				p.From.Type = obj.TYPE_MEM
  4498  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4499  				p.From.Type = obj.TYPE_ADDR
  4500  
  4501  			case Zm_o:
  4502  				ab.Put1(byte(op))
  4503  				ab.asmando(ctxt, cursym, p, &p.From, int(o.op[z+1]))
  4504  
  4505  			case Zr_m:
  4506  				ab.Put1(byte(op))
  4507  				ab.asmand(ctxt, cursym, p, &p.To, &p.From)
  4508  
  4509  			case Zvex:
  4510  				ab.asmvex(ctxt, &p.From, p.GetFrom3(), &p.To, o.op[z], o.op[z+1])
  4511  
  4512  			case Zvex_rm_v_r:
  4513  				ab.asmvex(ctxt, &p.From, p.GetFrom3(), &p.To, o.op[z], o.op[z+1])
  4514  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4515  
  4516  			case Zvex_rm_v_ro:
  4517  				ab.asmvex(ctxt, &p.From, p.GetFrom3(), &p.To, o.op[z], o.op[z+1])
  4518  				ab.asmando(ctxt, cursym, p, &p.From, int(o.op[z+2]))
  4519  
  4520  			case Zvex_i_rm_vo:
  4521  				ab.asmvex(ctxt, p.GetFrom3(), &p.To, nil, o.op[z], o.op[z+1])
  4522  				ab.asmando(ctxt, cursym, p, p.GetFrom3(), int(o.op[z+2]))
  4523  				ab.Put1(byte(p.From.Offset))
  4524  
  4525  			case Zvex_i_r_v:
  4526  				ab.asmvex(ctxt, p.GetFrom3(), &p.To, nil, o.op[z], o.op[z+1])
  4527  				regnum := byte(0x7)
  4528  				if p.GetFrom3().Reg >= REG_X0 && p.GetFrom3().Reg <= REG_X15 {
  4529  					regnum &= byte(p.GetFrom3().Reg - REG_X0)
  4530  				} else {
  4531  					regnum &= byte(p.GetFrom3().Reg - REG_Y0)
  4532  				}
  4533  				ab.Put1(o.op[z+2] | regnum)
  4534  				ab.Put1(byte(p.From.Offset))
  4535  
  4536  			case Zvex_i_rm_v_r:
  4537  				imm, from, from3, to := unpackOps4(p)
  4538  				ab.asmvex(ctxt, from, from3, to, o.op[z], o.op[z+1])
  4539  				ab.asmand(ctxt, cursym, p, from, to)
  4540  				ab.Put1(byte(imm.Offset))
  4541  
  4542  			case Zvex_i_rm_r:
  4543  				ab.asmvex(ctxt, p.GetFrom3(), nil, &p.To, o.op[z], o.op[z+1])
  4544  				ab.asmand(ctxt, cursym, p, p.GetFrom3(), &p.To)
  4545  				ab.Put1(byte(p.From.Offset))
  4546  
  4547  			case Zvex_v_rm_r:
  4548  				ab.asmvex(ctxt, p.GetFrom3(), &p.From, &p.To, o.op[z], o.op[z+1])
  4549  				ab.asmand(ctxt, cursym, p, p.GetFrom3(), &p.To)
  4550  
  4551  			case Zvex_r_v_rm:
  4552  				ab.asmvex(ctxt, &p.To, p.GetFrom3(), &p.From, o.op[z], o.op[z+1])
  4553  				ab.asmand(ctxt, cursym, p, &p.To, &p.From)
  4554  
  4555  			case Zvex_rm_r_vo:
  4556  				ab.asmvex(ctxt, &p.From, &p.To, p.GetFrom3(), o.op[z], o.op[z+1])
  4557  				ab.asmando(ctxt, cursym, p, &p.From, int(o.op[z+2]))
  4558  
  4559  			case Zvex_i_r_rm:
  4560  				ab.asmvex(ctxt, &p.To, nil, p.GetFrom3(), o.op[z], o.op[z+1])
  4561  				ab.asmand(ctxt, cursym, p, &p.To, p.GetFrom3())
  4562  				ab.Put1(byte(p.From.Offset))
  4563  
  4564  			case Zvex_hr_rm_v_r:
  4565  				hr, from, from3, to := unpackOps4(p)
  4566  				ab.asmvex(ctxt, from, from3, to, o.op[z], o.op[z+1])
  4567  				ab.asmand(ctxt, cursym, p, from, to)
  4568  				ab.Put1(byte(regIndex(hr.Reg) << 4))
  4569  
  4570  			case Zevex_k_rmo:
  4571  				ab.evex = newEVEXBits(z, &o.op)
  4572  				ab.asmevex(ctxt, p, &p.To, nil, nil, &p.From)
  4573  				ab.asmando(ctxt, cursym, p, &p.To, int(o.op[z+3]))
  4574  
  4575  			case Zevex_i_rm_vo:
  4576  				ab.evex = newEVEXBits(z, &o.op)
  4577  				ab.asmevex(ctxt, p, p.GetFrom3(), &p.To, nil, nil)
  4578  				ab.asmando(ctxt, cursym, p, p.GetFrom3(), int(o.op[z+3]))
  4579  				ab.Put1(byte(p.From.Offset))
  4580  
  4581  			case Zevex_i_rm_k_vo:
  4582  				imm, from, kmask, to := unpackOps4(p)
  4583  				ab.evex = newEVEXBits(z, &o.op)
  4584  				ab.asmevex(ctxt, p, from, to, nil, kmask)
  4585  				ab.asmando(ctxt, cursym, p, from, int(o.op[z+3]))
  4586  				ab.Put1(byte(imm.Offset))
  4587  
  4588  			case Zevex_i_r_rm:
  4589  				ab.evex = newEVEXBits(z, &o.op)
  4590  				ab.asmevex(ctxt, p, &p.To, nil, p.GetFrom3(), nil)
  4591  				ab.asmand(ctxt, cursym, p, &p.To, p.GetFrom3())
  4592  				ab.Put1(byte(p.From.Offset))
  4593  
  4594  			case Zevex_i_r_k_rm:
  4595  				imm, from, kmask, to := unpackOps4(p)
  4596  				ab.evex = newEVEXBits(z, &o.op)
  4597  				ab.asmevex(ctxt, p, to, nil, from, kmask)
  4598  				ab.asmand(ctxt, cursym, p, to, from)
  4599  				ab.Put1(byte(imm.Offset))
  4600  
  4601  			case Zevex_i_rm_r:
  4602  				ab.evex = newEVEXBits(z, &o.op)
  4603  				ab.asmevex(ctxt, p, p.GetFrom3(), nil, &p.To, nil)
  4604  				ab.asmand(ctxt, cursym, p, p.GetFrom3(), &p.To)
  4605  				ab.Put1(byte(p.From.Offset))
  4606  
  4607  			case Zevex_i_rm_k_r:
  4608  				imm, from, kmask, to := unpackOps4(p)
  4609  				ab.evex = newEVEXBits(z, &o.op)
  4610  				ab.asmevex(ctxt, p, from, nil, to, kmask)
  4611  				ab.asmand(ctxt, cursym, p, from, to)
  4612  				ab.Put1(byte(imm.Offset))
  4613  
  4614  			case Zevex_i_rm_v_r:
  4615  				imm, from, from3, to := unpackOps4(p)
  4616  				ab.evex = newEVEXBits(z, &o.op)
  4617  				ab.asmevex(ctxt, p, from, from3, to, nil)
  4618  				ab.asmand(ctxt, cursym, p, from, to)
  4619  				ab.Put1(byte(imm.Offset))
  4620  
  4621  			case Zevex_i_rm_v_k_r:
  4622  				imm, from, from3, kmask, to := unpackOps5(p)
  4623  				ab.evex = newEVEXBits(z, &o.op)
  4624  				ab.asmevex(ctxt, p, from, from3, to, kmask)
  4625  				ab.asmand(ctxt, cursym, p, from, to)
  4626  				ab.Put1(byte(imm.Offset))
  4627  
  4628  			case Zevex_r_v_rm:
  4629  				ab.evex = newEVEXBits(z, &o.op)
  4630  				ab.asmevex(ctxt, p, &p.To, p.GetFrom3(), &p.From, nil)
  4631  				ab.asmand(ctxt, cursym, p, &p.To, &p.From)
  4632  
  4633  			case Zevex_rm_v_r:
  4634  				ab.evex = newEVEXBits(z, &o.op)
  4635  				ab.asmevex(ctxt, p, &p.From, p.GetFrom3(), &p.To, nil)
  4636  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4637  
  4638  			case Zevex_rm_k_r:
  4639  				ab.evex = newEVEXBits(z, &o.op)
  4640  				ab.asmevex(ctxt, p, &p.From, nil, &p.To, p.GetFrom3())
  4641  				ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  4642  
  4643  			case Zevex_r_k_rm:
  4644  				ab.evex = newEVEXBits(z, &o.op)
  4645  				ab.asmevex(ctxt, p, &p.To, nil, &p.From, p.GetFrom3())
  4646  				ab.asmand(ctxt, cursym, p, &p.To, &p.From)
  4647  
  4648  			case Zevex_rm_v_k_r:
  4649  				from, from3, kmask, to := unpackOps4(p)
  4650  				ab.evex = newEVEXBits(z, &o.op)
  4651  				ab.asmevex(ctxt, p, from, from3, to, kmask)
  4652  				ab.asmand(ctxt, cursym, p, from, to)
  4653  
  4654  			case Zevex_r_v_k_rm:
  4655  				from, from3, kmask, to := unpackOps4(p)
  4656  				ab.evex = newEVEXBits(z, &o.op)
  4657  				ab.asmevex(ctxt, p, to, from3, from, kmask)
  4658  				ab.asmand(ctxt, cursym, p, to, from)
  4659  
  4660  			case Zr_m_xm:
  4661  				ab.mediaop(ctxt, o, op, int(yt.zoffset), z)
  4662  				ab.asmand(ctxt, cursym, p, &p.To, &p.From)
  4663  
  4664  			case Zr_m_xm_nr:
  4665  				ab.rexflag = 0
  4666  				ab.mediaop(ctxt, o, op, int(yt.zoffset), z)
  4667  				ab.asmand(ctxt, cursym, p, &p.To, &p.From)
  4668  
  4669  			case Zo_m:
  4670  				ab.Put1(byte(op))
  4671  				ab.asmando(ctxt, cursym, p, &p.To, int(o.op[z+1]))
  4672  
  4673  			case Zcallindreg:
  4674  				cursym.AddRel(ctxt, obj.Reloc{
  4675  					Type: objabi.R_CALLIND,
  4676  					Off:  int32(p.Pc),
  4677  				})
  4678  				fallthrough
  4679  
  4680  			case Zo_m64:
  4681  				ab.Put1(byte(op))
  4682  				ab.asmandsz(ctxt, cursym, p, &p.To, int(o.op[z+1]), 0, 1)
  4683  
  4684  			case Zm_ibo:
  4685  				ab.Put1(byte(op))
  4686  				ab.asmando(ctxt, cursym, p, &p.From, int(o.op[z+1]))
  4687  				ab.Put1(byte(vaddr(ctxt, p, &p.To, nil)))
  4688  
  4689  			case Zibo_m:
  4690  				ab.Put1(byte(op))
  4691  				ab.asmando(ctxt, cursym, p, &p.To, int(o.op[z+1]))
  4692  				ab.Put1(byte(vaddr(ctxt, p, &p.From, nil)))
  4693  
  4694  			case Zibo_m_xm:
  4695  				z = ab.mediaop(ctxt, o, op, int(yt.zoffset), z)
  4696  				ab.asmando(ctxt, cursym, p, &p.To, int(o.op[z+1]))
  4697  				ab.Put1(byte(vaddr(ctxt, p, &p.From, nil)))
  4698  
  4699  			case Z_ib, Zib_:
  4700  				var a *obj.Addr
  4701  				if yt.zcase == Zib_ {
  4702  					a = &p.From
  4703  				} else {
  4704  					a = &p.To
  4705  				}
  4706  				ab.Put1(byte(op))
  4707  				if p.As == AXABORT {
  4708  					ab.Put1(o.op[z+1])
  4709  				}
  4710  				ab.Put1(byte(vaddr(ctxt, p, a, nil)))
  4711  
  4712  			case Zib_rp:
  4713  				ab.rexflag |= regrex[p.To.Reg] & (Rxb | 0x40)
  4714  				ab.Put2(byte(op+reg[p.To.Reg]), byte(vaddr(ctxt, p, &p.From, nil)))
  4715  
  4716  			case Zil_rp:
  4717  				ab.rexflag |= regrex[p.To.Reg] & Rxb
  4718  				ab.Put1(byte(op + reg[p.To.Reg]))
  4719  				if o.prefix == Pe {
  4720  					v := vaddr(ctxt, p, &p.From, nil)
  4721  					ab.PutInt16(int16(v))
  4722  				} else {
  4723  					ab.relput4(ctxt, cursym, p, &p.From)
  4724  				}
  4725  
  4726  			case Zo_iw:
  4727  				ab.Put1(byte(op))
  4728  				if p.From.Type != obj.TYPE_NONE {
  4729  					v := vaddr(ctxt, p, &p.From, nil)
  4730  					ab.PutInt16(int16(v))
  4731  				}
  4732  
  4733  			case Ziq_rp:
  4734  				var rel obj.Reloc
  4735  				v := vaddr(ctxt, p, &p.From, &rel)
  4736  				l := int(v >> 32)
  4737  				if l == 0 && rel.Siz != 8 {
  4738  					ab.rexflag &^= (0x40 | Rxw)
  4739  
  4740  					ab.rexflag |= regrex[p.To.Reg] & Rxb
  4741  					ab.Put1(byte(0xb8 + reg[p.To.Reg]))
  4742  					if rel.Type != 0 {
  4743  						rel.Off = int32(p.Pc + int64(ab.Len()))
  4744  						cursym.AddRel(ctxt, rel)
  4745  					}
  4746  
  4747  					ab.PutInt32(int32(v))
  4748  				} else if l == -1 && uint64(v)&(uint64(1)<<31) != 0 { // sign extend
  4749  					ab.Put1(0xc7)
  4750  					ab.asmando(ctxt, cursym, p, &p.To, 0)
  4751  
  4752  					ab.PutInt32(int32(v)) // need all 8
  4753  				} else {
  4754  					ab.rexflag |= regrex[p.To.Reg] & Rxb
  4755  					ab.Put1(byte(op + reg[p.To.Reg]))
  4756  					if rel.Type != 0 {
  4757  						rel.Off = int32(p.Pc + int64(ab.Len()))
  4758  						cursym.AddRel(ctxt, rel)
  4759  					}
  4760  
  4761  					ab.PutInt64(v)
  4762  				}
  4763  
  4764  			case Zib_rr:
  4765  				ab.Put1(byte(op))
  4766  				ab.asmand(ctxt, cursym, p, &p.To, &p.To)
  4767  				ab.Put1(byte(vaddr(ctxt, p, &p.From, nil)))
  4768  
  4769  			case Z_il, Zil_:
  4770  				var a *obj.Addr
  4771  				if yt.zcase == Zil_ {
  4772  					a = &p.From
  4773  				} else {
  4774  					a = &p.To
  4775  				}
  4776  				ab.Put1(byte(op))
  4777  				if o.prefix == Pe {
  4778  					v := vaddr(ctxt, p, a, nil)
  4779  					ab.PutInt16(int16(v))
  4780  				} else {
  4781  					ab.relput4(ctxt, cursym, p, a)
  4782  				}
  4783  
  4784  			case Zm_ilo, Zilo_m:
  4785  				var a *obj.Addr
  4786  				ab.Put1(byte(op))
  4787  				if yt.zcase == Zilo_m {
  4788  					a = &p.From
  4789  					ab.asmando(ctxt, cursym, p, &p.To, int(o.op[z+1]))
  4790  				} else {
  4791  					a = &p.To
  4792  					ab.asmando(ctxt, cursym, p, &p.From, int(o.op[z+1]))
  4793  				}
  4794  
  4795  				if o.prefix == Pe {
  4796  					v := vaddr(ctxt, p, a, nil)
  4797  					ab.PutInt16(int16(v))
  4798  				} else {
  4799  					ab.relput4(ctxt, cursym, p, a)
  4800  				}
  4801  
  4802  			case Zil_rr:
  4803  				ab.Put1(byte(op))
  4804  				ab.asmand(ctxt, cursym, p, &p.To, &p.To)
  4805  				if o.prefix == Pe {
  4806  					v := vaddr(ctxt, p, &p.From, nil)
  4807  					ab.PutInt16(int16(v))
  4808  				} else {
  4809  					ab.relput4(ctxt, cursym, p, &p.From)
  4810  				}
  4811  
  4812  			case Z_rp:
  4813  				ab.rexflag |= regrex[p.To.Reg] & (Rxb | 0x40)
  4814  				ab.Put1(byte(op + reg[p.To.Reg]))
  4815  
  4816  			case Zrp_:
  4817  				ab.rexflag |= regrex[p.From.Reg] & (Rxb | 0x40)
  4818  				ab.Put1(byte(op + reg[p.From.Reg]))
  4819  
  4820  			case Zcallcon, Zjmpcon:
  4821  				if yt.zcase == Zcallcon {
  4822  					ab.Put1(byte(op))
  4823  				} else {
  4824  					ab.Put1(o.op[z+1])
  4825  				}
  4826  				cursym.AddRel(ctxt, obj.Reloc{
  4827  					Type: objabi.R_PCREL,
  4828  					Off:  int32(p.Pc + int64(ab.Len())),
  4829  					Siz:  4,
  4830  					Add:  p.To.Offset,
  4831  				})
  4832  				ab.PutInt32(0)
  4833  
  4834  			case Zcallind:
  4835  				ab.Put2(byte(op), o.op[z+1])
  4836  				typ := objabi.R_ADDR
  4837  				if ctxt.Arch.Family == sys.AMD64 {
  4838  					typ = objabi.R_PCREL
  4839  				}
  4840  				cursym.AddRel(ctxt, obj.Reloc{
  4841  					Type: typ,
  4842  					Off:  int32(p.Pc + int64(ab.Len())),
  4843  					Siz:  4,
  4844  					Sym:  p.To.Sym,
  4845  					Add:  p.To.Offset,
  4846  				})
  4847  				ab.PutInt32(0)
  4848  
  4849  			case Zcall, Zcallduff:
  4850  				if p.To.Sym == nil {
  4851  					ctxt.Diag("call without target")
  4852  					ctxt.DiagFlush()
  4853  					log.Fatalf("bad code")
  4854  				}
  4855  
  4856  				if yt.zcase == Zcallduff && ctxt.Flag_dynlink {
  4857  					ctxt.Diag("directly calling duff when dynamically linking Go")
  4858  				}
  4859  
  4860  				ab.Put1(byte(op))
  4861  				cursym.AddRel(ctxt, obj.Reloc{
  4862  					Type: objabi.R_CALL,
  4863  					Off:  int32(p.Pc + int64(ab.Len())),
  4864  					Siz:  4,
  4865  					Sym:  p.To.Sym,
  4866  					Add:  p.To.Offset,
  4867  				})
  4868  				ab.PutInt32(0)
  4869  
  4870  			// TODO: jump across functions needs reloc
  4871  			case Zbr, Zjmp, Zloop:
  4872  				if p.As == AXBEGIN {
  4873  					ab.Put1(byte(op))
  4874  				}
  4875  				if p.To.Sym != nil {
  4876  					if yt.zcase != Zjmp {
  4877  						ctxt.Diag("branch to ATEXT")
  4878  						ctxt.DiagFlush()
  4879  						log.Fatalf("bad code")
  4880  					}
  4881  
  4882  					ab.Put1(o.op[z+1])
  4883  					cursym.AddRel(ctxt, obj.Reloc{
  4884  						// Note: R_CALL instead of R_PCREL. R_CALL is more permissive in that
  4885  						// it can point to a trampoline instead of the destination itself.
  4886  						Type: objabi.R_CALL,
  4887  						Off:  int32(p.Pc + int64(ab.Len())),
  4888  						Siz:  4,
  4889  						Sym:  p.To.Sym,
  4890  					})
  4891  					ab.PutInt32(0)
  4892  					break
  4893  				}
  4894  
  4895  				// Assumes q is in this function.
  4896  				// TODO: Check in input, preserve in brchain.
  4897  
  4898  				// Fill in backward jump now.
  4899  				q := p.To.Target()
  4900  
  4901  				if q == nil {
  4902  					ctxt.Diag("jmp/branch/loop without target")
  4903  					ctxt.DiagFlush()
  4904  					log.Fatalf("bad code")
  4905  				}
  4906  
  4907  				if p.Back&branchBackwards != 0 {
  4908  					v := q.Pc - (p.Pc + 2)
  4909  					if v >= -128 && p.As != AXBEGIN {
  4910  						if p.As == AJCXZL {
  4911  							ab.Put1(0x67)
  4912  						}
  4913  						ab.Put2(byte(op), byte(v))
  4914  					} else if yt.zcase == Zloop {
  4915  						ctxt.Diag("loop too far: %v", p)
  4916  					} else {
  4917  						v -= 5 - 2
  4918  						if p.As == AXBEGIN {
  4919  							v--
  4920  						}
  4921  						if yt.zcase == Zbr {
  4922  							ab.Put1(0x0f)
  4923  							v--
  4924  						}
  4925  
  4926  						ab.Put1(o.op[z+1])
  4927  						ab.PutInt32(int32(v))
  4928  					}
  4929  
  4930  					break
  4931  				}
  4932  
  4933  				// Annotate target; will fill in later.
  4934  				p.Forwd = q.Rel
  4935  
  4936  				q.Rel = p
  4937  				if p.Back&branchShort != 0 && p.As != AXBEGIN {
  4938  					if p.As == AJCXZL {
  4939  						ab.Put1(0x67)
  4940  					}
  4941  					ab.Put2(byte(op), 0)
  4942  				} else if yt.zcase == Zloop {
  4943  					ctxt.Diag("loop too far: %v", p)
  4944  				} else {
  4945  					if yt.zcase == Zbr {
  4946  						ab.Put1(0x0f)
  4947  					}
  4948  					ab.Put1(o.op[z+1])
  4949  					ab.PutInt32(0)
  4950  				}
  4951  
  4952  			case Zbyte:
  4953  				var rel obj.Reloc
  4954  				v := vaddr(ctxt, p, &p.From, &rel)
  4955  				if rel.Siz != 0 {
  4956  					rel.Siz = uint8(op)
  4957  					rel.Off = int32(p.Pc + int64(ab.Len()))
  4958  					cursym.AddRel(ctxt, rel)
  4959  				}
  4960  
  4961  				ab.Put1(byte(v))
  4962  				if op > 1 {
  4963  					ab.Put1(byte(v >> 8))
  4964  					if op > 2 {
  4965  						ab.PutInt16(int16(v >> 16))
  4966  						if op > 4 {
  4967  							ab.PutInt32(int32(v >> 32))
  4968  						}
  4969  					}
  4970  				}
  4971  			}
  4972  
  4973  			return
  4974  		}
  4975  	}
  4976  	f3t = Ynone * Ymax
  4977  	if p.GetFrom3() != nil {
  4978  		f3t = oclass(ctxt, p, p.GetFrom3()) * Ymax
  4979  	}
  4980  	for mo := ymovtab; mo[0].as != 0; mo = mo[1:] {
  4981  		var pp obj.Prog
  4982  		var t []byte
  4983  		if p.As == mo[0].as {
  4984  			if ycover[ft+int(mo[0].ft)] != 0 && ycover[f3t+int(mo[0].f3t)] != 0 && ycover[tt+int(mo[0].tt)] != 0 {
  4985  				t = mo[0].op[:]
  4986  				switch mo[0].code {
  4987  				default:
  4988  					ctxt.Diag("asmins: unknown mov %d %v", mo[0].code, p)
  4989  
  4990  				case movLit:
  4991  					for z = 0; t[z] != 0; z++ {
  4992  						ab.Put1(t[z])
  4993  					}
  4994  
  4995  				case movRegMem:
  4996  					ab.Put1(t[0])
  4997  					ab.asmando(ctxt, cursym, p, &p.To, int(t[1]))
  4998  
  4999  				case movMemReg:
  5000  					ab.Put1(t[0])
  5001  					ab.asmando(ctxt, cursym, p, &p.From, int(t[1]))
  5002  
  5003  				case movRegMem2op: // r,m - 2op
  5004  					ab.Put2(t[0], t[1])
  5005  					ab.asmando(ctxt, cursym, p, &p.To, int(t[2]))
  5006  					ab.rexflag |= regrex[p.From.Reg] & (Rxr | 0x40)
  5007  
  5008  				case movMemReg2op:
  5009  					ab.Put2(t[0], t[1])
  5010  					ab.asmando(ctxt, cursym, p, &p.From, int(t[2]))
  5011  					ab.rexflag |= regrex[p.To.Reg] & (Rxr | 0x40)
  5012  
  5013  				case movFullPtr:
  5014  					if t[0] != 0 {
  5015  						ab.Put1(t[0])
  5016  					}
  5017  					switch p.To.Index {
  5018  					default:
  5019  						goto bad
  5020  
  5021  					case REG_DS:
  5022  						ab.Put1(0xc5)
  5023  
  5024  					case REG_SS:
  5025  						ab.Put2(0x0f, 0xb2)
  5026  
  5027  					case REG_ES:
  5028  						ab.Put1(0xc4)
  5029  
  5030  					case REG_FS:
  5031  						ab.Put2(0x0f, 0xb4)
  5032  
  5033  					case REG_GS:
  5034  						ab.Put2(0x0f, 0xb5)
  5035  					}
  5036  
  5037  					ab.asmand(ctxt, cursym, p, &p.From, &p.To)
  5038  
  5039  				case movDoubleShift:
  5040  					if t[0] == Pw {
  5041  						if ctxt.Arch.Family != sys.AMD64 {
  5042  							ctxt.Diag("asmins: illegal 64: %v", p)
  5043  						}
  5044  						ab.rexflag |= Pw
  5045  						t = t[1:]
  5046  					} else if t[0] == Pe {
  5047  						ab.Put1(Pe)
  5048  						t = t[1:]
  5049  					}
  5050  
  5051  					switch p.From.Type {
  5052  					default:
  5053  						goto bad
  5054  
  5055  					case obj.TYPE_CONST:
  5056  						ab.Put2(0x0f, t[0])
  5057  						ab.asmandsz(ctxt, cursym, p, &p.To, reg[p.GetFrom3().Reg], regrex[p.GetFrom3().Reg], 0)
  5058  						ab.Put1(byte(p.From.Offset))
  5059  
  5060  					case obj.TYPE_REG:
  5061  						switch p.From.Reg {
  5062  						default:
  5063  							goto bad
  5064  
  5065  						case REG_CL, REG_CX:
  5066  							ab.Put2(0x0f, t[1])
  5067  							ab.asmandsz(ctxt, cursym, p, &p.To, reg[p.GetFrom3().Reg], regrex[p.GetFrom3().Reg], 0)
  5068  						}
  5069  					}
  5070  
  5071  				// NOTE: The systems listed here are the ones that use the "TLS initial exec" model,
  5072  				// where you load the TLS base register into a register and then index off that
  5073  				// register to access the actual TLS variables. Systems that allow direct TLS access
  5074  				// are handled in prefixof above and should not be listed here.
  5075  				case movTLSReg:
  5076  					if ctxt.Arch.Family == sys.AMD64 && p.As != AMOVQ || ctxt.Arch.Family == sys.I386 && p.As != AMOVL {
  5077  						ctxt.Diag("invalid load of TLS: %v", p)
  5078  					}
  5079  
  5080  					if ctxt.Arch.Family == sys.I386 {
  5081  						// NOTE: The systems listed here are the ones that use the "TLS initial exec" model,
  5082  						// where you load the TLS base register into a register and then index off that
  5083  						// register to access the actual TLS variables. Systems that allow direct TLS access
  5084  						// are handled in prefixof above and should not be listed here.
  5085  						switch ctxt.Headtype {
  5086  						default:
  5087  							log.Fatalf("unknown TLS base location for %v", ctxt.Headtype)
  5088  
  5089  						case objabi.Hlinux, objabi.Hfreebsd:
  5090  							if ctxt.Flag_shared {
  5091  								// Note that this is not generating the same insns as the other cases.
  5092  								//     MOV TLS, dst
  5093  								// becomes
  5094  								//     call __x86.get_pc_thunk.dst
  5095  								//     movl (gotpc + g@gotntpoff)(dst), dst
  5096  								// which is encoded as
  5097  								//     call __x86.get_pc_thunk.dst
  5098  								//     movq 0(dst), dst
  5099  								// and R_CALL & R_TLS_IE relocs. This all assumes the only tls variable we access
  5100  								// is g, which we can't check here, but will when we assemble the second
  5101  								// instruction.
  5102  								dst := p.To.Reg
  5103  								ab.Put1(0xe8)
  5104  								cursym.AddRel(ctxt, obj.Reloc{
  5105  									Type: objabi.R_CALL,
  5106  									Off:  int32(p.Pc + int64(ab.Len())),
  5107  									Siz:  4,
  5108  									Sym:  ctxt.Lookup("__x86.get_pc_thunk." + strings.ToLower(rconv(int(dst)))),
  5109  								})
  5110  								ab.PutInt32(0)
  5111  
  5112  								ab.Put2(0x8B, byte(2<<6|reg[dst]|(reg[dst]<<3)))
  5113  								cursym.AddRel(ctxt, obj.Reloc{
  5114  									Type: objabi.R_TLS_IE,
  5115  									Off:  int32(p.Pc + int64(ab.Len())),
  5116  									Siz:  4,
  5117  									Add:  2,
  5118  								})
  5119  								ab.PutInt32(0)
  5120  							} else {
  5121  								// ELF TLS base is 0(GS).
  5122  								pp.From = p.From
  5123  
  5124  								pp.From.Type = obj.TYPE_MEM
  5125  								pp.From.Reg = REG_GS
  5126  								pp.From.Offset = 0
  5127  								pp.From.Index = REG_NONE
  5128  								pp.From.Scale = 0
  5129  								ab.Put2(0x65, // GS
  5130  									0x8B)
  5131  								ab.asmand(ctxt, cursym, p, &pp.From, &p.To)
  5132  							}
  5133  						case objabi.Hplan9:
  5134  							pp.From = obj.Addr{}
  5135  							pp.From.Type = obj.TYPE_MEM
  5136  							pp.From.Name = obj.NAME_EXTERN
  5137  							pp.From.Sym = plan9privates
  5138  							pp.From.Offset = 0
  5139  							pp.From.Index = REG_NONE
  5140  							ab.Put1(0x8B)
  5141  							ab.asmand(ctxt, cursym, p, &pp.From, &p.To)
  5142  						}
  5143  						break
  5144  					}
  5145  
  5146  					switch ctxt.Headtype {
  5147  					default:
  5148  						log.Fatalf("unknown TLS base location for %v", ctxt.Headtype)
  5149  
  5150  					case objabi.Hlinux, objabi.Hfreebsd:
  5151  						if !ctxt.Flag_shared {
  5152  							log.Fatalf("unknown TLS base location for linux/freebsd without -shared")
  5153  						}
  5154  						// Note that this is not generating the same insn as the other cases.
  5155  						//     MOV TLS, R_to
  5156  						// becomes
  5157  						//     movq g@gottpoff(%rip), R_to
  5158  						// which is encoded as
  5159  						//     movq 0(%rip), R_to
  5160  						// and a R_TLS_IE reloc. This all assumes the only tls variable we access
  5161  						// is g, which we can't check here, but will when we assemble the second
  5162  						// instruction.
  5163  						ab.rexflag = Pw | (regrex[p.To.Reg] & Rxr)
  5164  
  5165  						ab.Put2(0x8B, byte(0x05|(reg[p.To.Reg]<<3)))
  5166  						cursym.AddRel(ctxt, obj.Reloc{
  5167  							Type: objabi.R_TLS_IE,
  5168  							Off:  int32(p.Pc + int64(ab.Len())),
  5169  							Siz:  4,
  5170  							Add:  -4,
  5171  						})
  5172  						ab.PutInt32(0)
  5173  
  5174  					case objabi.Hplan9:
  5175  						pp.From = obj.Addr{}
  5176  						pp.From.Type = obj.TYPE_MEM
  5177  						pp.From.Name = obj.NAME_EXTERN
  5178  						pp.From.Sym = plan9privates
  5179  						pp.From.Offset = 0
  5180  						pp.From.Index = REG_NONE
  5181  						ab.rexflag |= Pw
  5182  						ab.Put1(0x8B)
  5183  						ab.asmand(ctxt, cursym, p, &pp.From, &p.To)
  5184  
  5185  					case objabi.Hsolaris: // TODO(rsc): Delete Hsolaris from list. Should not use this code. See progedit in obj6.c.
  5186  						// TLS base is 0(FS).
  5187  						pp.From = p.From
  5188  
  5189  						pp.From.Type = obj.TYPE_MEM
  5190  						pp.From.Name = obj.NAME_NONE
  5191  						pp.From.Reg = REG_NONE
  5192  						pp.From.Offset = 0
  5193  						pp.From.Index = REG_NONE
  5194  						pp.From.Scale = 0
  5195  						ab.rexflag |= Pw
  5196  						ab.Put2(0x64, // FS
  5197  							0x8B)
  5198  						ab.asmand(ctxt, cursym, p, &pp.From, &p.To)
  5199  					}
  5200  				}
  5201  				return
  5202  			}
  5203  		}
  5204  	}
  5205  	goto bad
  5206  
  5207  bad:
  5208  	if ctxt.Arch.Family != sys.AMD64 {
  5209  		// here, the assembly has failed.
  5210  		// if it's a byte instruction that has
  5211  		// unaddressable registers, try to
  5212  		// exchange registers and reissue the
  5213  		// instruction with the operands renamed.
  5214  		pp := *p
  5215  
  5216  		unbytereg(&pp.From, &pp.Ft)
  5217  		unbytereg(&pp.To, &pp.Tt)
  5218  
  5219  		z := int(p.From.Reg)
  5220  		if p.From.Type == obj.TYPE_REG && z >= REG_BP && z <= REG_DI {
  5221  			// TODO(rsc): Use this code for x86-64 too. It has bug fixes not present in the amd64 code base.
  5222  			// For now, different to keep bit-for-bit compatibility.
  5223  			if ctxt.Arch.Family == sys.I386 {
  5224  				breg := byteswapreg(ctxt, &p.To)
  5225  				if breg != REG_AX {
  5226  					ab.Put1(0x87) // xchg lhs,bx
  5227  					ab.asmando(ctxt, cursym, p, &p.From, reg[breg])
  5228  					subreg(&pp, z, breg)
  5229  					ab.doasm(ctxt, cursym, &pp)
  5230  					ab.Put1(0x87) // xchg lhs,bx
  5231  					ab.asmando(ctxt, cursym, p, &p.From, reg[breg])
  5232  				} else {
  5233  					ab.Put1(byte(0x90 + reg[z])) // xchg lsh,ax
  5234  					subreg(&pp, z, REG_AX)
  5235  					ab.doasm(ctxt, cursym, &pp)
  5236  					ab.Put1(byte(0x90 + reg[z])) // xchg lsh,ax
  5237  				}
  5238  				return
  5239  			}
  5240  
  5241  			if isax(&p.To) || p.To.Type == obj.TYPE_NONE {
  5242  				// We certainly don't want to exchange
  5243  				// with AX if the op is MUL or DIV.
  5244  				ab.Put1(0x87) // xchg lhs,bx
  5245  				ab.asmando(ctxt, cursym, p, &p.From, reg[REG_BX])
  5246  				subreg(&pp, z, REG_BX)
  5247  				ab.doasm(ctxt, cursym, &pp)
  5248  				ab.Put1(0x87) // xchg lhs,bx
  5249  				ab.asmando(ctxt, cursym, p, &p.From, reg[REG_BX])
  5250  			} else {
  5251  				ab.Put1(byte(0x90 + reg[z])) // xchg lsh,ax
  5252  				subreg(&pp, z, REG_AX)
  5253  				ab.doasm(ctxt, cursym, &pp)
  5254  				ab.Put1(byte(0x90 + reg[z])) // xchg lsh,ax
  5255  			}
  5256  			return
  5257  		}
  5258  
  5259  		z = int(p.To.Reg)
  5260  		if p.To.Type == obj.TYPE_REG && z >= REG_BP && z <= REG_DI {
  5261  			// TODO(rsc): Use this code for x86-64 too. It has bug fixes not present in the amd64 code base.
  5262  			// For now, different to keep bit-for-bit compatibility.
  5263  			if ctxt.Arch.Family == sys.I386 {
  5264  				breg := byteswapreg(ctxt, &p.From)
  5265  				if breg != REG_AX {
  5266  					ab.Put1(0x87) //xchg rhs,bx
  5267  					ab.asmando(ctxt, cursym, p, &p.To, reg[breg])
  5268  					subreg(&pp, z, breg)
  5269  					ab.doasm(ctxt, cursym, &pp)
  5270  					ab.Put1(0x87) // xchg rhs,bx
  5271  					ab.asmando(ctxt, cursym, p, &p.To, reg[breg])
  5272  				} else {
  5273  					ab.Put1(byte(0x90 + reg[z])) // xchg rsh,ax
  5274  					subreg(&pp, z, REG_AX)
  5275  					ab.doasm(ctxt, cursym, &pp)
  5276  					ab.Put1(byte(0x90 + reg[z])) // xchg rsh,ax
  5277  				}
  5278  				return
  5279  			}
  5280  
  5281  			if isax(&p.From) {
  5282  				ab.Put1(0x87) // xchg rhs,bx
  5283  				ab.asmando(ctxt, cursym, p, &p.To, reg[REG_BX])
  5284  				subreg(&pp, z, REG_BX)
  5285  				ab.doasm(ctxt, cursym, &pp)
  5286  				ab.Put1(0x87) // xchg rhs,bx
  5287  				ab.asmando(ctxt, cursym, p, &p.To, reg[REG_BX])
  5288  			} else {
  5289  				ab.Put1(byte(0x90 + reg[z])) // xchg rsh,ax
  5290  				subreg(&pp, z, REG_AX)
  5291  				ab.doasm(ctxt, cursym, &pp)
  5292  				ab.Put1(byte(0x90 + reg[z])) // xchg rsh,ax
  5293  			}
  5294  			return
  5295  		}
  5296  	}
  5297  
  5298  	ctxt.Diag("%s: invalid instruction: %v", cursym.Name, p)
  5299  }
  5300  
  5301  // byteswapreg returns a byte-addressable register (AX, BX, CX, DX)
  5302  // which is not referenced in a.
  5303  // If a is empty, it returns BX to account for MULB-like instructions
  5304  // that might use DX and AX.
  5305  func byteswapreg(ctxt *obj.Link, a *obj.Addr) int {
  5306  	cana, canb, canc, cand := true, true, true, true
  5307  	if a.Type == obj.TYPE_NONE {
  5308  		cana, cand = false, false
  5309  	}
  5310  
  5311  	if a.Type == obj.TYPE_REG || ((a.Type == obj.TYPE_MEM || a.Type == obj.TYPE_ADDR) && a.Name == obj.NAME_NONE) {
  5312  		switch a.Reg {
  5313  		case REG_NONE:
  5314  			cana, cand = false, false
  5315  		case REG_AX, REG_AL, REG_AH:
  5316  			cana = false
  5317  		case REG_BX, REG_BL, REG_BH:
  5318  			canb = false
  5319  		case REG_CX, REG_CL, REG_CH:
  5320  			canc = false
  5321  		case REG_DX, REG_DL, REG_DH:
  5322  			cand = false
  5323  		}
  5324  	}
  5325  
  5326  	if a.Type == obj.TYPE_MEM || a.Type == obj.TYPE_ADDR {
  5327  		switch a.Index {
  5328  		case REG_AX:
  5329  			cana = false
  5330  		case REG_BX:
  5331  			canb = false
  5332  		case REG_CX:
  5333  			canc = false
  5334  		case REG_DX:
  5335  			cand = false
  5336  		}
  5337  	}
  5338  
  5339  	switch {
  5340  	case cana:
  5341  		return REG_AX
  5342  	case canb:
  5343  		return REG_BX
  5344  	case canc:
  5345  		return REG_CX
  5346  	case cand:
  5347  		return REG_DX
  5348  	default:
  5349  		ctxt.Diag("impossible byte register")
  5350  		ctxt.DiagFlush()
  5351  		log.Fatalf("bad code")
  5352  		return 0
  5353  	}
  5354  }
  5355  
  5356  func isbadbyte(a *obj.Addr) bool {
  5357  	return a.Type == obj.TYPE_REG && (REG_BP <= a.Reg && a.Reg <= REG_DI || REG_BPB <= a.Reg && a.Reg <= REG_DIB)
  5358  }
  5359  
  5360  func (ab *AsmBuf) asmins(ctxt *obj.Link, cursym *obj.LSym, p *obj.Prog) {
  5361  	ab.Reset()
  5362  
  5363  	ab.rexflag = 0
  5364  	ab.vexflag = false
  5365  	ab.evexflag = false
  5366  	mark := ab.Len()
  5367  	ab.doasm(ctxt, cursym, p)
  5368  	if ab.rexflag != 0 && !ab.vexflag && !ab.evexflag {
  5369  		// as befits the whole approach of the architecture,
  5370  		// the rex prefix must appear before the first opcode byte
  5371  		// (and thus after any 66/67/f2/f3/26/2e/3e prefix bytes, but
  5372  		// before the 0f opcode escape!), or it might be ignored.
  5373  		// note that the handbook often misleadingly shows 66/f2/f3 in `opcode'.
  5374  		if ctxt.Arch.Family != sys.AMD64 {
  5375  			ctxt.Diag("asmins: illegal in mode %d: %v (%d %d)", ctxt.Arch.RegSize*8, p, p.Ft, p.Tt)
  5376  		}
  5377  		n := ab.Len()
  5378  		var np int
  5379  		for np = mark; np < n; np++ {
  5380  			c := ab.At(np)
  5381  			if c != 0xf2 && c != 0xf3 && (c < 0x64 || c > 0x67) && c != 0x2e && c != 0x3e && c != 0x26 {
  5382  				break
  5383  			}
  5384  		}
  5385  		ab.Insert(np, byte(0x40|ab.rexflag))
  5386  	}
  5387  
  5388  	n := ab.Len()
  5389  	for i := len(cursym.R) - 1; i >= 0; i-- {
  5390  		r := &cursym.R[i]
  5391  		if int64(r.Off) < p.Pc {
  5392  			break
  5393  		}
  5394  		if ab.rexflag != 0 && !ab.vexflag && !ab.evexflag {
  5395  			r.Off++
  5396  		}
  5397  		if r.Type == objabi.R_PCREL {
  5398  			if ctxt.Arch.Family == sys.AMD64 || p.As == obj.AJMP || p.As == obj.ACALL {
  5399  				// PC-relative addressing is relative to the end of the instruction,
  5400  				// but the relocations applied by the linker are relative to the end
  5401  				// of the relocation. Because immediate instruction
  5402  				// arguments can follow the PC-relative memory reference in the
  5403  				// instruction encoding, the two may not coincide. In this case,
  5404  				// adjust addend so that linker can keep relocating relative to the
  5405  				// end of the relocation.
  5406  				r.Add -= p.Pc + int64(n) - (int64(r.Off) + int64(r.Siz))
  5407  			} else if ctxt.Arch.Family == sys.I386 {
  5408  				// On 386 PC-relative addressing (for non-call/jmp instructions)
  5409  				// assumes that the previous instruction loaded the PC of the end
  5410  				// of that instruction into CX, so the adjustment is relative to
  5411  				// that.
  5412  				r.Add += int64(r.Off) - p.Pc + int64(r.Siz)
  5413  			}
  5414  		}
  5415  		if r.Type == objabi.R_GOTPCREL && ctxt.Arch.Family == sys.I386 {
  5416  			// On 386, R_GOTPCREL makes the same assumptions as R_PCREL.
  5417  			r.Add += int64(r.Off) - p.Pc + int64(r.Siz)
  5418  		}
  5419  
  5420  	}
  5421  }
  5422  
  5423  // unpackOps4 extracts 4 operands from p.
  5424  func unpackOps4(p *obj.Prog) (arg0, arg1, arg2, dst *obj.Addr) {
  5425  	return &p.From, &p.RestArgs[0].Addr, &p.RestArgs[1].Addr, &p.To
  5426  }
  5427  
  5428  // unpackOps5 extracts 5 operands from p.
  5429  func unpackOps5(p *obj.Prog) (arg0, arg1, arg2, arg3, dst *obj.Addr) {
  5430  	return &p.From, &p.RestArgs[0].Addr, &p.RestArgs[1].Addr, &p.RestArgs[2].Addr, &p.To
  5431  }
  5432  

View as plain text