Source file src/simd/archsimd/ops_internal_amd64.go

     1  // Code generated by 'simdgen -o godefs -goroot $GOROOT -arch amd64 -xedPath $XED_PATH go_amd64.yaml types.yaml categories.yaml'; DO NOT EDIT.
     2  
     3  //go:build goexperiment.simd
     4  
     5  package archsimd
     6  
     7  /* blend */
     8  
     9  // blend blends two vectors based on mask values, choosing either
    10  // the first or the second based on whether the third is false or true
    11  //
    12  // Asm: VPBLENDVB, CPU Feature: AVX
    13  func (x Int8x16) blend(y Int8x16, mask Int8x16) Int8x16
    14  
    15  // blend blends two vectors based on mask values, choosing either
    16  // the first or the second based on whether the third is false or true
    17  //
    18  // Asm: VPBLENDVB, CPU Feature: AVX2
    19  func (x Int8x32) blend(y Int8x32, mask Int8x32) Int8x32
    20  
    21  /* blendMasked */
    22  
    23  // blendMasked blends two vectors based on mask values, choosing either
    24  // the first or the second based on whether the third is false or true
    25  //
    26  // This operation is applied selectively under a write mask.
    27  //
    28  // Asm: VPBLENDMB, CPU Feature: AVX512
    29  func (x Int8x64) blendMasked(y Int8x64, mask Mask8x64) Int8x64
    30  
    31  // blendMasked blends two vectors based on mask values, choosing either
    32  // the first or the second based on whether the third is false or true
    33  //
    34  // This operation is applied selectively under a write mask.
    35  //
    36  // Asm: VPBLENDMW, CPU Feature: AVX512
    37  func (x Int16x32) blendMasked(y Int16x32, mask Mask16x32) Int16x32
    38  
    39  // blendMasked blends two vectors based on mask values, choosing either
    40  // the first or the second based on whether the third is false or true
    41  //
    42  // This operation is applied selectively under a write mask.
    43  //
    44  // Asm: VPBLENDMD, CPU Feature: AVX512
    45  func (x Int32x16) blendMasked(y Int32x16, mask Mask32x16) Int32x16
    46  
    47  // blendMasked blends two vectors based on mask values, choosing either
    48  // the first or the second based on whether the third is false or true
    49  //
    50  // This operation is applied selectively under a write mask.
    51  //
    52  // Asm: VPBLENDMQ, CPU Feature: AVX512
    53  func (x Int64x8) blendMasked(y Int64x8, mask Mask64x8) Int64x8
    54  
    55  /* broadcast1To2 */
    56  
    57  // broadcast1To2 copies the lowest element of its input to all 2 elements of
    58  // the output vector.
    59  //
    60  // Asm: VPBROADCASTQ, CPU Feature: AVX2
    61  func (x Float64x2) broadcast1To2() Float64x2
    62  
    63  // broadcast1To2 copies the lowest element of its input to all 2 elements of
    64  // the output vector.
    65  //
    66  // Asm: VPBROADCASTQ, CPU Feature: AVX2
    67  func (x Int64x2) broadcast1To2() Int64x2
    68  
    69  // broadcast1To2 copies the lowest element of its input to all 2 elements of
    70  // the output vector.
    71  //
    72  // Asm: VPBROADCASTQ, CPU Feature: AVX2
    73  func (x Uint64x2) broadcast1To2() Uint64x2
    74  
    75  /* broadcast1To2Masked */
    76  
    77  // broadcast1To2Masked copies the lowest element of its input to all 2 elements of
    78  // the output vector.
    79  //
    80  // This operation is applied selectively under a write mask.
    81  //
    82  // Asm: VPBROADCASTQ, CPU Feature: AVX512
    83  func (x Float64x2) broadcast1To2Masked(mask Mask64x2) Float64x2
    84  
    85  // broadcast1To2Masked copies the lowest element of its input to all 2 elements of
    86  // the output vector.
    87  //
    88  // This operation is applied selectively under a write mask.
    89  //
    90  // Asm: VPBROADCASTQ, CPU Feature: AVX512
    91  func (x Int64x2) broadcast1To2Masked(mask Mask64x2) Int64x2
    92  
    93  // broadcast1To2Masked copies the lowest element of its input to all 2 elements of
    94  // the output vector.
    95  //
    96  // This operation is applied selectively under a write mask.
    97  //
    98  // Asm: VPBROADCASTQ, CPU Feature: AVX512
    99  func (x Uint64x2) broadcast1To2Masked(mask Mask64x2) Uint64x2
   100  
   101  /* broadcast1To4 */
   102  
   103  // broadcast1To4 copies the lowest element of its input to all 4 elements of
   104  // the output vector.
   105  //
   106  // Asm: VBROADCASTSS, CPU Feature: AVX2
   107  func (x Float32x4) broadcast1To4() Float32x4
   108  
   109  // broadcast1To4 copies the lowest element of its input to all 4 elements of
   110  // the output vector.
   111  //
   112  // Asm: VBROADCASTSD, CPU Feature: AVX2
   113  func (x Float64x2) broadcast1To4() Float64x4
   114  
   115  // broadcast1To4 copies the lowest element of its input to all 4 elements of
   116  // the output vector.
   117  //
   118  // Asm: VPBROADCASTD, CPU Feature: AVX2
   119  func (x Int32x4) broadcast1To4() Int32x4
   120  
   121  // broadcast1To4 copies the lowest element of its input to all 4 elements of
   122  // the output vector.
   123  //
   124  // Asm: VPBROADCASTQ, CPU Feature: AVX2
   125  func (x Int64x2) broadcast1To4() Int64x4
   126  
   127  // broadcast1To4 copies the lowest element of its input to all 4 elements of
   128  // the output vector.
   129  //
   130  // Asm: VPBROADCASTD, CPU Feature: AVX2
   131  func (x Uint32x4) broadcast1To4() Uint32x4
   132  
   133  // broadcast1To4 copies the lowest element of its input to all 4 elements of
   134  // the output vector.
   135  //
   136  // Asm: VPBROADCASTQ, CPU Feature: AVX2
   137  func (x Uint64x2) broadcast1To4() Uint64x4
   138  
   139  /* broadcast1To4Masked */
   140  
   141  // broadcast1To4Masked copies the lowest element of its input to all 4 elements of
   142  // the output vector.
   143  //
   144  // This operation is applied selectively under a write mask.
   145  //
   146  // Asm: VBROADCASTSS, CPU Feature: AVX512
   147  func (x Float32x4) broadcast1To4Masked(mask Mask32x4) Float32x4
   148  
   149  // broadcast1To4Masked copies the lowest element of its input to all 4 elements of
   150  // the output vector.
   151  //
   152  // This operation is applied selectively under a write mask.
   153  //
   154  // Asm: VBROADCASTSD, CPU Feature: AVX512
   155  func (x Float64x2) broadcast1To4Masked(mask Mask64x2) Float64x4
   156  
   157  // broadcast1To4Masked copies the lowest element of its input to all 4 elements of
   158  // the output vector.
   159  //
   160  // This operation is applied selectively under a write mask.
   161  //
   162  // Asm: VPBROADCASTD, CPU Feature: AVX512
   163  func (x Int32x4) broadcast1To4Masked(mask Mask32x4) Int32x4
   164  
   165  // broadcast1To4Masked copies the lowest element of its input to all 4 elements of
   166  // the output vector.
   167  //
   168  // This operation is applied selectively under a write mask.
   169  //
   170  // Asm: VPBROADCASTQ, CPU Feature: AVX512
   171  func (x Int64x2) broadcast1To4Masked(mask Mask64x2) Int64x4
   172  
   173  // broadcast1To4Masked copies the lowest element of its input to all 4 elements of
   174  // the output vector.
   175  //
   176  // This operation is applied selectively under a write mask.
   177  //
   178  // Asm: VPBROADCASTD, CPU Feature: AVX512
   179  func (x Uint32x4) broadcast1To4Masked(mask Mask32x4) Uint32x4
   180  
   181  // broadcast1To4Masked copies the lowest element of its input to all 4 elements of
   182  // the output vector.
   183  //
   184  // This operation is applied selectively under a write mask.
   185  //
   186  // Asm: VPBROADCASTQ, CPU Feature: AVX512
   187  func (x Uint64x2) broadcast1To4Masked(mask Mask64x2) Uint64x4
   188  
   189  /* broadcast1To8 */
   190  
   191  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   192  // the output vector.
   193  //
   194  // Asm: VBROADCASTSS, CPU Feature: AVX2
   195  func (x Float32x4) broadcast1To8() Float32x8
   196  
   197  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   198  // the output vector.
   199  //
   200  // Asm: VBROADCASTSD, CPU Feature: AVX512
   201  func (x Float64x2) broadcast1To8() Float64x8
   202  
   203  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   204  // the output vector.
   205  //
   206  // Asm: VPBROADCASTW, CPU Feature: AVX2
   207  func (x Int16x8) broadcast1To8() Int16x8
   208  
   209  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   210  // the output vector.
   211  //
   212  // Asm: VPBROADCASTD, CPU Feature: AVX2
   213  func (x Int32x4) broadcast1To8() Int32x8
   214  
   215  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   216  // the output vector.
   217  //
   218  // Asm: VPBROADCASTQ, CPU Feature: AVX512
   219  func (x Int64x2) broadcast1To8() Int64x8
   220  
   221  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   222  // the output vector.
   223  //
   224  // Asm: VPBROADCASTW, CPU Feature: AVX2
   225  func (x Uint16x8) broadcast1To8() Uint16x8
   226  
   227  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   228  // the output vector.
   229  //
   230  // Asm: VPBROADCASTD, CPU Feature: AVX2
   231  func (x Uint32x4) broadcast1To8() Uint32x8
   232  
   233  // broadcast1To8 copies the lowest element of its input to all 8 elements of
   234  // the output vector.
   235  //
   236  // Asm: VPBROADCASTQ, CPU Feature: AVX512
   237  func (x Uint64x2) broadcast1To8() Uint64x8
   238  
   239  /* broadcast1To8Masked */
   240  
   241  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   242  // the output vector.
   243  //
   244  // This operation is applied selectively under a write mask.
   245  //
   246  // Asm: VBROADCASTSS, CPU Feature: AVX512
   247  func (x Float32x4) broadcast1To8Masked(mask Mask32x4) Float32x8
   248  
   249  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   250  // the output vector.
   251  //
   252  // This operation is applied selectively under a write mask.
   253  //
   254  // Asm: VBROADCASTSD, CPU Feature: AVX512
   255  func (x Float64x2) broadcast1To8Masked(mask Mask64x2) Float64x8
   256  
   257  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   258  // the output vector.
   259  //
   260  // This operation is applied selectively under a write mask.
   261  //
   262  // Asm: VPBROADCASTW, CPU Feature: AVX512
   263  func (x Int16x8) broadcast1To8Masked(mask Mask16x8) Int16x8
   264  
   265  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   266  // the output vector.
   267  //
   268  // This operation is applied selectively under a write mask.
   269  //
   270  // Asm: VPBROADCASTD, CPU Feature: AVX512
   271  func (x Int32x4) broadcast1To8Masked(mask Mask32x4) Int32x8
   272  
   273  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   274  // the output vector.
   275  //
   276  // This operation is applied selectively under a write mask.
   277  //
   278  // Asm: VPBROADCASTQ, CPU Feature: AVX512
   279  func (x Int64x2) broadcast1To8Masked(mask Mask64x2) Int64x8
   280  
   281  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   282  // the output vector.
   283  //
   284  // This operation is applied selectively under a write mask.
   285  //
   286  // Asm: VPBROADCASTW, CPU Feature: AVX512
   287  func (x Uint16x8) broadcast1To8Masked(mask Mask16x8) Uint16x8
   288  
   289  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   290  // the output vector.
   291  //
   292  // This operation is applied selectively under a write mask.
   293  //
   294  // Asm: VPBROADCASTD, CPU Feature: AVX512
   295  func (x Uint32x4) broadcast1To8Masked(mask Mask32x4) Uint32x8
   296  
   297  // broadcast1To8Masked copies the lowest element of its input to all 8 elements of
   298  // the output vector.
   299  //
   300  // This operation is applied selectively under a write mask.
   301  //
   302  // Asm: VPBROADCASTQ, CPU Feature: AVX512
   303  func (x Uint64x2) broadcast1To8Masked(mask Mask64x2) Uint64x8
   304  
   305  /* broadcast1To16 */
   306  
   307  // broadcast1To16 copies the lowest element of its input to all 16 elements of
   308  // the output vector.
   309  //
   310  // Asm: VBROADCASTSS, CPU Feature: AVX512
   311  func (x Float32x4) broadcast1To16() Float32x16
   312  
   313  // broadcast1To16 copies the lowest element of its input to all 16 elements of
   314  // the output vector.
   315  //
   316  // Asm: VPBROADCASTB, CPU Feature: AVX2
   317  func (x Int8x16) broadcast1To16() Int8x16
   318  
   319  // broadcast1To16 copies the lowest element of its input to all 16 elements of
   320  // the output vector.
   321  //
   322  // Asm: VPBROADCASTW, CPU Feature: AVX2
   323  func (x Int16x8) broadcast1To16() Int16x16
   324  
   325  // broadcast1To16 copies the lowest element of its input to all 16 elements of
   326  // the output vector.
   327  //
   328  // Asm: VPBROADCASTD, CPU Feature: AVX512
   329  func (x Int32x4) broadcast1To16() Int32x16
   330  
   331  // broadcast1To16 copies the lowest element of its input to all 16 elements of
   332  // the output vector.
   333  //
   334  // Asm: VPBROADCASTB, CPU Feature: AVX2
   335  func (x Uint8x16) broadcast1To16() Uint8x16
   336  
   337  // broadcast1To16 copies the lowest element of its input to all 16 elements of
   338  // the output vector.
   339  //
   340  // Asm: VPBROADCASTW, CPU Feature: AVX2
   341  func (x Uint16x8) broadcast1To16() Uint16x16
   342  
   343  // broadcast1To16 copies the lowest element of its input to all 16 elements of
   344  // the output vector.
   345  //
   346  // Asm: VPBROADCASTD, CPU Feature: AVX512
   347  func (x Uint32x4) broadcast1To16() Uint32x16
   348  
   349  /* broadcast1To16Masked */
   350  
   351  // broadcast1To16Masked copies the lowest element of its input to all 16 elements of
   352  // the output vector.
   353  //
   354  // This operation is applied selectively under a write mask.
   355  //
   356  // Asm: VBROADCASTSS, CPU Feature: AVX512
   357  func (x Float32x4) broadcast1To16Masked(mask Mask32x4) Float32x16
   358  
   359  // broadcast1To16Masked copies the lowest element of its input to all 16 elements of
   360  // the output vector.
   361  //
   362  // This operation is applied selectively under a write mask.
   363  //
   364  // Asm: VPBROADCASTB, CPU Feature: AVX512
   365  func (x Int8x16) broadcast1To16Masked(mask Mask8x16) Int8x16
   366  
   367  // broadcast1To16Masked copies the lowest element of its input to all 16 elements of
   368  // the output vector.
   369  //
   370  // This operation is applied selectively under a write mask.
   371  //
   372  // Asm: VPBROADCASTW, CPU Feature: AVX512
   373  func (x Int16x8) broadcast1To16Masked(mask Mask16x8) Int16x16
   374  
   375  // broadcast1To16Masked copies the lowest element of its input to all 16 elements of
   376  // the output vector.
   377  //
   378  // This operation is applied selectively under a write mask.
   379  //
   380  // Asm: VPBROADCASTD, CPU Feature: AVX512
   381  func (x Int32x4) broadcast1To16Masked(mask Mask32x4) Int32x16
   382  
   383  // broadcast1To16Masked copies the lowest element of its input to all 16 elements of
   384  // the output vector.
   385  //
   386  // This operation is applied selectively under a write mask.
   387  //
   388  // Asm: VPBROADCASTB, CPU Feature: AVX512
   389  func (x Uint8x16) broadcast1To16Masked(mask Mask8x16) Uint8x16
   390  
   391  // broadcast1To16Masked copies the lowest element of its input to all 16 elements of
   392  // the output vector.
   393  //
   394  // This operation is applied selectively under a write mask.
   395  //
   396  // Asm: VPBROADCASTW, CPU Feature: AVX512
   397  func (x Uint16x8) broadcast1To16Masked(mask Mask16x8) Uint16x16
   398  
   399  // broadcast1To16Masked copies the lowest element of its input to all 16 elements of
   400  // the output vector.
   401  //
   402  // This operation is applied selectively under a write mask.
   403  //
   404  // Asm: VPBROADCASTD, CPU Feature: AVX512
   405  func (x Uint32x4) broadcast1To16Masked(mask Mask32x4) Uint32x16
   406  
   407  /* broadcast1To32 */
   408  
   409  // broadcast1To32 copies the lowest element of its input to all 32 elements of
   410  // the output vector.
   411  //
   412  // Asm: VPBROADCASTB, CPU Feature: AVX2
   413  func (x Int8x16) broadcast1To32() Int8x32
   414  
   415  // broadcast1To32 copies the lowest element of its input to all 32 elements of
   416  // the output vector.
   417  //
   418  // Asm: VPBROADCASTW, CPU Feature: AVX512
   419  func (x Int16x8) broadcast1To32() Int16x32
   420  
   421  // broadcast1To32 copies the lowest element of its input to all 32 elements of
   422  // the output vector.
   423  //
   424  // Asm: VPBROADCASTB, CPU Feature: AVX2
   425  func (x Uint8x16) broadcast1To32() Uint8x32
   426  
   427  // broadcast1To32 copies the lowest element of its input to all 32 elements of
   428  // the output vector.
   429  //
   430  // Asm: VPBROADCASTW, CPU Feature: AVX512
   431  func (x Uint16x8) broadcast1To32() Uint16x32
   432  
   433  /* broadcast1To32Masked */
   434  
   435  // broadcast1To32Masked copies the lowest element of its input to all 32 elements of
   436  // the output vector.
   437  //
   438  // This operation is applied selectively under a write mask.
   439  //
   440  // Asm: VPBROADCASTB, CPU Feature: AVX512
   441  func (x Int8x16) broadcast1To32Masked(mask Mask8x16) Int8x32
   442  
   443  // broadcast1To32Masked copies the lowest element of its input to all 32 elements of
   444  // the output vector.
   445  //
   446  // This operation is applied selectively under a write mask.
   447  //
   448  // Asm: VPBROADCASTW, CPU Feature: AVX512
   449  func (x Int16x8) broadcast1To32Masked(mask Mask16x8) Int16x32
   450  
   451  // broadcast1To32Masked copies the lowest element of its input to all 32 elements of
   452  // the output vector.
   453  //
   454  // This operation is applied selectively under a write mask.
   455  //
   456  // Asm: VPBROADCASTB, CPU Feature: AVX512
   457  func (x Uint8x16) broadcast1To32Masked(mask Mask8x16) Uint8x32
   458  
   459  // broadcast1To32Masked copies the lowest element of its input to all 32 elements of
   460  // the output vector.
   461  //
   462  // This operation is applied selectively under a write mask.
   463  //
   464  // Asm: VPBROADCASTW, CPU Feature: AVX512
   465  func (x Uint16x8) broadcast1To32Masked(mask Mask16x8) Uint16x32
   466  
   467  /* broadcast1To64 */
   468  
   469  // broadcast1To64 copies the lowest element of its input to all 64 elements of
   470  // the output vector.
   471  //
   472  // Asm: VPBROADCASTB, CPU Feature: AVX512
   473  func (x Int8x16) broadcast1To64() Int8x64
   474  
   475  // broadcast1To64 copies the lowest element of its input to all 64 elements of
   476  // the output vector.
   477  //
   478  // Asm: VPBROADCASTB, CPU Feature: AVX512
   479  func (x Uint8x16) broadcast1To64() Uint8x64
   480  
   481  /* broadcast1To64Masked */
   482  
   483  // broadcast1To64Masked copies the lowest element of its input to all 64 elements of
   484  // the output vector.
   485  //
   486  // This operation is applied selectively under a write mask.
   487  //
   488  // Asm: VPBROADCASTB, CPU Feature: AVX512
   489  func (x Int8x16) broadcast1To64Masked(mask Mask8x16) Int8x64
   490  
   491  // broadcast1To64Masked copies the lowest element of its input to all 64 elements of
   492  // the output vector.
   493  //
   494  // This operation is applied selectively under a write mask.
   495  //
   496  // Asm: VPBROADCASTB, CPU Feature: AVX512
   497  func (x Uint8x16) broadcast1To64Masked(mask Mask8x16) Uint8x64
   498  
   499  /* carrylessMultiply */
   500  
   501  // carrylessMultiply computes one of four possible Galois polynomial
   502  // products of selected high and low halves of x and y,
   503  // depending on the value of xyHiLo, returning the 128-bit
   504  // product in the concatenated two elements of the result.
   505  // Bit 0 selects the low (0) or high (1) element of x and
   506  // bit 4 selects the low (0x00) or high (0x10) element of y.
   507  //
   508  // A non-constant value of xyHiLo may result in significantly worse performance for this operation.
   509  //
   510  // Asm: VPCLMULQDQ, CPU Feature: AVX
   511  func (x Uint64x2) carrylessMultiply(xyHiLo uint8, y Uint64x2) Uint64x2
   512  
   513  // carrylessMultiply computes one of two possible Galois polynomial
   514  // products of selected high and low halves of each of the two
   515  // 128-bit lanes of x and y, depending on the value of xyHiLo,
   516  // and returns the four 128-bit products in the result's lanes.
   517  // Bit 0 selects the low (0) or high (1) elements of x's lanes and
   518  // bit 4 selects the low (0x00) or high (0x10) elements of y's lanes.
   519  //
   520  // A non-constant value of xyHiLo may result in significantly worse performance for this operation.
   521  //
   522  // Asm: VPCLMULQDQ, CPU Feature: AVX512VPCLMULQDQ
   523  func (x Uint64x4) carrylessMultiply(xyHiLo uint8, y Uint64x4) Uint64x4
   524  
   525  // carrylessMultiply computes one of four possible Galois polynomial
   526  // products of selected high and low halves of each of the four
   527  // 128-bit lanes of x and y, depending on the value of xyHiLo,
   528  // and returns the four 128-bit products in the result's lanes.
   529  // Bit 0 selects the low (0) or high (1) elements of x's lanes and
   530  // bit 4 selects the low (0x00) or high (0x10) elements of y's lanes.
   531  //
   532  // A non-constant value of xyHiLo may result in significantly worse performance for this operation.
   533  //
   534  // Asm: VPCLMULQDQ, CPU Feature: AVX512VPCLMULQDQ
   535  func (x Uint64x8) carrylessMultiply(xyHiLo uint8, y Uint64x8) Uint64x8
   536  
   537  /* concatSelectedConstant */
   538  
   539  // concatSelectedConstant concatenates selected elements from x and y into the lower and upper
   540  // halves of the output.  The selection is chosen by the constant parameter h1h0l1l0
   541  // where each {h,l}{1,0} is two bits specify which element from y or x to select.
   542  // For example, {0,1,2,3}.concatSelectedConstant(0b_11_01_00_10, {4,5,6,7}) returns
   543  // {2, 0, 5, 7} (don't forget that the binary constant is written big-endian).
   544  //
   545  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   546  //
   547  // Asm: VSHUFPS, CPU Feature: AVX
   548  func (x Float32x4) concatSelectedConstant(h1h0l1l0 uint8, y Float32x4) Float32x4
   549  
   550  // concatSelectedConstant concatenates selected elements from x and y into the lower and upper
   551  // halves of the output.  The selection is chosen by the constant parameter hilo
   552  // where hi and lo are each one bit specifying which 64-bit element to select
   553  // from y and x.  For example {4,5}.concatSelectedConstant(0b10, {6,7})
   554  // returns {4,7}; bit 0, selecting from x, is zero, and selects 4, and bit 1,
   555  // selecting from y, is 1, and selects 7.
   556  //
   557  // A non-constant value of hilo may result in significantly worse performance for this operation.
   558  //
   559  // Asm: VSHUFPD, CPU Feature: AVX
   560  func (x Float64x2) concatSelectedConstant(hilo uint8, y Float64x2) Float64x2
   561  
   562  // concatSelectedConstant concatenates selected elements from x and y into the lower and upper
   563  // halves of the output.  The selection is chosen by the constant parameter h1h0l1l0
   564  // where each {h,l}{1,0} is two bits specify which element from y or x to select.
   565  // For example, {0,1,2,3}.concatSelectedConstant(0b_11_01_00_10, {4,5,6,7}) returns
   566  // {2, 0, 5, 7} (don't forget that the binary constant is written big-endian).
   567  //
   568  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   569  //
   570  // Asm: VSHUFPS, CPU Feature: AVX
   571  func (x Int32x4) concatSelectedConstant(h1h0l1l0 uint8, y Int32x4) Int32x4
   572  
   573  // concatSelectedConstant concatenates selected elements from x and y into the lower and upper
   574  // halves of the output.  The selection is chosen by the constant parameter hilo
   575  // where hi and lo are each one bit specifying which 64-bit element to select
   576  // from y and x.  For example {4,5}.concatSelectedConstant(0b10, {6,7})
   577  // returns {4,7}; bit 0, selecting from x, is zero, and selects 4, and bit 1,
   578  // selecting from y, is 1, and selects 7.
   579  //
   580  // A non-constant value of hilo may result in significantly worse performance for this operation.
   581  //
   582  // Asm: VSHUFPD, CPU Feature: AVX
   583  func (x Int64x2) concatSelectedConstant(hilo uint8, y Int64x2) Int64x2
   584  
   585  // concatSelectedConstant concatenates selected elements from x and y into the lower and upper
   586  // halves of the output.  The selection is chosen by the constant parameter h1h0l1l0
   587  // where each {h,l}{1,0} is two bits specify which element from y or x to select.
   588  // For example, {0,1,2,3}.concatSelectedConstant(0b_11_01_00_10, {4,5,6,7}) returns
   589  // {2, 0, 5, 7} (don't forget that the binary constant is written big-endian).
   590  //
   591  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   592  //
   593  // Asm: VSHUFPS, CPU Feature: AVX
   594  func (x Uint32x4) concatSelectedConstant(h1h0l1l0 uint8, y Uint32x4) Uint32x4
   595  
   596  // concatSelectedConstant concatenates selected elements from x and y into the lower and upper
   597  // halves of the output.  The selection is chosen by the constant parameter hilo
   598  // where hi and lo are each one bit specifying which 64-bit element to select
   599  // from y and x.  For example {4,5}.concatSelectedConstant(0b10, {6,7})
   600  // returns {4,7}; bit 0, selecting from x, is zero, and selects 4, and bit 1,
   601  // selecting from y, is 1, and selects 7.
   602  //
   603  // A non-constant value of hilo may result in significantly worse performance for this operation.
   604  //
   605  // Asm: VSHUFPD, CPU Feature: AVX
   606  func (x Uint64x2) concatSelectedConstant(hilo uint8, y Uint64x2) Uint64x2
   607  
   608  /* concatSelectedConstantGrouped */
   609  
   610  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   611  // into the lower and upper halves of corresponding subvectors of the output.
   612  // The selection is chosen by the constant parameter h1h0l1l0
   613  // where each {h,l}{1,0} is two bits specifying which element from y or x to select.
   614  // For example,
   615  // {0,1,2,3,8,9,10,11}.concatSelectedConstantGrouped(0b_11_01_00_10, {4,5,6,7,12,13,14,15})
   616  // returns {2,0,5,7,10,8,13,15}
   617  // (don't forget that the binary constant is written big-endian).
   618  //
   619  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   620  //
   621  // Asm: VSHUFPS, CPU Feature: AVX
   622  func (x Float32x8) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Float32x8) Float32x8
   623  
   624  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   625  // into the lower and upper halves of corresponding subvectors of the output.
   626  // The selection is chosen by the constant parameter h1h0l1l0
   627  // where each {h,l}{1,0} is two bits specifying which element from y or x to select.
   628  // For example,
   629  //
   630  //	{0,1,2,3,8,9,10,11, 20,21,22,23,28,29,210,211}.concatSelectedConstantGrouped(
   631  //	 0b_11_01_00_10, {4,5,6,7,12,13,14,15, 24,25,26,27,212,213,214,215})
   632  //
   633  // returns {2,0,5,7,10,8,13,15, 22,20,25,27,210,28,213,215}
   634  //
   635  // (don't forget that the binary constant is written big-endian).
   636  //
   637  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   638  //
   639  // Asm: VSHUFPS, CPU Feature: AVX512
   640  func (x Float32x16) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Float32x16) Float32x16
   641  
   642  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   643  // into the lower and upper halves of corresponding subvectors of the output.
   644  // The selections are specified by the constant parameter hilos where each
   645  // hi and lo pair select 64-bit elements from the corresponding 128-bit
   646  // subvectors of x and y.
   647  //
   648  // For example {4,5,8,9}.concatSelectedConstantGrouped(0b_11_10, {6,7,10,11})
   649  // returns {4,7,9,11}; bit 0 is zero, selecting element 0 from x's least
   650  // 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7),
   651  // then 1, selecting element 1 from x's upper 128 bits (9), then 1,
   652  // selecting element 1 from y's upper 128 bits (11).
   653  // This differs from the same method applied to a 32x8 vector, where
   654  // the 8-bit constant performs the same selection on both subvectors.
   655  //
   656  // A non-constant value of hilos may result in significantly worse performance for this operation.
   657  //
   658  // Asm: VSHUFPD, CPU Feature: AVX
   659  func (x Float64x4) concatSelectedConstantGrouped(hilos uint8, y Float64x4) Float64x4
   660  
   661  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   662  // into the lower and upper halves of corresponding subvectors of the output.
   663  // The selections are specified by the constant parameter hilos where each
   664  // hi and lo pair select 64-bit elements from the corresponding 128-bit
   665  // subvectors of x and y.
   666  //
   667  // For example {4,5,8,9,12,13,16,17}.concatSelectedConstantGrouped(0b11_00_11_10, {6,7,10,11,14,15,18,19})
   668  // returns {4,7,9,11,12,14,17,19}; bit 0 is zero, selecting element 0 from x's
   669  // least 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7),
   670  // then 1, selecting element 1 from x's next 128 bits (9), then 1,
   671  // selecting element 1 from y's upper 128 bits (11).  The next two 0 bits select
   672  // the lower elements from x and y's 3rd 128 bit groups (12, 14), the last two
   673  // 1 bits select the upper elements from x and y's last 128 bits (17, 19).
   674  // This differs from the same method applied to a 32x8 or 32x16 vector, where
   675  // the 8-bit constant performs the same selection on all the subvectors.
   676  //
   677  // A non-constant value of hilos may result in significantly worse performance for this operation.
   678  //
   679  // Asm: VSHUFPD, CPU Feature: AVX512
   680  func (x Float64x8) concatSelectedConstantGrouped(hilos uint8, y Float64x8) Float64x8
   681  
   682  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   683  // into the lower and upper halves of corresponding subvectors of the output.
   684  // The selection is chosen by the constant parameter h1h0l1l0
   685  // where each {h,l}{1,0} is two bits specifying which element from y or x to select.
   686  // For example,
   687  // {0,1,2,3,8,9,10,11}.concatSelectedConstantGrouped(0b_11_01_00_10, {4,5,6,7,12,13,14,15})
   688  // returns {2,0,5,7,10,8,13,15}
   689  // (don't forget that the binary constant is written big-endian).
   690  //
   691  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   692  //
   693  // Asm: VSHUFPS, CPU Feature: AVX
   694  func (x Int32x8) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Int32x8) Int32x8
   695  
   696  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   697  // into the lower and upper halves of corresponding subvectors of the output.
   698  // The selection is chosen by the constant parameter h1h0l1l0
   699  // where each {h,l}{1,0} is two bits specifying which element from y or x to select.
   700  // For example,
   701  //
   702  //	{0,1,2,3,8,9,10,11, 20,21,22,23,28,29,210,211}.concatSelectedConstantGrouped(
   703  //	 0b_11_01_00_10, {4,5,6,7,12,13,14,15, 24,25,26,27,212,213,214,215})
   704  //
   705  // returns {2,0,5,7,10,8,13,15, 22,20,25,27,210,28,213,215}
   706  //
   707  // (don't forget that the binary constant is written big-endian).
   708  //
   709  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   710  //
   711  // Asm: VSHUFPS, CPU Feature: AVX512
   712  func (x Int32x16) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Int32x16) Int32x16
   713  
   714  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   715  // into the lower and upper halves of corresponding subvectors of the output.
   716  // The selections are specified by the constant parameter hilos where each
   717  // hi and lo pair select 64-bit elements from the corresponding 128-bit
   718  // subvectors of x and y.
   719  //
   720  // For example {4,5,8,9}.concatSelectedConstantGrouped(0b_11_10, {6,7,10,11})
   721  // returns {4,7,9,11}; bit 0 is zero, selecting element 0 from x's least
   722  // 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7),
   723  // then 1, selecting element 1 from x's upper 128 bits (9), then 1,
   724  // selecting element 1 from y's upper 128 bits (11).
   725  // This differs from the same method applied to a 32x8 vector, where
   726  // the 8-bit constant performs the same selection on both subvectors.
   727  //
   728  // A non-constant value of hilos may result in significantly worse performance for this operation.
   729  //
   730  // Asm: VSHUFPD, CPU Feature: AVX
   731  func (x Int64x4) concatSelectedConstantGrouped(hilos uint8, y Int64x4) Int64x4
   732  
   733  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   734  // into the lower and upper halves of corresponding subvectors of the output.
   735  // The selections are specified by the constant parameter hilos where each
   736  // hi and lo pair select 64-bit elements from the corresponding 128-bit
   737  // subvectors of x and y.
   738  //
   739  // For example {4,5,8,9,12,13,16,17}.concatSelectedConstantGrouped(0b11_00_11_10, {6,7,10,11,14,15,18,19})
   740  // returns {4,7,9,11,12,14,17,19}; bit 0 is zero, selecting element 0 from x's
   741  // least 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7),
   742  // then 1, selecting element 1 from x's next 128 bits (9), then 1,
   743  // selecting element 1 from y's upper 128 bits (11).  The next two 0 bits select
   744  // the lower elements from x and y's 3rd 128 bit groups (12, 14), the last two
   745  // 1 bits select the upper elements from x and y's last 128 bits (17, 19).
   746  // This differs from the same method applied to a 32x8 or 32x16 vector, where
   747  // the 8-bit constant performs the same selection on all the subvectors.
   748  //
   749  // A non-constant value of hilos may result in significantly worse performance for this operation.
   750  //
   751  // Asm: VSHUFPD, CPU Feature: AVX512
   752  func (x Int64x8) concatSelectedConstantGrouped(hilos uint8, y Int64x8) Int64x8
   753  
   754  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   755  // into the lower and upper halves of corresponding subvectors of the output.
   756  // The selection is chosen by the constant parameter h1h0l1l0
   757  // where each {h,l}{1,0} is two bits specifying which element from y or x to select.
   758  // For example,
   759  // {0,1,2,3,8,9,10,11}.concatSelectedConstantGrouped(0b_11_01_00_10, {4,5,6,7,12,13,14,15})
   760  // returns {2,0,5,7,10,8,13,15}
   761  // (don't forget that the binary constant is written big-endian).
   762  //
   763  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   764  //
   765  // Asm: VSHUFPS, CPU Feature: AVX
   766  func (x Uint32x8) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Uint32x8) Uint32x8
   767  
   768  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   769  // into the lower and upper halves of corresponding subvectors of the output.
   770  // The selection is chosen by the constant parameter h1h0l1l0
   771  // where each {h,l}{1,0} is two bits specifying which element from y or x to select.
   772  // For example,
   773  //
   774  //	{0,1,2,3,8,9,10,11, 20,21,22,23,28,29,210,211}.concatSelectedConstantGrouped(
   775  //	 0b_11_01_00_10, {4,5,6,7,12,13,14,15, 24,25,26,27,212,213,214,215})
   776  //
   777  // returns {2,0,5,7,10,8,13,15, 22,20,25,27,210,28,213,215}
   778  //
   779  // (don't forget that the binary constant is written big-endian).
   780  //
   781  // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation.
   782  //
   783  // Asm: VSHUFPS, CPU Feature: AVX512
   784  func (x Uint32x16) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Uint32x16) Uint32x16
   785  
   786  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   787  // into the lower and upper halves of corresponding subvectors of the output.
   788  // The selections are specified by the constant parameter hilos where each
   789  // hi and lo pair select 64-bit elements from the corresponding 128-bit
   790  // subvectors of x and y.
   791  //
   792  // For example {4,5,8,9}.concatSelectedConstantGrouped(0b_11_10, {6,7,10,11})
   793  // returns {4,7,9,11}; bit 0 is zero, selecting element 0 from x's least
   794  // 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7),
   795  // then 1, selecting element 1 from x's upper 128 bits (9), then 1,
   796  // selecting element 1 from y's upper 128 bits (11).
   797  // This differs from the same method applied to a 32x8 vector, where
   798  // the 8-bit constant performs the same selection on both subvectors.
   799  //
   800  // A non-constant value of hilos may result in significantly worse performance for this operation.
   801  //
   802  // Asm: VSHUFPD, CPU Feature: AVX
   803  func (x Uint64x4) concatSelectedConstantGrouped(hilos uint8, y Uint64x4) Uint64x4
   804  
   805  // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y
   806  // into the lower and upper halves of corresponding subvectors of the output.
   807  // The selections are specified by the constant parameter hilos where each
   808  // hi and lo pair select 64-bit elements from the corresponding 128-bit
   809  // subvectors of x and y.
   810  //
   811  // For example {4,5,8,9,12,13,16,17}.concatSelectedConstantGrouped(0b11_00_11_10, {6,7,10,11,14,15,18,19})
   812  // returns {4,7,9,11,12,14,17,19}; bit 0 is zero, selecting element 0 from x's
   813  // least 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7),
   814  // then 1, selecting element 1 from x's next 128 bits (9), then 1,
   815  // selecting element 1 from y's upper 128 bits (11).  The next two 0 bits select
   816  // the lower elements from x and y's 3rd 128 bit groups (12, 14), the last two
   817  // 1 bits select the upper elements from x and y's last 128 bits (17, 19).
   818  // This differs from the same method applied to a 32x8 or 32x16 vector, where
   819  // the 8-bit constant performs the same selection on all the subvectors.
   820  //
   821  // A non-constant value of hilos may result in significantly worse performance for this operation.
   822  //
   823  // Asm: VSHUFPD, CPU Feature: AVX512
   824  func (x Uint64x8) concatSelectedConstantGrouped(hilos uint8, y Uint64x8) Uint64x8
   825  
   826  /* permuteScalars */
   827  
   828  // permuteScalars performs a permutation of vector x using constant indices:
   829  //
   830  //	result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]]}
   831  //
   832  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   833  //
   834  // A non-constant value of indices may result in significantly worse performance for this operation.
   835  //
   836  // Asm: VPSHUFD, CPU Feature: AVX
   837  func (x Int32x4) permuteScalars(indices uint8) Int32x4
   838  
   839  // permuteScalars performs a permutation of vector x using constant indices:
   840  //
   841  //	result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]]}
   842  //
   843  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   844  //
   845  // A non-constant value of indices may result in significantly worse performance for this operation.
   846  //
   847  // Asm: VPSHUFD, CPU Feature: AVX
   848  func (x Uint32x4) permuteScalars(indices uint8) Uint32x4
   849  
   850  /* permuteScalarsGrouped */
   851  
   852  // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices:
   853  //
   854  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...}
   855  //
   856  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   857  // Each group is of size 128-bit.
   858  //
   859  // A non-constant value of indices may result in significantly worse performance for this operation.
   860  //
   861  // Asm: VPSHUFD, CPU Feature: AVX2
   862  func (x Int32x8) permuteScalarsGrouped(indices uint8) Int32x8
   863  
   864  // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices:
   865  //
   866  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...}
   867  //
   868  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   869  // Each group is of size 128-bit.
   870  //
   871  // A non-constant value of indices may result in significantly worse performance for this operation.
   872  //
   873  // Asm: VPSHUFD, CPU Feature: AVX512
   874  func (x Int32x16) permuteScalarsGrouped(indices uint8) Int32x16
   875  
   876  // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices:
   877  //
   878  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...}
   879  //
   880  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   881  // Each group is of size 128-bit.
   882  //
   883  // A non-constant value of indices may result in significantly worse performance for this operation.
   884  //
   885  // Asm: VPSHUFD, CPU Feature: AVX2
   886  func (x Uint32x8) permuteScalarsGrouped(indices uint8) Uint32x8
   887  
   888  // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices:
   889  //
   890  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...}
   891  //
   892  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   893  // Each group is of size 128-bit.
   894  //
   895  // A non-constant value of indices may result in significantly worse performance for this operation.
   896  //
   897  // Asm: VPSHUFD, CPU Feature: AVX512
   898  func (x Uint32x16) permuteScalarsGrouped(indices uint8) Uint32x16
   899  
   900  /* permuteScalarsHi */
   901  
   902  // permuteScalarsHi performs a permutation of vector x using constant indices:
   903  //
   904  //	result = {x[0], x[1], x[2], x[3], x[indices[0:2]+4], x[indices[2:4]+4], x[indices[4:6]+4], x[indices[6:8]+4]}
   905  //
   906  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   907  //
   908  // A non-constant value of indices may result in significantly worse performance for this operation.
   909  //
   910  // Asm: VPSHUFHW, CPU Feature: AVX
   911  func (x Int16x8) permuteScalarsHi(indices uint8) Int16x8
   912  
   913  // permuteScalarsHi performs a permutation of vector x using constant indices:
   914  //
   915  //	result = {x[0], x[1], x[2], x[3], x[indices[0:2]+4], x[indices[2:4]+4], x[indices[4:6]+4], x[indices[6:8]+4]}
   916  //
   917  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   918  //
   919  // A non-constant value of indices may result in significantly worse performance for this operation.
   920  //
   921  // Asm: VPSHUFHW, CPU Feature: AVX
   922  func (x Uint16x8) permuteScalarsHi(indices uint8) Uint16x8
   923  
   924  /* permuteScalarsHiGrouped */
   925  
   926  // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices:
   927  // result =
   928  //
   929  //	{x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4],
   930  //	 x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...}
   931  //
   932  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   933  // Each group is of size 128-bit.
   934  //
   935  // A non-constant value of indices may result in significantly worse performance for this operation.
   936  //
   937  // Asm: VPSHUFHW, CPU Feature: AVX2
   938  func (x Int16x16) permuteScalarsHiGrouped(indices uint8) Int16x16
   939  
   940  // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices:
   941  // result =
   942  //
   943  //	{x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4],
   944  //	 x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...}
   945  //
   946  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   947  // Each group is of size 128-bit.
   948  //
   949  // A non-constant value of indices may result in significantly worse performance for this operation.
   950  //
   951  // Asm: VPSHUFHW, CPU Feature: AVX512
   952  func (x Int16x32) permuteScalarsHiGrouped(indices uint8) Int16x32
   953  
   954  // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices:
   955  // result =
   956  //
   957  //	{x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4],
   958  //	 x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...}
   959  //
   960  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   961  // Each group is of size 128-bit.
   962  //
   963  // A non-constant value of indices may result in significantly worse performance for this operation.
   964  //
   965  // Asm: VPSHUFHW, CPU Feature: AVX2
   966  func (x Uint16x16) permuteScalarsHiGrouped(indices uint8) Uint16x16
   967  
   968  // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices:
   969  // result =
   970  //
   971  //	{x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4],
   972  //	 x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...}
   973  //
   974  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   975  // Each group is of size 128-bit.
   976  //
   977  // A non-constant value of indices may result in significantly worse performance for this operation.
   978  //
   979  // Asm: VPSHUFHW, CPU Feature: AVX512
   980  func (x Uint16x32) permuteScalarsHiGrouped(indices uint8) Uint16x32
   981  
   982  /* permuteScalarsLo */
   983  
   984  // permuteScalarsLo performs a permutation of vector x using constant indices:
   985  //
   986  //	result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]], x[4], x[5], x[6], x[7]}
   987  //
   988  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
   989  //
   990  // A non-constant value of indices may result in significantly worse performance for this operation.
   991  //
   992  // Asm: VPSHUFLW, CPU Feature: AVX
   993  func (x Int16x8) permuteScalarsLo(indices uint8) Int16x8
   994  
   995  // permuteScalarsLo performs a permutation of vector x using constant indices:
   996  //
   997  //	result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]], x[4], x[5], x[6], x[7]}
   998  //
   999  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
  1000  //
  1001  // A non-constant value of indices may result in significantly worse performance for this operation.
  1002  //
  1003  // Asm: VPSHUFLW, CPU Feature: AVX
  1004  func (x Uint16x8) permuteScalarsLo(indices uint8) Uint16x8
  1005  
  1006  /* permuteScalarsLoGrouped */
  1007  
  1008  // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices:
  1009  //
  1010  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7],
  1011  //	 x_group1[indices[0:2]], ...}
  1012  //
  1013  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
  1014  // Each group is of size 128-bit.
  1015  //
  1016  // A non-constant value of indices may result in significantly worse performance for this operation.
  1017  //
  1018  // Asm: VPSHUFLW, CPU Feature: AVX2
  1019  func (x Int16x16) permuteScalarsLoGrouped(indices uint8) Int16x16
  1020  
  1021  // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices:
  1022  //
  1023  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7],
  1024  //	 x_group1[indices[0:2]], ...}
  1025  //
  1026  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
  1027  // Each group is of size 128-bit.
  1028  //
  1029  // A non-constant value of indices may result in significantly worse performance for this operation.
  1030  //
  1031  // Asm: VPSHUFLW, CPU Feature: AVX512
  1032  func (x Int16x32) permuteScalarsLoGrouped(indices uint8) Int16x32
  1033  
  1034  // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices:
  1035  //
  1036  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7],
  1037  //	 x_group1[indices[0:2]], ...}
  1038  //
  1039  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
  1040  // Each group is of size 128-bit.
  1041  //
  1042  // A non-constant value of indices may result in significantly worse performance for this operation.
  1043  //
  1044  // Asm: VPSHUFLW, CPU Feature: AVX2
  1045  func (x Uint16x16) permuteScalarsLoGrouped(indices uint8) Uint16x16
  1046  
  1047  // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices:
  1048  //
  1049  //	result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7],
  1050  //	 x_group1[indices[0:2]], ...}
  1051  //
  1052  // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index.
  1053  // Each group is of size 128-bit.
  1054  //
  1055  // A non-constant value of indices may result in significantly worse performance for this operation.
  1056  //
  1057  // Asm: VPSHUFLW, CPU Feature: AVX512
  1058  func (x Uint16x32) permuteScalarsLoGrouped(indices uint8) Uint16x32
  1059  
  1060  /* tern */
  1061  
  1062  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1063  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1064  //
  1065  // A non-constant value of table may result in significantly worse performance for this operation.
  1066  //
  1067  // Asm: VPTERNLOGD, CPU Feature: AVX512
  1068  func (x Int32x4) tern(table uint8, y Int32x4, z Int32x4) Int32x4
  1069  
  1070  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1071  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1072  //
  1073  // A non-constant value of table may result in significantly worse performance for this operation.
  1074  //
  1075  // Asm: VPTERNLOGD, CPU Feature: AVX512
  1076  func (x Int32x8) tern(table uint8, y Int32x8, z Int32x8) Int32x8
  1077  
  1078  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1079  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1080  //
  1081  // A non-constant value of table may result in significantly worse performance for this operation.
  1082  //
  1083  // Asm: VPTERNLOGD, CPU Feature: AVX512
  1084  func (x Int32x16) tern(table uint8, y Int32x16, z Int32x16) Int32x16
  1085  
  1086  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1087  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1088  //
  1089  // A non-constant value of table may result in significantly worse performance for this operation.
  1090  //
  1091  // Asm: VPTERNLOGQ, CPU Feature: AVX512
  1092  func (x Int64x2) tern(table uint8, y Int64x2, z Int64x2) Int64x2
  1093  
  1094  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1095  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1096  //
  1097  // A non-constant value of table may result in significantly worse performance for this operation.
  1098  //
  1099  // Asm: VPTERNLOGQ, CPU Feature: AVX512
  1100  func (x Int64x4) tern(table uint8, y Int64x4, z Int64x4) Int64x4
  1101  
  1102  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1103  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1104  //
  1105  // A non-constant value of table may result in significantly worse performance for this operation.
  1106  //
  1107  // Asm: VPTERNLOGQ, CPU Feature: AVX512
  1108  func (x Int64x8) tern(table uint8, y Int64x8, z Int64x8) Int64x8
  1109  
  1110  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1111  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1112  //
  1113  // A non-constant value of table may result in significantly worse performance for this operation.
  1114  //
  1115  // Asm: VPTERNLOGD, CPU Feature: AVX512
  1116  func (x Uint32x4) tern(table uint8, y Uint32x4, z Uint32x4) Uint32x4
  1117  
  1118  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1119  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1120  //
  1121  // A non-constant value of table may result in significantly worse performance for this operation.
  1122  //
  1123  // Asm: VPTERNLOGD, CPU Feature: AVX512
  1124  func (x Uint32x8) tern(table uint8, y Uint32x8, z Uint32x8) Uint32x8
  1125  
  1126  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1127  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1128  //
  1129  // A non-constant value of table may result in significantly worse performance for this operation.
  1130  //
  1131  // Asm: VPTERNLOGD, CPU Feature: AVX512
  1132  func (x Uint32x16) tern(table uint8, y Uint32x16, z Uint32x16) Uint32x16
  1133  
  1134  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1135  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1136  //
  1137  // A non-constant value of table may result in significantly worse performance for this operation.
  1138  //
  1139  // Asm: VPTERNLOGQ, CPU Feature: AVX512
  1140  func (x Uint64x2) tern(table uint8, y Uint64x2, z Uint64x2) Uint64x2
  1141  
  1142  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1143  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1144  //
  1145  // A non-constant value of table may result in significantly worse performance for this operation.
  1146  //
  1147  // Asm: VPTERNLOGQ, CPU Feature: AVX512
  1148  func (x Uint64x4) tern(table uint8, y Uint64x4, z Uint64x4) Uint64x4
  1149  
  1150  // tern performs a logical operation on three vectors based on the 8-bit truth table.
  1151  // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z))
  1152  //
  1153  // A non-constant value of table may result in significantly worse performance for this operation.
  1154  //
  1155  // Asm: VPTERNLOGQ, CPU Feature: AVX512
  1156  func (x Uint64x8) tern(table uint8, y Uint64x8, z Uint64x8) Uint64x8
  1157  

View as plain text