Source file src/simd/archsimd/ops_internal_amd64.go
1 // Code generated by 'simdgen -o godefs -goroot $GOROOT -arch amd64 -xedPath $XED_PATH go_amd64.yaml types.yaml categories.yaml'; DO NOT EDIT. 2 3 //go:build goexperiment.simd 4 5 package archsimd 6 7 /* blend */ 8 9 // blend blends two vectors based on mask values, choosing either 10 // the first or the second based on whether the third is false or true 11 // 12 // Asm: VPBLENDVB, CPU Feature: AVX 13 func (x Int8x16) blend(y Int8x16, mask Int8x16) Int8x16 14 15 // blend blends two vectors based on mask values, choosing either 16 // the first or the second based on whether the third is false or true 17 // 18 // Asm: VPBLENDVB, CPU Feature: AVX2 19 func (x Int8x32) blend(y Int8x32, mask Int8x32) Int8x32 20 21 /* blendMasked */ 22 23 // blendMasked blends two vectors based on mask values, choosing either 24 // the first or the second based on whether the third is false or true 25 // 26 // This operation is applied selectively under a write mask. 27 // 28 // Asm: VPBLENDMB, CPU Feature: AVX512 29 func (x Int8x64) blendMasked(y Int8x64, mask Mask8x64) Int8x64 30 31 // blendMasked blends two vectors based on mask values, choosing either 32 // the first or the second based on whether the third is false or true 33 // 34 // This operation is applied selectively under a write mask. 35 // 36 // Asm: VPBLENDMW, CPU Feature: AVX512 37 func (x Int16x32) blendMasked(y Int16x32, mask Mask16x32) Int16x32 38 39 // blendMasked blends two vectors based on mask values, choosing either 40 // the first or the second based on whether the third is false or true 41 // 42 // This operation is applied selectively under a write mask. 43 // 44 // Asm: VPBLENDMD, CPU Feature: AVX512 45 func (x Int32x16) blendMasked(y Int32x16, mask Mask32x16) Int32x16 46 47 // blendMasked blends two vectors based on mask values, choosing either 48 // the first or the second based on whether the third is false or true 49 // 50 // This operation is applied selectively under a write mask. 51 // 52 // Asm: VPBLENDMQ, CPU Feature: AVX512 53 func (x Int64x8) blendMasked(y Int64x8, mask Mask64x8) Int64x8 54 55 /* broadcast1To2 */ 56 57 // broadcast1To2 copies the lowest element of its input to all 2 elements of 58 // the output vector. 59 // 60 // Asm: VPBROADCASTQ, CPU Feature: AVX2 61 func (x Float64x2) broadcast1To2() Float64x2 62 63 // broadcast1To2 copies the lowest element of its input to all 2 elements of 64 // the output vector. 65 // 66 // Asm: VPBROADCASTQ, CPU Feature: AVX2 67 func (x Int64x2) broadcast1To2() Int64x2 68 69 // broadcast1To2 copies the lowest element of its input to all 2 elements of 70 // the output vector. 71 // 72 // Asm: VPBROADCASTQ, CPU Feature: AVX2 73 func (x Uint64x2) broadcast1To2() Uint64x2 74 75 /* broadcast1To2Masked */ 76 77 // broadcast1To2Masked copies the lowest element of its input to all 2 elements of 78 // the output vector. 79 // 80 // This operation is applied selectively under a write mask. 81 // 82 // Asm: VPBROADCASTQ, CPU Feature: AVX512 83 func (x Float64x2) broadcast1To2Masked(mask Mask64x2) Float64x2 84 85 // broadcast1To2Masked copies the lowest element of its input to all 2 elements of 86 // the output vector. 87 // 88 // This operation is applied selectively under a write mask. 89 // 90 // Asm: VPBROADCASTQ, CPU Feature: AVX512 91 func (x Int64x2) broadcast1To2Masked(mask Mask64x2) Int64x2 92 93 // broadcast1To2Masked copies the lowest element of its input to all 2 elements of 94 // the output vector. 95 // 96 // This operation is applied selectively under a write mask. 97 // 98 // Asm: VPBROADCASTQ, CPU Feature: AVX512 99 func (x Uint64x2) broadcast1To2Masked(mask Mask64x2) Uint64x2 100 101 /* broadcast1To4 */ 102 103 // broadcast1To4 copies the lowest element of its input to all 4 elements of 104 // the output vector. 105 // 106 // Asm: VBROADCASTSS, CPU Feature: AVX2 107 func (x Float32x4) broadcast1To4() Float32x4 108 109 // broadcast1To4 copies the lowest element of its input to all 4 elements of 110 // the output vector. 111 // 112 // Asm: VBROADCASTSD, CPU Feature: AVX2 113 func (x Float64x2) broadcast1To4() Float64x4 114 115 // broadcast1To4 copies the lowest element of its input to all 4 elements of 116 // the output vector. 117 // 118 // Asm: VPBROADCASTD, CPU Feature: AVX2 119 func (x Int32x4) broadcast1To4() Int32x4 120 121 // broadcast1To4 copies the lowest element of its input to all 4 elements of 122 // the output vector. 123 // 124 // Asm: VPBROADCASTQ, CPU Feature: AVX2 125 func (x Int64x2) broadcast1To4() Int64x4 126 127 // broadcast1To4 copies the lowest element of its input to all 4 elements of 128 // the output vector. 129 // 130 // Asm: VPBROADCASTD, CPU Feature: AVX2 131 func (x Uint32x4) broadcast1To4() Uint32x4 132 133 // broadcast1To4 copies the lowest element of its input to all 4 elements of 134 // the output vector. 135 // 136 // Asm: VPBROADCASTQ, CPU Feature: AVX2 137 func (x Uint64x2) broadcast1To4() Uint64x4 138 139 /* broadcast1To4Masked */ 140 141 // broadcast1To4Masked copies the lowest element of its input to all 4 elements of 142 // the output vector. 143 // 144 // This operation is applied selectively under a write mask. 145 // 146 // Asm: VBROADCASTSS, CPU Feature: AVX512 147 func (x Float32x4) broadcast1To4Masked(mask Mask32x4) Float32x4 148 149 // broadcast1To4Masked copies the lowest element of its input to all 4 elements of 150 // the output vector. 151 // 152 // This operation is applied selectively under a write mask. 153 // 154 // Asm: VBROADCASTSD, CPU Feature: AVX512 155 func (x Float64x2) broadcast1To4Masked(mask Mask64x2) Float64x4 156 157 // broadcast1To4Masked copies the lowest element of its input to all 4 elements of 158 // the output vector. 159 // 160 // This operation is applied selectively under a write mask. 161 // 162 // Asm: VPBROADCASTD, CPU Feature: AVX512 163 func (x Int32x4) broadcast1To4Masked(mask Mask32x4) Int32x4 164 165 // broadcast1To4Masked copies the lowest element of its input to all 4 elements of 166 // the output vector. 167 // 168 // This operation is applied selectively under a write mask. 169 // 170 // Asm: VPBROADCASTQ, CPU Feature: AVX512 171 func (x Int64x2) broadcast1To4Masked(mask Mask64x2) Int64x4 172 173 // broadcast1To4Masked copies the lowest element of its input to all 4 elements of 174 // the output vector. 175 // 176 // This operation is applied selectively under a write mask. 177 // 178 // Asm: VPBROADCASTD, CPU Feature: AVX512 179 func (x Uint32x4) broadcast1To4Masked(mask Mask32x4) Uint32x4 180 181 // broadcast1To4Masked copies the lowest element of its input to all 4 elements of 182 // the output vector. 183 // 184 // This operation is applied selectively under a write mask. 185 // 186 // Asm: VPBROADCASTQ, CPU Feature: AVX512 187 func (x Uint64x2) broadcast1To4Masked(mask Mask64x2) Uint64x4 188 189 /* broadcast1To8 */ 190 191 // broadcast1To8 copies the lowest element of its input to all 8 elements of 192 // the output vector. 193 // 194 // Asm: VBROADCASTSS, CPU Feature: AVX2 195 func (x Float32x4) broadcast1To8() Float32x8 196 197 // broadcast1To8 copies the lowest element of its input to all 8 elements of 198 // the output vector. 199 // 200 // Asm: VBROADCASTSD, CPU Feature: AVX512 201 func (x Float64x2) broadcast1To8() Float64x8 202 203 // broadcast1To8 copies the lowest element of its input to all 8 elements of 204 // the output vector. 205 // 206 // Asm: VPBROADCASTW, CPU Feature: AVX2 207 func (x Int16x8) broadcast1To8() Int16x8 208 209 // broadcast1To8 copies the lowest element of its input to all 8 elements of 210 // the output vector. 211 // 212 // Asm: VPBROADCASTD, CPU Feature: AVX2 213 func (x Int32x4) broadcast1To8() Int32x8 214 215 // broadcast1To8 copies the lowest element of its input to all 8 elements of 216 // the output vector. 217 // 218 // Asm: VPBROADCASTQ, CPU Feature: AVX512 219 func (x Int64x2) broadcast1To8() Int64x8 220 221 // broadcast1To8 copies the lowest element of its input to all 8 elements of 222 // the output vector. 223 // 224 // Asm: VPBROADCASTW, CPU Feature: AVX2 225 func (x Uint16x8) broadcast1To8() Uint16x8 226 227 // broadcast1To8 copies the lowest element of its input to all 8 elements of 228 // the output vector. 229 // 230 // Asm: VPBROADCASTD, CPU Feature: AVX2 231 func (x Uint32x4) broadcast1To8() Uint32x8 232 233 // broadcast1To8 copies the lowest element of its input to all 8 elements of 234 // the output vector. 235 // 236 // Asm: VPBROADCASTQ, CPU Feature: AVX512 237 func (x Uint64x2) broadcast1To8() Uint64x8 238 239 /* broadcast1To8Masked */ 240 241 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 242 // the output vector. 243 // 244 // This operation is applied selectively under a write mask. 245 // 246 // Asm: VBROADCASTSS, CPU Feature: AVX512 247 func (x Float32x4) broadcast1To8Masked(mask Mask32x4) Float32x8 248 249 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 250 // the output vector. 251 // 252 // This operation is applied selectively under a write mask. 253 // 254 // Asm: VBROADCASTSD, CPU Feature: AVX512 255 func (x Float64x2) broadcast1To8Masked(mask Mask64x2) Float64x8 256 257 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 258 // the output vector. 259 // 260 // This operation is applied selectively under a write mask. 261 // 262 // Asm: VPBROADCASTW, CPU Feature: AVX512 263 func (x Int16x8) broadcast1To8Masked(mask Mask16x8) Int16x8 264 265 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 266 // the output vector. 267 // 268 // This operation is applied selectively under a write mask. 269 // 270 // Asm: VPBROADCASTD, CPU Feature: AVX512 271 func (x Int32x4) broadcast1To8Masked(mask Mask32x4) Int32x8 272 273 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 274 // the output vector. 275 // 276 // This operation is applied selectively under a write mask. 277 // 278 // Asm: VPBROADCASTQ, CPU Feature: AVX512 279 func (x Int64x2) broadcast1To8Masked(mask Mask64x2) Int64x8 280 281 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 282 // the output vector. 283 // 284 // This operation is applied selectively under a write mask. 285 // 286 // Asm: VPBROADCASTW, CPU Feature: AVX512 287 func (x Uint16x8) broadcast1To8Masked(mask Mask16x8) Uint16x8 288 289 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 290 // the output vector. 291 // 292 // This operation is applied selectively under a write mask. 293 // 294 // Asm: VPBROADCASTD, CPU Feature: AVX512 295 func (x Uint32x4) broadcast1To8Masked(mask Mask32x4) Uint32x8 296 297 // broadcast1To8Masked copies the lowest element of its input to all 8 elements of 298 // the output vector. 299 // 300 // This operation is applied selectively under a write mask. 301 // 302 // Asm: VPBROADCASTQ, CPU Feature: AVX512 303 func (x Uint64x2) broadcast1To8Masked(mask Mask64x2) Uint64x8 304 305 /* broadcast1To16 */ 306 307 // broadcast1To16 copies the lowest element of its input to all 16 elements of 308 // the output vector. 309 // 310 // Asm: VBROADCASTSS, CPU Feature: AVX512 311 func (x Float32x4) broadcast1To16() Float32x16 312 313 // broadcast1To16 copies the lowest element of its input to all 16 elements of 314 // the output vector. 315 // 316 // Asm: VPBROADCASTB, CPU Feature: AVX2 317 func (x Int8x16) broadcast1To16() Int8x16 318 319 // broadcast1To16 copies the lowest element of its input to all 16 elements of 320 // the output vector. 321 // 322 // Asm: VPBROADCASTW, CPU Feature: AVX2 323 func (x Int16x8) broadcast1To16() Int16x16 324 325 // broadcast1To16 copies the lowest element of its input to all 16 elements of 326 // the output vector. 327 // 328 // Asm: VPBROADCASTD, CPU Feature: AVX512 329 func (x Int32x4) broadcast1To16() Int32x16 330 331 // broadcast1To16 copies the lowest element of its input to all 16 elements of 332 // the output vector. 333 // 334 // Asm: VPBROADCASTB, CPU Feature: AVX2 335 func (x Uint8x16) broadcast1To16() Uint8x16 336 337 // broadcast1To16 copies the lowest element of its input to all 16 elements of 338 // the output vector. 339 // 340 // Asm: VPBROADCASTW, CPU Feature: AVX2 341 func (x Uint16x8) broadcast1To16() Uint16x16 342 343 // broadcast1To16 copies the lowest element of its input to all 16 elements of 344 // the output vector. 345 // 346 // Asm: VPBROADCASTD, CPU Feature: AVX512 347 func (x Uint32x4) broadcast1To16() Uint32x16 348 349 /* broadcast1To16Masked */ 350 351 // broadcast1To16Masked copies the lowest element of its input to all 16 elements of 352 // the output vector. 353 // 354 // This operation is applied selectively under a write mask. 355 // 356 // Asm: VBROADCASTSS, CPU Feature: AVX512 357 func (x Float32x4) broadcast1To16Masked(mask Mask32x4) Float32x16 358 359 // broadcast1To16Masked copies the lowest element of its input to all 16 elements of 360 // the output vector. 361 // 362 // This operation is applied selectively under a write mask. 363 // 364 // Asm: VPBROADCASTB, CPU Feature: AVX512 365 func (x Int8x16) broadcast1To16Masked(mask Mask8x16) Int8x16 366 367 // broadcast1To16Masked copies the lowest element of its input to all 16 elements of 368 // the output vector. 369 // 370 // This operation is applied selectively under a write mask. 371 // 372 // Asm: VPBROADCASTW, CPU Feature: AVX512 373 func (x Int16x8) broadcast1To16Masked(mask Mask16x8) Int16x16 374 375 // broadcast1To16Masked copies the lowest element of its input to all 16 elements of 376 // the output vector. 377 // 378 // This operation is applied selectively under a write mask. 379 // 380 // Asm: VPBROADCASTD, CPU Feature: AVX512 381 func (x Int32x4) broadcast1To16Masked(mask Mask32x4) Int32x16 382 383 // broadcast1To16Masked copies the lowest element of its input to all 16 elements of 384 // the output vector. 385 // 386 // This operation is applied selectively under a write mask. 387 // 388 // Asm: VPBROADCASTB, CPU Feature: AVX512 389 func (x Uint8x16) broadcast1To16Masked(mask Mask8x16) Uint8x16 390 391 // broadcast1To16Masked copies the lowest element of its input to all 16 elements of 392 // the output vector. 393 // 394 // This operation is applied selectively under a write mask. 395 // 396 // Asm: VPBROADCASTW, CPU Feature: AVX512 397 func (x Uint16x8) broadcast1To16Masked(mask Mask16x8) Uint16x16 398 399 // broadcast1To16Masked copies the lowest element of its input to all 16 elements of 400 // the output vector. 401 // 402 // This operation is applied selectively under a write mask. 403 // 404 // Asm: VPBROADCASTD, CPU Feature: AVX512 405 func (x Uint32x4) broadcast1To16Masked(mask Mask32x4) Uint32x16 406 407 /* broadcast1To32 */ 408 409 // broadcast1To32 copies the lowest element of its input to all 32 elements of 410 // the output vector. 411 // 412 // Asm: VPBROADCASTB, CPU Feature: AVX2 413 func (x Int8x16) broadcast1To32() Int8x32 414 415 // broadcast1To32 copies the lowest element of its input to all 32 elements of 416 // the output vector. 417 // 418 // Asm: VPBROADCASTW, CPU Feature: AVX512 419 func (x Int16x8) broadcast1To32() Int16x32 420 421 // broadcast1To32 copies the lowest element of its input to all 32 elements of 422 // the output vector. 423 // 424 // Asm: VPBROADCASTB, CPU Feature: AVX2 425 func (x Uint8x16) broadcast1To32() Uint8x32 426 427 // broadcast1To32 copies the lowest element of its input to all 32 elements of 428 // the output vector. 429 // 430 // Asm: VPBROADCASTW, CPU Feature: AVX512 431 func (x Uint16x8) broadcast1To32() Uint16x32 432 433 /* broadcast1To32Masked */ 434 435 // broadcast1To32Masked copies the lowest element of its input to all 32 elements of 436 // the output vector. 437 // 438 // This operation is applied selectively under a write mask. 439 // 440 // Asm: VPBROADCASTB, CPU Feature: AVX512 441 func (x Int8x16) broadcast1To32Masked(mask Mask8x16) Int8x32 442 443 // broadcast1To32Masked copies the lowest element of its input to all 32 elements of 444 // the output vector. 445 // 446 // This operation is applied selectively under a write mask. 447 // 448 // Asm: VPBROADCASTW, CPU Feature: AVX512 449 func (x Int16x8) broadcast1To32Masked(mask Mask16x8) Int16x32 450 451 // broadcast1To32Masked copies the lowest element of its input to all 32 elements of 452 // the output vector. 453 // 454 // This operation is applied selectively under a write mask. 455 // 456 // Asm: VPBROADCASTB, CPU Feature: AVX512 457 func (x Uint8x16) broadcast1To32Masked(mask Mask8x16) Uint8x32 458 459 // broadcast1To32Masked copies the lowest element of its input to all 32 elements of 460 // the output vector. 461 // 462 // This operation is applied selectively under a write mask. 463 // 464 // Asm: VPBROADCASTW, CPU Feature: AVX512 465 func (x Uint16x8) broadcast1To32Masked(mask Mask16x8) Uint16x32 466 467 /* broadcast1To64 */ 468 469 // broadcast1To64 copies the lowest element of its input to all 64 elements of 470 // the output vector. 471 // 472 // Asm: VPBROADCASTB, CPU Feature: AVX512 473 func (x Int8x16) broadcast1To64() Int8x64 474 475 // broadcast1To64 copies the lowest element of its input to all 64 elements of 476 // the output vector. 477 // 478 // Asm: VPBROADCASTB, CPU Feature: AVX512 479 func (x Uint8x16) broadcast1To64() Uint8x64 480 481 /* broadcast1To64Masked */ 482 483 // broadcast1To64Masked copies the lowest element of its input to all 64 elements of 484 // the output vector. 485 // 486 // This operation is applied selectively under a write mask. 487 // 488 // Asm: VPBROADCASTB, CPU Feature: AVX512 489 func (x Int8x16) broadcast1To64Masked(mask Mask8x16) Int8x64 490 491 // broadcast1To64Masked copies the lowest element of its input to all 64 elements of 492 // the output vector. 493 // 494 // This operation is applied selectively under a write mask. 495 // 496 // Asm: VPBROADCASTB, CPU Feature: AVX512 497 func (x Uint8x16) broadcast1To64Masked(mask Mask8x16) Uint8x64 498 499 /* carrylessMultiply */ 500 501 // carrylessMultiply computes one of four possible Galois polynomial 502 // products of selected high and low halves of x and y, 503 // depending on the value of xyHiLo, returning the 128-bit 504 // product in the concatenated two elements of the result. 505 // Bit 0 selects the low (0) or high (1) element of x and 506 // bit 4 selects the low (0x00) or high (0x10) element of y. 507 // 508 // A non-constant value of xyHiLo may result in significantly worse performance for this operation. 509 // 510 // Asm: VPCLMULQDQ, CPU Feature: AVX 511 func (x Uint64x2) carrylessMultiply(xyHiLo uint8, y Uint64x2) Uint64x2 512 513 // carrylessMultiply computes one of two possible Galois polynomial 514 // products of selected high and low halves of each of the two 515 // 128-bit lanes of x and y, depending on the value of xyHiLo, 516 // and returns the four 128-bit products in the result's lanes. 517 // Bit 0 selects the low (0) or high (1) elements of x's lanes and 518 // bit 4 selects the low (0x00) or high (0x10) elements of y's lanes. 519 // 520 // A non-constant value of xyHiLo may result in significantly worse performance for this operation. 521 // 522 // Asm: VPCLMULQDQ, CPU Feature: AVX512VPCLMULQDQ 523 func (x Uint64x4) carrylessMultiply(xyHiLo uint8, y Uint64x4) Uint64x4 524 525 // carrylessMultiply computes one of four possible Galois polynomial 526 // products of selected high and low halves of each of the four 527 // 128-bit lanes of x and y, depending on the value of xyHiLo, 528 // and returns the four 128-bit products in the result's lanes. 529 // Bit 0 selects the low (0) or high (1) elements of x's lanes and 530 // bit 4 selects the low (0x00) or high (0x10) elements of y's lanes. 531 // 532 // A non-constant value of xyHiLo may result in significantly worse performance for this operation. 533 // 534 // Asm: VPCLMULQDQ, CPU Feature: AVX512VPCLMULQDQ 535 func (x Uint64x8) carrylessMultiply(xyHiLo uint8, y Uint64x8) Uint64x8 536 537 /* concatSelectedConstant */ 538 539 // concatSelectedConstant concatenates selected elements from x and y into the lower and upper 540 // halves of the output. The selection is chosen by the constant parameter h1h0l1l0 541 // where each {h,l}{1,0} is two bits specify which element from y or x to select. 542 // For example, {0,1,2,3}.concatSelectedConstant(0b_11_01_00_10, {4,5,6,7}) returns 543 // {2, 0, 5, 7} (don't forget that the binary constant is written big-endian). 544 // 545 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 546 // 547 // Asm: VSHUFPS, CPU Feature: AVX 548 func (x Float32x4) concatSelectedConstant(h1h0l1l0 uint8, y Float32x4) Float32x4 549 550 // concatSelectedConstant concatenates selected elements from x and y into the lower and upper 551 // halves of the output. The selection is chosen by the constant parameter hilo 552 // where hi and lo are each one bit specifying which 64-bit element to select 553 // from y and x. For example {4,5}.concatSelectedConstant(0b10, {6,7}) 554 // returns {4,7}; bit 0, selecting from x, is zero, and selects 4, and bit 1, 555 // selecting from y, is 1, and selects 7. 556 // 557 // A non-constant value of hilo may result in significantly worse performance for this operation. 558 // 559 // Asm: VSHUFPD, CPU Feature: AVX 560 func (x Float64x2) concatSelectedConstant(hilo uint8, y Float64x2) Float64x2 561 562 // concatSelectedConstant concatenates selected elements from x and y into the lower and upper 563 // halves of the output. The selection is chosen by the constant parameter h1h0l1l0 564 // where each {h,l}{1,0} is two bits specify which element from y or x to select. 565 // For example, {0,1,2,3}.concatSelectedConstant(0b_11_01_00_10, {4,5,6,7}) returns 566 // {2, 0, 5, 7} (don't forget that the binary constant is written big-endian). 567 // 568 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 569 // 570 // Asm: VSHUFPS, CPU Feature: AVX 571 func (x Int32x4) concatSelectedConstant(h1h0l1l0 uint8, y Int32x4) Int32x4 572 573 // concatSelectedConstant concatenates selected elements from x and y into the lower and upper 574 // halves of the output. The selection is chosen by the constant parameter hilo 575 // where hi and lo are each one bit specifying which 64-bit element to select 576 // from y and x. For example {4,5}.concatSelectedConstant(0b10, {6,7}) 577 // returns {4,7}; bit 0, selecting from x, is zero, and selects 4, and bit 1, 578 // selecting from y, is 1, and selects 7. 579 // 580 // A non-constant value of hilo may result in significantly worse performance for this operation. 581 // 582 // Asm: VSHUFPD, CPU Feature: AVX 583 func (x Int64x2) concatSelectedConstant(hilo uint8, y Int64x2) Int64x2 584 585 // concatSelectedConstant concatenates selected elements from x and y into the lower and upper 586 // halves of the output. The selection is chosen by the constant parameter h1h0l1l0 587 // where each {h,l}{1,0} is two bits specify which element from y or x to select. 588 // For example, {0,1,2,3}.concatSelectedConstant(0b_11_01_00_10, {4,5,6,7}) returns 589 // {2, 0, 5, 7} (don't forget that the binary constant is written big-endian). 590 // 591 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 592 // 593 // Asm: VSHUFPS, CPU Feature: AVX 594 func (x Uint32x4) concatSelectedConstant(h1h0l1l0 uint8, y Uint32x4) Uint32x4 595 596 // concatSelectedConstant concatenates selected elements from x and y into the lower and upper 597 // halves of the output. The selection is chosen by the constant parameter hilo 598 // where hi and lo are each one bit specifying which 64-bit element to select 599 // from y and x. For example {4,5}.concatSelectedConstant(0b10, {6,7}) 600 // returns {4,7}; bit 0, selecting from x, is zero, and selects 4, and bit 1, 601 // selecting from y, is 1, and selects 7. 602 // 603 // A non-constant value of hilo may result in significantly worse performance for this operation. 604 // 605 // Asm: VSHUFPD, CPU Feature: AVX 606 func (x Uint64x2) concatSelectedConstant(hilo uint8, y Uint64x2) Uint64x2 607 608 /* concatSelectedConstantGrouped */ 609 610 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 611 // into the lower and upper halves of corresponding subvectors of the output. 612 // The selection is chosen by the constant parameter h1h0l1l0 613 // where each {h,l}{1,0} is two bits specifying which element from y or x to select. 614 // For example, 615 // {0,1,2,3,8,9,10,11}.concatSelectedConstantGrouped(0b_11_01_00_10, {4,5,6,7,12,13,14,15}) 616 // returns {2,0,5,7,10,8,13,15} 617 // (don't forget that the binary constant is written big-endian). 618 // 619 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 620 // 621 // Asm: VSHUFPS, CPU Feature: AVX 622 func (x Float32x8) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Float32x8) Float32x8 623 624 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 625 // into the lower and upper halves of corresponding subvectors of the output. 626 // The selection is chosen by the constant parameter h1h0l1l0 627 // where each {h,l}{1,0} is two bits specifying which element from y or x to select. 628 // For example, 629 // 630 // {0,1,2,3,8,9,10,11, 20,21,22,23,28,29,210,211}.concatSelectedConstantGrouped( 631 // 0b_11_01_00_10, {4,5,6,7,12,13,14,15, 24,25,26,27,212,213,214,215}) 632 // 633 // returns {2,0,5,7,10,8,13,15, 22,20,25,27,210,28,213,215} 634 // 635 // (don't forget that the binary constant is written big-endian). 636 // 637 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 638 // 639 // Asm: VSHUFPS, CPU Feature: AVX512 640 func (x Float32x16) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Float32x16) Float32x16 641 642 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 643 // into the lower and upper halves of corresponding subvectors of the output. 644 // The selections are specified by the constant parameter hilos where each 645 // hi and lo pair select 64-bit elements from the corresponding 128-bit 646 // subvectors of x and y. 647 // 648 // For example {4,5,8,9}.concatSelectedConstantGrouped(0b_11_10, {6,7,10,11}) 649 // returns {4,7,9,11}; bit 0 is zero, selecting element 0 from x's least 650 // 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7), 651 // then 1, selecting element 1 from x's upper 128 bits (9), then 1, 652 // selecting element 1 from y's upper 128 bits (11). 653 // This differs from the same method applied to a 32x8 vector, where 654 // the 8-bit constant performs the same selection on both subvectors. 655 // 656 // A non-constant value of hilos may result in significantly worse performance for this operation. 657 // 658 // Asm: VSHUFPD, CPU Feature: AVX 659 func (x Float64x4) concatSelectedConstantGrouped(hilos uint8, y Float64x4) Float64x4 660 661 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 662 // into the lower and upper halves of corresponding subvectors of the output. 663 // The selections are specified by the constant parameter hilos where each 664 // hi and lo pair select 64-bit elements from the corresponding 128-bit 665 // subvectors of x and y. 666 // 667 // For example {4,5,8,9,12,13,16,17}.concatSelectedConstantGrouped(0b11_00_11_10, {6,7,10,11,14,15,18,19}) 668 // returns {4,7,9,11,12,14,17,19}; bit 0 is zero, selecting element 0 from x's 669 // least 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7), 670 // then 1, selecting element 1 from x's next 128 bits (9), then 1, 671 // selecting element 1 from y's upper 128 bits (11). The next two 0 bits select 672 // the lower elements from x and y's 3rd 128 bit groups (12, 14), the last two 673 // 1 bits select the upper elements from x and y's last 128 bits (17, 19). 674 // This differs from the same method applied to a 32x8 or 32x16 vector, where 675 // the 8-bit constant performs the same selection on all the subvectors. 676 // 677 // A non-constant value of hilos may result in significantly worse performance for this operation. 678 // 679 // Asm: VSHUFPD, CPU Feature: AVX512 680 func (x Float64x8) concatSelectedConstantGrouped(hilos uint8, y Float64x8) Float64x8 681 682 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 683 // into the lower and upper halves of corresponding subvectors of the output. 684 // The selection is chosen by the constant parameter h1h0l1l0 685 // where each {h,l}{1,0} is two bits specifying which element from y or x to select. 686 // For example, 687 // {0,1,2,3,8,9,10,11}.concatSelectedConstantGrouped(0b_11_01_00_10, {4,5,6,7,12,13,14,15}) 688 // returns {2,0,5,7,10,8,13,15} 689 // (don't forget that the binary constant is written big-endian). 690 // 691 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 692 // 693 // Asm: VSHUFPS, CPU Feature: AVX 694 func (x Int32x8) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Int32x8) Int32x8 695 696 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 697 // into the lower and upper halves of corresponding subvectors of the output. 698 // The selection is chosen by the constant parameter h1h0l1l0 699 // where each {h,l}{1,0} is two bits specifying which element from y or x to select. 700 // For example, 701 // 702 // {0,1,2,3,8,9,10,11, 20,21,22,23,28,29,210,211}.concatSelectedConstantGrouped( 703 // 0b_11_01_00_10, {4,5,6,7,12,13,14,15, 24,25,26,27,212,213,214,215}) 704 // 705 // returns {2,0,5,7,10,8,13,15, 22,20,25,27,210,28,213,215} 706 // 707 // (don't forget that the binary constant is written big-endian). 708 // 709 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 710 // 711 // Asm: VSHUFPS, CPU Feature: AVX512 712 func (x Int32x16) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Int32x16) Int32x16 713 714 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 715 // into the lower and upper halves of corresponding subvectors of the output. 716 // The selections are specified by the constant parameter hilos where each 717 // hi and lo pair select 64-bit elements from the corresponding 128-bit 718 // subvectors of x and y. 719 // 720 // For example {4,5,8,9}.concatSelectedConstantGrouped(0b_11_10, {6,7,10,11}) 721 // returns {4,7,9,11}; bit 0 is zero, selecting element 0 from x's least 722 // 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7), 723 // then 1, selecting element 1 from x's upper 128 bits (9), then 1, 724 // selecting element 1 from y's upper 128 bits (11). 725 // This differs from the same method applied to a 32x8 vector, where 726 // the 8-bit constant performs the same selection on both subvectors. 727 // 728 // A non-constant value of hilos may result in significantly worse performance for this operation. 729 // 730 // Asm: VSHUFPD, CPU Feature: AVX 731 func (x Int64x4) concatSelectedConstantGrouped(hilos uint8, y Int64x4) Int64x4 732 733 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 734 // into the lower and upper halves of corresponding subvectors of the output. 735 // The selections are specified by the constant parameter hilos where each 736 // hi and lo pair select 64-bit elements from the corresponding 128-bit 737 // subvectors of x and y. 738 // 739 // For example {4,5,8,9,12,13,16,17}.concatSelectedConstantGrouped(0b11_00_11_10, {6,7,10,11,14,15,18,19}) 740 // returns {4,7,9,11,12,14,17,19}; bit 0 is zero, selecting element 0 from x's 741 // least 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7), 742 // then 1, selecting element 1 from x's next 128 bits (9), then 1, 743 // selecting element 1 from y's upper 128 bits (11). The next two 0 bits select 744 // the lower elements from x and y's 3rd 128 bit groups (12, 14), the last two 745 // 1 bits select the upper elements from x and y's last 128 bits (17, 19). 746 // This differs from the same method applied to a 32x8 or 32x16 vector, where 747 // the 8-bit constant performs the same selection on all the subvectors. 748 // 749 // A non-constant value of hilos may result in significantly worse performance for this operation. 750 // 751 // Asm: VSHUFPD, CPU Feature: AVX512 752 func (x Int64x8) concatSelectedConstantGrouped(hilos uint8, y Int64x8) Int64x8 753 754 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 755 // into the lower and upper halves of corresponding subvectors of the output. 756 // The selection is chosen by the constant parameter h1h0l1l0 757 // where each {h,l}{1,0} is two bits specifying which element from y or x to select. 758 // For example, 759 // {0,1,2,3,8,9,10,11}.concatSelectedConstantGrouped(0b_11_01_00_10, {4,5,6,7,12,13,14,15}) 760 // returns {2,0,5,7,10,8,13,15} 761 // (don't forget that the binary constant is written big-endian). 762 // 763 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 764 // 765 // Asm: VSHUFPS, CPU Feature: AVX 766 func (x Uint32x8) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Uint32x8) Uint32x8 767 768 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 769 // into the lower and upper halves of corresponding subvectors of the output. 770 // The selection is chosen by the constant parameter h1h0l1l0 771 // where each {h,l}{1,0} is two bits specifying which element from y or x to select. 772 // For example, 773 // 774 // {0,1,2,3,8,9,10,11, 20,21,22,23,28,29,210,211}.concatSelectedConstantGrouped( 775 // 0b_11_01_00_10, {4,5,6,7,12,13,14,15, 24,25,26,27,212,213,214,215}) 776 // 777 // returns {2,0,5,7,10,8,13,15, 22,20,25,27,210,28,213,215} 778 // 779 // (don't forget that the binary constant is written big-endian). 780 // 781 // A non-constant value of h1h0l1l0 may result in significantly worse performance for this operation. 782 // 783 // Asm: VSHUFPS, CPU Feature: AVX512 784 func (x Uint32x16) concatSelectedConstantGrouped(h1h0l1l0 uint8, y Uint32x16) Uint32x16 785 786 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 787 // into the lower and upper halves of corresponding subvectors of the output. 788 // The selections are specified by the constant parameter hilos where each 789 // hi and lo pair select 64-bit elements from the corresponding 128-bit 790 // subvectors of x and y. 791 // 792 // For example {4,5,8,9}.concatSelectedConstantGrouped(0b_11_10, {6,7,10,11}) 793 // returns {4,7,9,11}; bit 0 is zero, selecting element 0 from x's least 794 // 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7), 795 // then 1, selecting element 1 from x's upper 128 bits (9), then 1, 796 // selecting element 1 from y's upper 128 bits (11). 797 // This differs from the same method applied to a 32x8 vector, where 798 // the 8-bit constant performs the same selection on both subvectors. 799 // 800 // A non-constant value of hilos may result in significantly worse performance for this operation. 801 // 802 // Asm: VSHUFPD, CPU Feature: AVX 803 func (x Uint64x4) concatSelectedConstantGrouped(hilos uint8, y Uint64x4) Uint64x4 804 805 // concatSelectedConstantGrouped concatenates selected elements from 128-bit subvectors of x and y 806 // into the lower and upper halves of corresponding subvectors of the output. 807 // The selections are specified by the constant parameter hilos where each 808 // hi and lo pair select 64-bit elements from the corresponding 128-bit 809 // subvectors of x and y. 810 // 811 // For example {4,5,8,9,12,13,16,17}.concatSelectedConstantGrouped(0b11_00_11_10, {6,7,10,11,14,15,18,19}) 812 // returns {4,7,9,11,12,14,17,19}; bit 0 is zero, selecting element 0 from x's 813 // least 128-bits (4), then 1, selects the element 1 from y's least 128-bits (7), 814 // then 1, selecting element 1 from x's next 128 bits (9), then 1, 815 // selecting element 1 from y's upper 128 bits (11). The next two 0 bits select 816 // the lower elements from x and y's 3rd 128 bit groups (12, 14), the last two 817 // 1 bits select the upper elements from x and y's last 128 bits (17, 19). 818 // This differs from the same method applied to a 32x8 or 32x16 vector, where 819 // the 8-bit constant performs the same selection on all the subvectors. 820 // 821 // A non-constant value of hilos may result in significantly worse performance for this operation. 822 // 823 // Asm: VSHUFPD, CPU Feature: AVX512 824 func (x Uint64x8) concatSelectedConstantGrouped(hilos uint8, y Uint64x8) Uint64x8 825 826 /* permuteScalars */ 827 828 // permuteScalars performs a permutation of vector x using constant indices: 829 // 830 // result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]]} 831 // 832 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 833 // 834 // A non-constant value of indices may result in significantly worse performance for this operation. 835 // 836 // Asm: VPSHUFD, CPU Feature: AVX 837 func (x Int32x4) permuteScalars(indices uint8) Int32x4 838 839 // permuteScalars performs a permutation of vector x using constant indices: 840 // 841 // result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]]} 842 // 843 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 844 // 845 // A non-constant value of indices may result in significantly worse performance for this operation. 846 // 847 // Asm: VPSHUFD, CPU Feature: AVX 848 func (x Uint32x4) permuteScalars(indices uint8) Uint32x4 849 850 /* permuteScalarsGrouped */ 851 852 // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices: 853 // 854 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...} 855 // 856 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 857 // Each group is of size 128-bit. 858 // 859 // A non-constant value of indices may result in significantly worse performance for this operation. 860 // 861 // Asm: VPSHUFD, CPU Feature: AVX2 862 func (x Int32x8) permuteScalarsGrouped(indices uint8) Int32x8 863 864 // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices: 865 // 866 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...} 867 // 868 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 869 // Each group is of size 128-bit. 870 // 871 // A non-constant value of indices may result in significantly worse performance for this operation. 872 // 873 // Asm: VPSHUFD, CPU Feature: AVX512 874 func (x Int32x16) permuteScalarsGrouped(indices uint8) Int32x16 875 876 // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices: 877 // 878 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...} 879 // 880 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 881 // Each group is of size 128-bit. 882 // 883 // A non-constant value of indices may result in significantly worse performance for this operation. 884 // 885 // Asm: VPSHUFD, CPU Feature: AVX2 886 func (x Uint32x8) permuteScalarsGrouped(indices uint8) Uint32x8 887 888 // permuteScalarsGrouped performs a grouped permutation of vector x using constant indices: 889 // 890 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x_group1[indices[0:2]], ...} 891 // 892 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 893 // Each group is of size 128-bit. 894 // 895 // A non-constant value of indices may result in significantly worse performance for this operation. 896 // 897 // Asm: VPSHUFD, CPU Feature: AVX512 898 func (x Uint32x16) permuteScalarsGrouped(indices uint8) Uint32x16 899 900 /* permuteScalarsHi */ 901 902 // permuteScalarsHi performs a permutation of vector x using constant indices: 903 // 904 // result = {x[0], x[1], x[2], x[3], x[indices[0:2]+4], x[indices[2:4]+4], x[indices[4:6]+4], x[indices[6:8]+4]} 905 // 906 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 907 // 908 // A non-constant value of indices may result in significantly worse performance for this operation. 909 // 910 // Asm: VPSHUFHW, CPU Feature: AVX 911 func (x Int16x8) permuteScalarsHi(indices uint8) Int16x8 912 913 // permuteScalarsHi performs a permutation of vector x using constant indices: 914 // 915 // result = {x[0], x[1], x[2], x[3], x[indices[0:2]+4], x[indices[2:4]+4], x[indices[4:6]+4], x[indices[6:8]+4]} 916 // 917 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 918 // 919 // A non-constant value of indices may result in significantly worse performance for this operation. 920 // 921 // Asm: VPSHUFHW, CPU Feature: AVX 922 func (x Uint16x8) permuteScalarsHi(indices uint8) Uint16x8 923 924 /* permuteScalarsHiGrouped */ 925 926 // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices: 927 // result = 928 // 929 // {x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4], 930 // x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...} 931 // 932 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 933 // Each group is of size 128-bit. 934 // 935 // A non-constant value of indices may result in significantly worse performance for this operation. 936 // 937 // Asm: VPSHUFHW, CPU Feature: AVX2 938 func (x Int16x16) permuteScalarsHiGrouped(indices uint8) Int16x16 939 940 // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices: 941 // result = 942 // 943 // {x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4], 944 // x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...} 945 // 946 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 947 // Each group is of size 128-bit. 948 // 949 // A non-constant value of indices may result in significantly worse performance for this operation. 950 // 951 // Asm: VPSHUFHW, CPU Feature: AVX512 952 func (x Int16x32) permuteScalarsHiGrouped(indices uint8) Int16x32 953 954 // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices: 955 // result = 956 // 957 // {x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4], 958 // x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...} 959 // 960 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 961 // Each group is of size 128-bit. 962 // 963 // A non-constant value of indices may result in significantly worse performance for this operation. 964 // 965 // Asm: VPSHUFHW, CPU Feature: AVX2 966 func (x Uint16x16) permuteScalarsHiGrouped(indices uint8) Uint16x16 967 968 // permuteScalarsHiGrouped performs a grouped permutation of vector x using constant indices: 969 // result = 970 // 971 // {x_group0[0], x_group0[1], x_group0[2], x_group0[3], x_group0[indices[0:2]+4], x_group0[indices[2:4]+4], x_group0[indices[4:6]+4], x_group0[indices[6:8]+4], 972 // x_group1[0], x_group1[1], x_group1[2], x_group1[3], x_group1[indices[0:2]+4], ...} 973 // 974 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 975 // Each group is of size 128-bit. 976 // 977 // A non-constant value of indices may result in significantly worse performance for this operation. 978 // 979 // Asm: VPSHUFHW, CPU Feature: AVX512 980 func (x Uint16x32) permuteScalarsHiGrouped(indices uint8) Uint16x32 981 982 /* permuteScalarsLo */ 983 984 // permuteScalarsLo performs a permutation of vector x using constant indices: 985 // 986 // result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]], x[4], x[5], x[6], x[7]} 987 // 988 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 989 // 990 // A non-constant value of indices may result in significantly worse performance for this operation. 991 // 992 // Asm: VPSHUFLW, CPU Feature: AVX 993 func (x Int16x8) permuteScalarsLo(indices uint8) Int16x8 994 995 // permuteScalarsLo performs a permutation of vector x using constant indices: 996 // 997 // result = {x[indices[0:2]], x[indices[2:4]], x[indices[4:6]], x[indices[6:8]], x[4], x[5], x[6], x[7]} 998 // 999 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 1000 // 1001 // A non-constant value of indices may result in significantly worse performance for this operation. 1002 // 1003 // Asm: VPSHUFLW, CPU Feature: AVX 1004 func (x Uint16x8) permuteScalarsLo(indices uint8) Uint16x8 1005 1006 /* permuteScalarsLoGrouped */ 1007 1008 // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices: 1009 // 1010 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7], 1011 // x_group1[indices[0:2]], ...} 1012 // 1013 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 1014 // Each group is of size 128-bit. 1015 // 1016 // A non-constant value of indices may result in significantly worse performance for this operation. 1017 // 1018 // Asm: VPSHUFLW, CPU Feature: AVX2 1019 func (x Int16x16) permuteScalarsLoGrouped(indices uint8) Int16x16 1020 1021 // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices: 1022 // 1023 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7], 1024 // x_group1[indices[0:2]], ...} 1025 // 1026 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 1027 // Each group is of size 128-bit. 1028 // 1029 // A non-constant value of indices may result in significantly worse performance for this operation. 1030 // 1031 // Asm: VPSHUFLW, CPU Feature: AVX512 1032 func (x Int16x32) permuteScalarsLoGrouped(indices uint8) Int16x32 1033 1034 // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices: 1035 // 1036 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7], 1037 // x_group1[indices[0:2]], ...} 1038 // 1039 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 1040 // Each group is of size 128-bit. 1041 // 1042 // A non-constant value of indices may result in significantly worse performance for this operation. 1043 // 1044 // Asm: VPSHUFLW, CPU Feature: AVX2 1045 func (x Uint16x16) permuteScalarsLoGrouped(indices uint8) Uint16x16 1046 1047 // permuteScalarsLoGrouped performs a grouped permutation of vector x using constant indices: 1048 // 1049 // result = {x_group0[indices[0:2]], x_group0[indices[2:4]], x_group0[indices[4:6]], x_group0[indices[6:8]], x[4], x[5], x[6], x[7], 1050 // x_group1[indices[0:2]], ...} 1051 // 1052 // Indices is four 2-bit values packed into a byte, thus indices[0:2] is the first index. 1053 // Each group is of size 128-bit. 1054 // 1055 // A non-constant value of indices may result in significantly worse performance for this operation. 1056 // 1057 // Asm: VPSHUFLW, CPU Feature: AVX512 1058 func (x Uint16x32) permuteScalarsLoGrouped(indices uint8) Uint16x32 1059 1060 /* tern */ 1061 1062 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1063 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1064 // 1065 // A non-constant value of table may result in significantly worse performance for this operation. 1066 // 1067 // Asm: VPTERNLOGD, CPU Feature: AVX512 1068 func (x Int32x4) tern(table uint8, y Int32x4, z Int32x4) Int32x4 1069 1070 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1071 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1072 // 1073 // A non-constant value of table may result in significantly worse performance for this operation. 1074 // 1075 // Asm: VPTERNLOGD, CPU Feature: AVX512 1076 func (x Int32x8) tern(table uint8, y Int32x8, z Int32x8) Int32x8 1077 1078 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1079 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1080 // 1081 // A non-constant value of table may result in significantly worse performance for this operation. 1082 // 1083 // Asm: VPTERNLOGD, CPU Feature: AVX512 1084 func (x Int32x16) tern(table uint8, y Int32x16, z Int32x16) Int32x16 1085 1086 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1087 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1088 // 1089 // A non-constant value of table may result in significantly worse performance for this operation. 1090 // 1091 // Asm: VPTERNLOGQ, CPU Feature: AVX512 1092 func (x Int64x2) tern(table uint8, y Int64x2, z Int64x2) Int64x2 1093 1094 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1095 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1096 // 1097 // A non-constant value of table may result in significantly worse performance for this operation. 1098 // 1099 // Asm: VPTERNLOGQ, CPU Feature: AVX512 1100 func (x Int64x4) tern(table uint8, y Int64x4, z Int64x4) Int64x4 1101 1102 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1103 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1104 // 1105 // A non-constant value of table may result in significantly worse performance for this operation. 1106 // 1107 // Asm: VPTERNLOGQ, CPU Feature: AVX512 1108 func (x Int64x8) tern(table uint8, y Int64x8, z Int64x8) Int64x8 1109 1110 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1111 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1112 // 1113 // A non-constant value of table may result in significantly worse performance for this operation. 1114 // 1115 // Asm: VPTERNLOGD, CPU Feature: AVX512 1116 func (x Uint32x4) tern(table uint8, y Uint32x4, z Uint32x4) Uint32x4 1117 1118 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1119 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1120 // 1121 // A non-constant value of table may result in significantly worse performance for this operation. 1122 // 1123 // Asm: VPTERNLOGD, CPU Feature: AVX512 1124 func (x Uint32x8) tern(table uint8, y Uint32x8, z Uint32x8) Uint32x8 1125 1126 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1127 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1128 // 1129 // A non-constant value of table may result in significantly worse performance for this operation. 1130 // 1131 // Asm: VPTERNLOGD, CPU Feature: AVX512 1132 func (x Uint32x16) tern(table uint8, y Uint32x16, z Uint32x16) Uint32x16 1133 1134 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1135 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1136 // 1137 // A non-constant value of table may result in significantly worse performance for this operation. 1138 // 1139 // Asm: VPTERNLOGQ, CPU Feature: AVX512 1140 func (x Uint64x2) tern(table uint8, y Uint64x2, z Uint64x2) Uint64x2 1141 1142 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1143 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1144 // 1145 // A non-constant value of table may result in significantly worse performance for this operation. 1146 // 1147 // Asm: VPTERNLOGQ, CPU Feature: AVX512 1148 func (x Uint64x4) tern(table uint8, y Uint64x4, z Uint64x4) Uint64x4 1149 1150 // tern performs a logical operation on three vectors based on the 8-bit truth table. 1151 // Bitwise, the result is equal to 1 & (table >> (x<<2 + y<<1 + z)) 1152 // 1153 // A non-constant value of table may result in significantly worse performance for this operation. 1154 // 1155 // Asm: VPTERNLOGQ, CPU Feature: AVX512 1156 func (x Uint64x8) tern(table uint8, y Uint64x8, z Uint64x8) Uint64x8 1157