/ 128-bit load over 64-bit stride
| 373 | |
| 374 | //// 128-bit load over 64-bit stride |
| 375 | NPY_FINLINE npyv_s64 npyv_loadn2_till_s64(const npy_int64 *ptr, npy_intp stride, npy_uintp nlane, |
| 376 | npy_int64 fill_lo, npy_int64 fill_hi) |
| 377 | { |
| 378 | assert(nlane > 0); |
| 379 | __m256i a = npyv_loadl_s64(ptr); |
| 380 | #if defined(_MSC_VER) && defined(_M_IX86) |
| 381 | __m128i fill =_mm_setr_epi32( |
| 382 | (int)fill_lo, (int)(fill_lo >> 32), |
| 383 | (int)fill_hi, (int)(fill_hi >> 32) |
| 384 | ); |
| 385 | #else |
| 386 | __m128i fill = _mm_set_epi64x(fill_hi, fill_lo); |
| 387 | #endif |
| 388 | __m128i b = nlane > 1 ? _mm_loadu_si128((const __m128i*)(ptr + stride)) : fill; |
| 389 | __m256i ret = _mm256_inserti128_si256(a, b, 1); |
| 390 | #if NPY_SIMD_GUARD_PARTIAL_LOAD |
| 391 | volatile __m256i workaround = ret; |
| 392 | ret = _mm256_or_si256(workaround, ret); |
| 393 | #endif |
| 394 | return ret; |
| 395 | } |
| 396 | // fill zero to rest lanes |
| 397 | NPY_FINLINE npyv_s64 npyv_loadn2_tillz_s64(const npy_int64 *ptr, npy_intp stride, npy_uintp nlane) |
| 398 | { return npyv_loadn2_till_s64(ptr, stride, nlane, 0, 0); } |
no outgoing calls
no test coverage detected