* Non-contiguous partial load *********************************/ / 32
| 305 | *********************************/ |
| 306 | //// 32 |
| 307 | NPY_FINLINE npyv_s32 |
| 308 | npyv_loadn_till_s32(const npy_int32 *ptr, npy_intp stride, npy_uintp nlane, npy_int32 fill) |
| 309 | { |
| 310 | assert(nlane > 0); |
| 311 | assert(llabs(stride) <= NPY_SIMD_MAXLOAD_STRIDE32); |
| 312 | const __m256i vfill = _mm256_set1_epi32(fill); |
| 313 | const __m256i steps = _mm256_setr_epi32(0, 1, 2, 3, 4, 5, 6, 7); |
| 314 | const __m256i idx = _mm256_mullo_epi32(_mm256_set1_epi32((int)stride), steps); |
| 315 | __m256i vnlane = _mm256_set1_epi32(nlane > 8 ? 8 : (int)nlane); |
| 316 | __m256i mask = _mm256_cmpgt_epi32(vnlane, steps); |
| 317 | __m256i ret = _mm256_mask_i32gather_epi32(vfill, (const int*)ptr, idx, mask, 4); |
| 318 | #if NPY_SIMD_GUARD_PARTIAL_LOAD |
| 319 | volatile __m256i workaround = ret; |
| 320 | ret = _mm256_or_si256(workaround, ret); |
| 321 | #endif |
| 322 | return ret; |
| 323 | } |
| 324 | // fill zero to rest lanes |
| 325 | NPY_FINLINE npyv_s32 |
| 326 | npyv_loadn_tillz_s32(const npy_int32 *ptr, npy_intp stride, npy_uintp nlane) |
no outgoing calls
no test coverage detected