zeroshade commented on code in PR #1348:
URL: https://github.com/apache/arrow-go/pull/1348#discussion_r4199537145


##########
arrow/compute/internal/kernels/_lib/get_take_indices_avx2_amd64.s:
##########
@@ -0,0 +1,136 @@
+       .intel_syntax noprefix
+       .file   "get_take_indices_avx2_amd64.cc"
+       .text
+       .globl  get_take_indices_uint32_avx2
+       .p2align        4
+       .type   get_take_indices_uint32_avx2,@function
+get_take_indices_uint32_avx2:           # @get_take_indices_uint32_avx2
+# %bb.0:
+       push    r14
+       push    rbx
+       test    rcx, rcx
+       jle     .LBB0_10
+# %bb.1:
+       dec     rcx
+       je      .LBB0_2
+# %bb.11:
+       vpxor   xmm0, xmm0, xmm0
+       vmovdqu xmm1, xmmword ptr [rdx + 384]
+       xor     eax, eax
+       vmovdqu xmm2, xmmword ptr [rdx + 352]
+       vmovdqu xmm3, xmmword ptr [rdx + 368]
+       xor     r9d, r9d
+       jmp     .LBB0_12
+       .p2align        4
+.LBB0_14:                               #   in Loop: Header=BB0_12 Depth=1
+       vpor    xmm4, xmm0, xmm2
+       vpor    xmm5, xmm0, xmm3
+       vmovdqu xmmword ptr [rsi + 4*rax], xmm4
+       vmovdqu xmmword ptr [rsi + 4*rax + 16], xmm5
+       add     rax, 8
+.LBB0_19:                               #   in Loop: Header=BB0_12 Depth=1
+       vpaddd  xmm0, xmm0, xmm1
+       inc     r9
+       cmp     rcx, r9
+       je      .LBB0_3
+.LBB0_12:                               # =>This Inner Loop Header: Depth=1
+       movzx   r10d, byte ptr [rdi + r9]
+       test    r10d, r10d
+       je      .LBB0_19
+# %bb.13:                               #   in Loop: Header=BB0_12 Depth=1
+       cmp     r10d, 255
+       je      .LBB0_14
+# %bb.15:                               #   in Loop: Header=BB0_12 Depth=1
+       mov     r11d, r10d
+       and     r11d, 15
+       mov     r14d, r10d
+       shr     r14d, 4
+       movzx   ebx, byte ptr [rdx + r11 + 336]
+       movzx   r11d, byte ptr [rdx + r14 + 336]
+       test    rbx, rbx
+       je      .LBB0_17
+# %bb.16:                               #   in Loop: Header=BB0_12 Depth=1
+       mov     r14d, r10d

Review Comment:
   Regenerating from the `.cc` puts `ebp` here (`mov ebp, r10d; shl bpl, 4`), 
and `vpaddd` instead of `vpor` at L26/27/56. This file looks hand-edited.



##########
arrow/compute/internal/kernels/Makefile:
##########
@@ -76,6 +76,9 @@ _lib/scalar_comparison_sse4_amd64.s: _lib/scalar_comparison.cc
 _lib/filter_uint32_avx2_amd64.s: _lib/filter_uint32.cc
        $(CXX) -std=c++17 -S $(C_FLAGS) $(ASM_FLAGS_AVX2) $^ -o $@ ; 
$(PERL_FIXUP_ROTATE) $@
 
+_lib/get_take_indices_avx2_amd64.s: _lib/get_take_indices_avx2_amd64.cc

Review Comment:
   This recipe doesn't reproduce the committed `_lib` output. Please make sure 
`make` round-trips to the checked-in files.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to