[x265] [PATCH 6 of 8] asm: intra_pred_ang4_7_sse2 improved ~1% 614.99 -> 607.50 with nits and tweaks

dtyx265 at gmail.com dtyx265 at gmail.com
Sat Mar 28 22:35:24 CET 2015


# HG changeset patch
# User David T Yuen <dtyx265 at gmail.com>
# Date 1427577671 25200
# Node ID 3b2a26ca89555b3b9929bbfdb451d715b2a0bfbe
# Parent  befc4da18fcf11e471a17bb5a58a14d8c6803162
asm: intra_pred_ang4_7_sse2 improved ~1% 614.99 -> 607.50 with nits and tweaks

Corrected parameter count
Changed r3 and r4 to r3d and r4d
tweaked unpacking for performance

diff -r befc4da18fcf -r 3b2a26ca8955 source/common/x86/intrapred8.asm
--- a/source/common/x86/intrapred8.asm	Sat Mar 28 14:14:42 2015 -0700
+++ b/source/common/x86/intrapred8.asm	Sat Mar 28 14:21:11 2015 -0700
@@ -1483,21 +1483,21 @@
     mova        m7, [r3 +  1 * 16]  ; [20]
     jmp         mangle(private_prefix %+ _ %+ intra_pred_ang4_3 %+ SUFFIX %+ .do_filter4x4)
 
-cglobal intra_pred_ang4_7, 3,5,8
-    xor         r4, r4
-    inc         r4
+cglobal intra_pred_ang4_7, 4,5,8
+    xor         r4d, r4d
+    inc         r4d
     cmp         r3m, byte 29
-    mov         r3, 9
-    cmove       r3, r4
+    mov         r3d, 9
+    cmove       r3d, r4d
 
     movh        m0, [r2 + r3]    ; [8 7 6 5 4 3 2 1]
-    mova        m1, m0
-    psrldq      m1, 1           ; [x 8 7 6 5 4 3 2]
-    punpcklbw   m0, m1          ; [x 8 8 7 7 6 6 5 5 4 4 3 3 2 2 1]
-    mova        m3, m0
-    psrldq      m3, 2           ; [x x x x x x x x 6 5 5 4 4 3 3 2]
-    punpcklqdq  m2, m0, m3
+    punpcklbw   m0, m0
+    psrldq      m0, 1
+    mova        m2, m0
+    psrldq      m2, 2           ; [x x x x x x x x 6 5 5 4 4 3 3 2]
     punpcklqdq  m0, m0
+    punpcklqdq  m2, m2
+    movhlps     m2, m0
 
     lea         r3, [pw_ang_table + 20 * 16]
     mova        m4, [r3 - 11 * 16]  ; [ 9]


More information about the x265-devel mailing list