summaryrefslogtreecommitdiff
path: root/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S
diff options
context:
space:
mode:
Diffstat (limited to 'xc/extras/Mesa/src/X86/3dnow_xform_masked3.S')
-rw-r--r--xc/extras/Mesa/src/X86/3dnow_xform_masked3.S916
1 files changed, 448 insertions, 468 deletions
diff --git a/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S b/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S
index fbc66c34b..45686e03f 100644
--- a/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S
+++ b/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S
@@ -1,729 +1,709 @@
+
+/*
+ * Mesa 3-D graphics library
+ * Version: 3.4
+ *
+ * Copyright (C) 1999-2000 Brian Paul All Rights Reserved.
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a
+ * copy of this software and associated documentation files (the "Software"),
+ * to deal in the Software without restriction, including without limitation
+ * the rights to use, copy, modify, merge, publish, distribute, sublicense,
+ * and/or sell copies of the Software, and to permit persons to whom the
+ * Software is furnished to do so, subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included
+ * in all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
+ * OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+ * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
+ * BRIAN PAUL BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
+ * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+ */
+
#include "assyntax.h"
+#include "xform_args.h"
+
+ SEG_TEXT
- SEG_TEXT
+#define FRAME_OFFSET 16
ALIGNTEXT16
-GLOBL GLNAME(gl_3dnow_transform_points3_general_masked)
+GLOBL GLNAME( gl_3dnow_transform_points3_general_masked )
GLNAME( gl_3dnow_transform_points3_general_masked ):
PUSH_L ( ESI )
- MOV_L ( REGOFF(8, ESP), ECX )
- MOV_L ( REGOFF(12, ESP), ESI )
- MOV_L ( REGOFF(16, ESP), EAX )
- MOV_L ( CONST(4), REGOFF(16, ECX) )
- OR_B ( CONST(15), REGOFF(20, ECX) )
- MOV_L ( REGOFF(8, EAX), EDX )
- MOV_L ( EDX, REGOFF(8, ECX) )
-
-ALIGNTEXT32
-
- PUSH_L ( ESI )
PUSH_L ( EDI )
PUSH_L ( EBX )
PUSH_L ( EBP )
- MOV_L ( REGOFF(4, ECX), EDX )
- MOV_L ( ESI, ECX )
- MOV_L ( REGOFF(8, EAX), ESI ) /* count */
- MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */
- MOV_L ( REGOFF(4, EAX), EAX )
- MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */
- MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */
+ MOV_L ( ARG_DEST, ECX )
+ MOV_L ( ARG_MATRIX, ESI )
+ MOV_L ( ARG_SOURCE, EAX )
+ MOV_L ( CONST(4), REGOFF(V4F_SIZE, ECX) )
+ OR_B ( CONST(VEC_SIZE_4), REGOFF(V4F_FLAGS, ECX) )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), EDX )
+ MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) )
- FEMMS
+ MOV_L ( REGOFF(V4F_START, ECX), EDX )
+ MOV_L ( ESI, ECX )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), ESI )
+ MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI )
+ MOV_L ( REGOFF(V4F_START, EAX), EAX )
+ MOV_L ( ARG_CLIP, EBP )
+ MOV_B ( ARG_FLAG, BL )
- MOVD ( REGIND(ECX), MM0 ) /* | m00 */
- MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */
+ MOVD ( REGIND(ECX), MM0 ) /* | m00 */
+ MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */
- PSLLQ ( CONST(32), MM7 ) /* m10 | */
- POR ( MM7, MM0 ) /* m10 | m00 */
+ PSLLQ ( CONST(32), MM7 ) /* m10 | */
+ POR ( MM7, MM0 ) /* m10 | m00 */
- MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */
- MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
+ MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */
+ MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
- PSLLQ ( CONST(32), MM7 ) /* m11 | */
- POR ( MM7, MM1 ) /* m11 | m01 */
+ PSLLQ ( CONST(32), MM7 ) /* m11 | */
+ POR ( MM7, MM1 ) /* m11 | m01 */
- MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */
- MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */
+ MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */
+ MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */
- CMP_L ( CONST(0), ESI )
- JE ( LLBL(G3TPGM_6) )
+ TEST_L ( ESI, ESI )
+ JZ ( LLBL( G3TPGM_6 ) )
PUSH_L ( EBP )
PUSH_L ( EAX )
PUSH_L ( EDX )
PUSH_L ( ESI )
+ALIGNTEXT16
+LLBL( G3TPGM_2 ):
-ALIGNTEXT32
-LLBL(G3TPGM_2):
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TPGM_3 ) ) /* skip vertex */
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TPGM_3) /* skip vertex */ )
+ MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
+ MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
- MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
- MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
+ MOVQ ( MM4, MM5 ) /* x1 | x0 */
+ PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */
- MOVQ ( MM4, MM5 ) /* x1 | x0 */
- PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */
+ PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */
+ PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */
- PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */
- PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */
+ PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */
+ PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */
- PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */
- PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */
+ PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */
+ PFADD ( MM4, MM6 ) /* r1 | r0 */
- PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */
- PFADD ( MM4, MM6 ) /* r1 | r0 */
+ MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */
- MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */
+ALIGNTEXT16
+LLBL( G3TPGM_3 ):
-LLBL(G3TPGM_3):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
- JA ( LLBL(G3TPGM_2) /* cnt > 0 ? -> process next vertex */ )
- /* and now the second stripe ... */
+ JNZ ( LLBL( G3TPGM_2 ) ) /* cnt > 0 ? -> process next vertex */
- POP_L ( ESI ) /* reset counter & pointers */
+ /* and now the second stripe ... */
+ POP_L ( ESI ) /* reset counter & pointers */
POP_L ( EDX )
POP_L ( EAX )
POP_L ( EBP )
- MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */
- MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */
+ MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */
+ MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */
- PSLLQ ( CONST(32), MM7 ) /* m12 | */
- POR ( MM7, MM0 ) /* m12 | m02 */
+ PSLLQ ( CONST(32), MM7 ) /* m12 | */
+ POR ( MM7, MM0 ) /* m12 | m02 */
- MOVD ( REGOFF(12, ECX), MM1 ) /* | m03 */
- MOVD ( REGOFF(28, ECX), MM7 ) /* | m13 */
+ MOVD ( REGOFF(12, ECX), MM1 ) /* | m03 */
+ MOVD ( REGOFF(28, ECX), MM7 ) /* | m13 */
- PSLLQ ( CONST(32), MM7 ) /* m13 | */
- POR ( MM7, MM1 ) /* m13 | m03 */
+ PSLLQ ( CONST(32), MM7 ) /* m13 | */
+ POR ( MM7, MM1 ) /* m13 | m03 */
- MOVQ ( REGOFF(40, ECX), MM2 ) /* m23 | m22 */
- MOVQ ( REGOFF(56, ECX), MM3 ) /* m33 | m32 */
+ MOVQ ( REGOFF(40, ECX), MM2 ) /* m23 | m22 */
+ MOVQ ( REGOFF(56, ECX), MM3 ) /* m33 | m32 */
+ALIGNTEXT16
+LLBL( G3TPGM_4 ):
-ALIGNTEXT32
-LLBL(G3TPGM_4):
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TPGM_5) /* skip vertex */ )
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TPGM_5 ) ) /* skip vertex */
- MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
- MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
+ MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
+ MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
- MOVQ ( MM4, MM5 ) /* x1 | x0 */
- PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */
+ MOVQ ( MM4, MM5 ) /* x1 | x0 */
+ PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */
- PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */
- PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */
+ PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */
+ PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */
- PFMUL ( MM2, MM6 ) /* x2*m23 | x2*m22 */
- PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */
+ PFMUL ( MM2, MM6 ) /* x2*m23 | x2*m22 */
+ PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */
- PFADD ( MM3, MM6 ) /* x2*m23+m33 | x2*m22+m32 */
- PFADD ( MM4, MM6 ) /* r3 | r2 */
+ PFADD ( MM3, MM6 ) /* x2*m23+m33 | x2*m22+m32 */
+ PFADD ( MM4, MM6 ) /* r3 | r2 */
- MOVQ ( MM6, REGOFF(8, EDX) ) /* write r2, r3 */
+ MOVQ ( MM6, REGOFF(8, EDX) ) /* write r2, r3 */
-LLBL(G3TPGM_5):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ALIGNTEXT16
+LLBL( G3TPGM_5 ):
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
- JA ( LLBL(G3TPGM_4) /* cnt > 0 ? -> process next vertex */ )
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
-LLBL(G3TPGM_6):
+ JNZ ( LLBL( G3TPGM_4 ) ) /* cnt > 0 ? -> process next vertex */
- FEMMS
+LLBL( G3TPGM_6 ):
+ FEMMS
POP_L ( EBP )
POP_L ( EBX )
POP_L ( EDI )
POP_L ( ESI )
-
- POP_L ( ESI )
RET
-
-
ALIGNTEXT16
-GLOBL GLNAME(gl_3dnow_transform_points3_identity_masked)
-GLNAME( gl_3dnow_transform_points3_identity_masked ):
-
- PUSH_L ( ESI )
- MOV_L ( REGOFF(8, ESP), ECX )
- MOV_L ( REGOFF(12, ESP), ESI )
- MOV_L ( REGOFF(16, ESP), EAX )
- MOV_L ( CONST(3), REGOFF(16, ECX) )
- OR_B ( CONST(7), REGOFF(20, ECX) )
- MOV_L ( REGOFF(8, EAX), EDX )
- MOV_L ( EDX, REGOFF(8, ECX) )
-
-ALIGNTEXT32
+GLOBL GLNAME( gl_3dnow_transform_points3_perspective_masked )
+GLNAME( gl_3dnow_transform_points3_perspective_masked ):
PUSH_L ( ESI )
PUSH_L ( EDI )
PUSH_L ( EBX )
PUSH_L ( EBP )
- MOV_L ( REGOFF(4, ECX), EDX )
+ MOV_L ( ARG_DEST, ECX )
+ MOV_L ( ARG_MATRIX, ESI )
+ MOV_L ( ARG_SOURCE, EAX )
+ MOV_L ( CONST(4), REGOFF(V4F_SIZE, ECX) )
+ OR_B ( CONST(VEC_SIZE_4), REGOFF(V4F_FLAGS, ECX) )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), EDX )
+ MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) )
+
+ MOV_L ( REGOFF(V4F_START, ECX), EDX )
MOV_L ( ESI, ECX )
- MOV_L ( REGOFF(8, EAX), ESI ) /* count */
- MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */
- MOV_L ( REGOFF(4, EAX), EAX )
- MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */
- MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */
+ MOV_L ( REGOFF(V4F_COUNT, EAX), ESI )
+ MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI )
+ MOV_L ( REGOFF(V4F_START, EAX), EAX )
+ MOV_L ( ARG_CLIP, EBP )
+ MOV_B ( ARG_FLAG, BL )
- FEMMS
+ MOVD ( REGIND(ECX), MM0 ) /* | m00 */
+ MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
+ PSLLQ ( CONST(32), MM7 ) /* m11 | */
+ POR ( MM7, MM0 ) /* m11 | m00 */
-ALIGNTEXT32
-LLBL(G3TPIM_2):
+ MOVQ ( REGOFF(32, ECX), MM1 ) /* m21 | m20 */
+ MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TPIM_3) /* skip vertex */ )
+ MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */
- MOVQ ( REGIND(EAX), MM0 ) /* x1 | x0 */
- MOVD ( REGOFF(8, EAX), MM1 ) /* | x2 */
+ TEST_L ( ESI, ESI )
+ JZ ( LLBL( G3TPPM_4 ) )
- MOVQ ( MM0, REGIND(EDX) ) /* r1 | r0 */
- MOVD ( MM1, REGOFF(8, EDX) ) /* | r2 */
+ALIGNTEXT16
+LLBL( G3TPPM_2 ):
-LLBL(G3TPIM_3):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TPPM_3 ) ) /* skip vertex */
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
+ MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
- JA ( LLBL(G3TPIM_2) /* cnt > 0 ? -> process next vertex */ )
+ PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */
+ MOVQ ( MM5, MM6 ) /* | x2 */
-LLBL(G3TPIM_4):
- FEMMS
+ PUNPCKLDQ ( MM5, MM5 ) /* x2 | x2 */
+ PFMUL ( MM1, MM5 ) /* x2*m21 | x2*m20 */
- POP_L ( EBP )
- POP_L ( EBX )
- POP_L ( EDI )
- POP_L ( ESI )
+ PFADD ( MM4, MM5 ) /* x1*m11+x2*m21 | x0*m00+x2*m20 */
+ MOVQ ( MM5, REGIND(EDX) ) /* write r0, r1 */
- POP_L ( ESI )
- RET
+ MOVQ ( MM6, MM5 ) /* | x2 */
+ PFMUL ( MM2, MM5 ) /* | x2*m22 */
+ PFADD ( MM3, MM5 ) /* | x2*m22+m32 */
+ PFSUBR ( MM7, MM6 ) /* (LO mm7 == 0) | -x2 */
+ MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */
+ MOVD ( MM6, REGOFF(12, EDX) ) /* write r3 */
ALIGNTEXT16
-GLOBL GLNAME(gl_3dnow_transform_points3_2d_masked)
-GLNAME( gl_3dnow_transform_points3_2d_masked ):
+LLBL( G3TPPM_3 ):
- PUSH_L ( ESI )
- MOV_L ( REGOFF(8, ESP), ECX )
- MOV_L ( REGOFF(12, ESP), ESI )
- MOV_L ( REGOFF(16, ESP), EAX )
- MOV_L ( CONST(3), REGOFF(16, ECX) )
- OR_B ( CONST(7), REGOFF(20, ECX) )
- MOV_L ( REGOFF(8, EAX), EDX )
- MOV_L ( EDX, REGOFF(8, ECX) )
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
-ALIGNTEXT32
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
- PUSH_L ( ESI )
- PUSH_L ( EDI )
- PUSH_L ( EBX )
- PUSH_L ( EBP )
+ JNZ ( LLBL( G3TPPM_2 ) ) /* cnt > 0 ? -> process next vertex */
- MOV_L ( REGOFF(4, ECX), EDX )
- MOV_L ( ESI, ECX )
- MOV_L ( REGOFF(8, EAX), ESI ) /* count */
- MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */
- MOV_L ( REGOFF(4, EAX), EAX )
- MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */
- MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */
+LLBL( G3TPPM_4 ):
FEMMS
+ POP_L ( EBP )
+ POP_L ( EBX )
+ POP_L ( EDI )
+ POP_L ( ESI )
+ RET
- MOVD ( REGIND(ECX), MM0 ) /* | m00 */
- MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */
- PSLLQ ( CONST(32), MM7 ) /* m10 | */
- POR ( MM7, MM0 ) /* m10 | m00 */
- MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */
- MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
- PSLLQ ( CONST(32), MM7 ) /* m11 | */
- POR ( MM7, MM1 ) /* m11 | m01 */
+ALIGNTEXT16
+GLOBL GLNAME( gl_3dnow_transform_points3_3d_masked )
+GLNAME( gl_3dnow_transform_points3_3d_masked ):
- MOVQ ( REGOFF(48, ECX), MM2 ) /* m31 | m30 */
- CMP_L ( CONST(0), ESI )
- JE ( LLBL(G3TP2M_4) )
+ PUSH_L ( ESI )
+ PUSH_L ( EDI )
+ PUSH_L ( EBX )
+ PUSH_L ( EBP )
+ MOV_L ( ARG_DEST, ECX )
+ MOV_L ( ARG_MATRIX, ESI )
+ MOV_L ( ARG_SOURCE, EAX )
+ MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) )
+ OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), EDX )
+ MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) )
-ALIGNTEXT32
-LLBL(G3TP2M_2):
+ MOV_L ( REGOFF(V4F_START, ECX), EDX )
+ MOV_L ( ESI, ECX )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), ESI )
+ MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI )
+ MOV_L ( REGOFF(V4F_START, EAX), EAX )
+ MOV_L ( ARG_CLIP, EBP )
+ MOV_B ( ARG_FLAG, BL )
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TP2M_3) /* skip vertex */ )
+ MOVD ( REGIND(ECX), MM0 ) /* | m00 */
+ MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */
- MOVQ ( REGIND(EAX), MM3 ) /* x1 | x0 */
- MOVQ ( MM3, MM4 ) /* x1 | x0 */
+ PSLLQ ( CONST(32), MM7 ) /* m10 | */
+ POR ( MM7, MM0 ) /* m10 | m00 */
- MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
- PFMUL ( MM0, MM3 ) /* x1*m10 | x0*m00 */
+ MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */
+ MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
- PFMUL ( MM1, MM4 ) /* x1*m11 | x0*m01 */
- PFACC ( MM4, MM3 ) /* x0*m00+x1*m10 | x0*m01+x1*m11 */
+ PSLLQ ( CONST(32), MM7 ) /* m11 | */
+ POR ( MM7, MM1 ) /* m11 | m01 */
- PFADD ( MM2, MM3 ) /* x0*...*m10+m30| x0*m01+x1*m11+m31 */
- MOVQ ( MM3, REGIND(EDX) ) /* write r0, r1 */
+ MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */
+ MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */
- MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */
+ TEST_L ( ESI, ESI )
+ JZ ( LLBL( G3TP3M_6 ) )
-LLBL(G3TP2M_3):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ PUSH_L ( EBP )
+ PUSH_L ( EAX )
+ PUSH_L ( EDX )
+ PUSH_L ( ESI )
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ALIGNTEXT16
+LLBL( G3TP3M_2 ):
- JA ( LLBL(G3TP2M_2) /* cnt > 0 ? -> process next vertex */ )
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TP3M_3 ) ) /* skip vertex */
-LLBL(G3TP2M_4):
- FEMMS
+ MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
+ MOVQ ( MM4, MM5 ) /* x1 | x0 */
- POP_L ( EBP )
- POP_L ( EBX )
- POP_L ( EDI )
- POP_L ( ESI )
+ PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */
+ MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
- POP_L ( ESI )
- RET
+ PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */
+ PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */
+ PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */
+ PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */
+ PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */
+ PFADD ( MM4, MM6 ) /* r1 | r0 */
+ MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */
ALIGNTEXT16
-GLOBL GLNAME(gl_3dnow_transform_points3_2d_no_rot_masked)
-GLNAME( gl_3dnow_transform_points3_2d_no_rot_masked ):
+LLBL( G3TP3M_3 ):
- PUSH_L ( ESI )
- MOV_L ( REGOFF(8, ESP), ECX )
- MOV_L ( REGOFF(12, ESP), ESI )
- MOV_L ( REGOFF(16, ESP), EAX )
- MOV_L ( CONST(3), REGOFF(16, ECX) )
- OR_B ( CONST(7), REGOFF(20, ECX) )
- MOV_L ( REGOFF(8, EAX), EDX )
- MOV_L ( EDX, REGOFF(8, ECX) )
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
-ALIGNTEXT32
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
- PUSH_L ( ESI )
- PUSH_L ( EDI )
- PUSH_L ( EBX )
- PUSH_L ( EBP )
+ JNZ ( LLBL( G3TP3M_2 ) ) /* cnt > 0 ? -> process next vertex */
- MOV_L ( REGOFF(4, ECX), EDX )
- MOV_L ( ESI, ECX )
- MOV_L ( REGOFF(8, EAX), ESI ) /* count */
- MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */
- MOV_L ( REGOFF(4, EAX), EAX )
- MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */
- MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */
+ /* and now the second stripe ... */
+ MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */
- FEMMS
+ MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */
+ PSLLQ ( CONST(32), MM7 ) /* m12 | */
- MOVD ( REGIND(ECX), MM0 ) /* | m00 */
- MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
+ POR ( MM7, MM0 ) /* m12 | m02 */
+ MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */
- PSLLQ ( CONST(32), MM7 ) /* m11 | */
- POR ( MM7, MM0 ) /* m11 | m00 */
+ MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */
- MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */
- CMP_L ( CONST(0), ESI )
- JE ( LLBL(G3TP2NRM_4) )
+ POP_L ( ESI ) /* reset counter & pointers */
+ POP_L ( EDX )
+ POP_L ( EAX )
+ POP_L ( EBP )
+ALIGNTEXT16
+LLBL( G3TP3M_4 ):
-ALIGNTEXT32
-LLBL(G3TP2NRM_2):
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TP3M_5 ) ) /* skip vertex */
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TP2NRM_3) /* skip vertex */ )
+ MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
+ MOVQ ( MM4, MM5 ) /* x1 | x0 */
- MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
- MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
+ PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */
+ MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
- PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */
- MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */
+ PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */
+ PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */
- PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */
- MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */
+ PFMUL ( MM2, MM6 ) /* | x2*m22 */
+ PFADD ( MM3, MM6 ) /* | x2*m22+m32 */
-LLBL(G3TP2NRM_3):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ PFADD ( MM4, MM6 ) /* | r2 */
+ MOVD ( MM6, REGOFF(8, EDX) ) /* write r2 */
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ALIGNTEXT16
+LLBL( G3TP3M_5 ):
- JA ( LLBL(G3TP2NRM_2) /* cnt > 0 ? -> process next vertex */ )
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
-LLBL(G3TP2NRM_4):
- FEMMS
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
+
+ JNZ ( LLBL( G3TP3M_4 ) ) /* cnt > 0 ? -> process next vertex */
+LLBL( G3TP3M_6 ):
+
+ FEMMS
POP_L ( EBP )
POP_L ( EBX )
POP_L ( EDI )
POP_L ( ESI )
-
- POP_L ( ESI )
RET
-
ALIGNTEXT16
-GLOBL GLNAME(gl_3dnow_transform_points3_3d_masked)
-GLNAME( gl_3dnow_transform_points3_3d_masked ):
-
- PUSH_L ( ESI )
- MOV_L ( REGOFF(8, ESP), ECX )
- MOV_L ( REGOFF(12, ESP), ESI )
- MOV_L ( REGOFF(16, ESP), EAX )
- MOV_L ( CONST(3), REGOFF(16, ECX) )
- OR_B ( CONST(7), REGOFF(20, ECX) )
- MOV_L ( REGOFF(8, EAX), EDX )
- MOV_L ( EDX, REGOFF(8, ECX) )
-
-ALIGNTEXT32
+GLOBL GLNAME( gl_3dnow_transform_points3_3d_no_rot_masked )
+GLNAME( gl_3dnow_transform_points3_3d_no_rot_masked ):
PUSH_L ( ESI )
PUSH_L ( EDI )
PUSH_L ( EBX )
PUSH_L ( EBP )
- MOV_L ( REGOFF(4, ECX), EDX )
+ MOV_L ( ARG_DEST, ECX )
+ MOV_L ( ARG_MATRIX, ESI )
+ MOV_L ( ARG_SOURCE, EAX )
+ MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) )
+ OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), EDX )
+ MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) )
+
+ MOV_L ( REGOFF(V4F_START, ECX), EDX )
MOV_L ( ESI, ECX )
- MOV_L ( REGOFF(8, EAX), ESI ) /* count */
- MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */
- MOV_L ( REGOFF(4, EAX), EAX )
- MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */
- MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */
+ MOV_L ( REGOFF(V4F_COUNT, EAX), ESI )
+ MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI )
+ MOV_L ( REGOFF(V4F_START, EAX), EAX )
+ MOV_L ( ARG_CLIP, EBP )
+ MOV_B ( ARG_FLAG, BL )
- FEMMS
+ MOVD ( REGIND(ECX), MM0 ) /* | m00 */
+ MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
- MOVD ( REGIND(ECX), MM0 ) /* | m00 */
- MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */
+ PSLLQ ( CONST(32), MM7 ) /* m11 | */
+ POR ( MM7, MM0 ) /* m11 | m00 */
- PSLLQ ( CONST(32), MM7 ) /* m10 | */
- POR ( MM7, MM0 ) /* m10 | m00 */
+ MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */
+ PUNPCKLDQ ( MM2, MM2 ) /* m22 | m22 */
- MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */
- MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
+ MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */
+ MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */
- PSLLQ ( CONST(32), MM7 ) /* m11 | */
- POR ( MM7, MM1 ) /* m11 | m01 */
+ PUNPCKLDQ ( MM3, MM3 ) /* m32 | m32 */
- MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */
- MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */
- CMP_L ( CONST(0), ESI )
- JE ( LLBL(G3TP3M_6) )
+ TEST_L ( ESI, ESI )
+ JZ ( LLBL( G3TP3NRM_4 ) )
- PUSH_L ( EBP )
- PUSH_L ( EAX )
- PUSH_L ( EDX )
- PUSH_L ( ESI )
+ALIGNTEXT16
+LLBL( G3TP3NRM_2 ):
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TP3NRM_3 ) ) /* skip vertex */
-ALIGNTEXT32
-LLBL(G3TP3M_2):
+ MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
+ MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TP3M_3) /* skip vertex */ )
+ PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */
+ PFMUL ( MM2, MM5 ) /* | x2*m22 */
- MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
- MOVQ ( MM4, MM5 ) /* x1 | x0 */
+ PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */
+ PFADD ( MM3, MM5 ) /* | x2*m22+m32 */
- PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */
- MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
+ MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */
+ MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */
- PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */
- PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */
+LLBL( G3TP3NRM_3 ):
- PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */
- PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
- PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */
- PFADD ( MM4, MM6 ) /* r1 | r0 */
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
- MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */
+ JNZ ( LLBL( G3TP3NRM_2 ) ) /* cnt > 0 ? -> process next vertex */
-LLBL(G3TP3M_3):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+LLBL( G3TP3NRM_4 ):
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ FEMMS
+ POP_L ( EBP )
+ POP_L ( EBX )
+ POP_L ( EDI )
+ POP_L ( ESI )
+ RET
- JA ( LLBL(G3TP3M_2) /* cnt > 0 ? -> process next vertex */ )
- /* and now the second stripe ... */
- MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */
- MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */
- PSLLQ ( CONST(32), MM7 ) /* m12 | */
- POR ( MM7, MM0 ) /* m12 | m02 */
- MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */
- MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */
- POP_L ( ESI ) /* reset counter & pointers */
+ALIGNTEXT16
+GLOBL GLNAME( gl_3dnow_transform_points3_2d_masked )
+GLNAME( gl_3dnow_transform_points3_2d_masked ):
- POP_L ( EDX )
- POP_L ( EAX )
+ PUSH_L ( ESI )
+ PUSH_L ( EDI )
+ PUSH_L ( EBX )
+ PUSH_L ( EBP )
- POP_L ( EBP )
+ MOV_L ( ARG_DEST, ECX )
+ MOV_L ( ARG_MATRIX, ESI )
+ MOV_L ( ARG_SOURCE, EAX )
+ MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) )
+ OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), EDX )
+ MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) )
+ MOV_L ( REGOFF(V4F_START, ECX), EDX )
+ MOV_L ( ESI, ECX )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), ESI )
+ MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI )
+ MOV_L ( REGOFF(V4F_START, EAX), EAX )
+ MOV_L ( ARG_CLIP, EBP )
+ MOV_B ( ARG_FLAG, BL )
-ALIGNTEXT32
-LLBL(G3TP3M_4):
+ MOVD ( REGIND(ECX), MM0 ) /* | m00 */
+ MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TP3M_5) /* skip vertex */ )
+ PSLLQ ( CONST(32), MM7 ) /* m10 | */
+ POR ( MM7, MM0 ) /* m10 | m00 */
- MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
- MOVQ ( MM4, MM5 ) /* x1 | x0 */
+ MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */
+ MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
- PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */
- MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */
+ PSLLQ ( CONST(32), MM7 ) /* m11 | */
+ POR ( MM7, MM1 ) /* m11 | m01 */
- PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */
- PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */
+ MOVQ ( REGOFF(48, ECX), MM2 ) /* m31 | m30 */
- PFMUL ( MM2, MM6 ) /* | x2*m22 */
- PFADD ( MM3, MM6 ) /* | x2*m22+m32 */
+ TEST_L ( ESI, ESI )
+ JZ ( LLBL( G3TP2M_4 ) )
- PFADD ( MM4, MM6 ) /* | r2 */
- MOVD ( MM6, REGOFF(8, EDX) ) /* write r2 */
+ALIGNTEXT16
+LLBL( G3TP2M_2 ):
-LLBL(G3TP3M_5):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TP2M_3 ) ) /* skip vertex */
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ MOVQ ( REGIND(EAX), MM3 ) /* x1 | x0 */
+ MOVQ ( MM3, MM4 ) /* x1 | x0 */
- JA ( LLBL(G3TP3M_4) /* cnt > 0 ? -> process next vertex */ )
+ MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
+ PFMUL ( MM0, MM3 ) /* x1*m10 | x0*m00 */
-LLBL(G3TP3M_6):
- FEMMS
+ PFMUL ( MM1, MM4 ) /* x1*m11 | x0*m01 */
+ PFACC ( MM4, MM3 ) /* x0*m00+x1*m10 | x0*m01+x1*m11 */
+
+ PFADD ( MM2, MM3 ) /* x0*...*m10+m30 | x0*...*m11+m31 */
+ MOVQ ( MM3, REGIND(EDX) ) /* write r0, r1 */
+
+ MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */
+ALIGNTEXT16
+LLBL( G3TP2M_3 ):
+
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
+
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
+
+ JNZ ( LLBL( G3TP2M_2 ) ) /* cnt > 0 ? -> process next vertex */
+
+LLBL( G3TP2M_4 ):
+
+ FEMMS
POP_L ( EBP )
POP_L ( EBX )
POP_L ( EDI )
POP_L ( ESI )
-
- POP_L ( ESI )
RET
-
-
ALIGNTEXT16
-GLOBL GLNAME(gl_3dnow_transform_points3_3d_no_rot_masked)
-GLNAME( gl_3dnow_transform_points3_3d_no_rot_masked ):
-
- PUSH_L ( ESI )
- MOV_L ( REGOFF(8, ESP), ECX )
- MOV_L ( REGOFF(12, ESP), ESI )
- MOV_L ( REGOFF(16, ESP), EAX )
- MOV_L ( CONST(3), REGOFF(16, ECX) )
- OR_B ( CONST(7), REGOFF(20, ECX) )
- MOV_L ( REGOFF(8, EAX), EDX )
- MOV_L ( EDX, REGOFF(8, ECX) )
-
-ALIGNTEXT32
+GLOBL GLNAME( gl_3dnow_transform_points3_2d_no_rot_masked )
+GLNAME( gl_3dnow_transform_points3_2d_no_rot_masked ):
PUSH_L ( ESI )
PUSH_L ( EDI )
PUSH_L ( EBX )
PUSH_L ( EBP )
- MOV_L ( REGOFF(4, ECX), EDX )
- MOV_L ( ESI, ECX )
- MOV_L ( REGOFF(8, EAX), ESI ) /* count */
- MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */
- MOV_L ( REGOFF(4, EAX), EAX )
- MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */
- MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */
-
- FEMMS
+ MOV_L ( ARG_DEST, ECX )
+ MOV_L ( ARG_MATRIX, ESI )
+ MOV_L ( ARG_SOURCE, EAX )
+ MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) )
+ OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), EDX )
+ MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) )
- MOVD ( REGIND(ECX), MM0 ) /* | m00 */
- MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
+ MOV_L ( REGOFF(V4F_START, ECX), EDX )
+ MOV_L ( ESI, ECX )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), ESI )
+ MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI )
+ MOV_L ( REGOFF(V4F_START, EAX), EAX )
+ MOV_L ( ARG_CLIP, EBP )
+ MOV_B ( ARG_FLAG, BL )
- PSLLQ ( CONST(32), MM7 ) /* m11 | */
- POR ( MM7, MM0 ) /* m11 | m00 */
+ MOVD ( REGIND(ECX), MM0 ) /* | m00 */
+ MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
- MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */
- PUNPCKLDQ ( MM2, MM2 ) /* m22 | m22 */
+ PSLLQ ( CONST(32), MM7 ) /* m11 | */
+ POR ( MM7, MM0 ) /* m11 | m00 */
- MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */
- MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */
+ MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */
- PUNPCKLDQ ( MM3, MM3 ) /* m32 | m32 */
- CMP_L ( CONST(0), ESI )
- JE ( LLBL(G3TP3NRM_4) )
+ TEST_L ( ESI, ESI )
+ JZ ( LLBL( G3TP2NRM_4 ) )
+ALIGNTEXT16
+LLBL( G3TP2NRM_2 ):
-ALIGNTEXT32
-LLBL(G3TP3NRM_2):
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TP2NRM_3 ) ) /* skip vertex */
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TP3NRM_3) /* skip vertex */ )
+ MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
+ MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
- MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
- MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
+ PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */
+ MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */
- PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */
- PFMUL ( MM2, MM5 ) /* | x2*m22 */
+ PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */
+ MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */
- PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */
- PFADD ( MM3, MM5 ) /* | x2*m22+m32 */
+ALIGNTEXT16
+LLBL( G3TP2NRM_3 ):
- MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */
- MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
-LLBL(G3TP3NRM_3):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ JNZ ( LLBL( G3TP2NRM_2 ) ) /* cnt > 0 ? -> process next vertex */
- JA ( LLBL(G3TP3NRM_2) /* cnt > 0 ? -> process next vertex */ )
+LLBL( G3TP2NRM_4 ):
-LLBL(G3TP3NRM_4):
FEMMS
-
POP_L ( EBP )
POP_L ( EBX )
POP_L ( EDI )
POP_L ( ESI )
-
- POP_L ( ESI )
RET
ALIGNTEXT16
-GLOBL GLNAME(gl_3dnow_transform_points3_perspective_masked)
-GLNAME( gl_3dnow_transform_points3_perspective_masked ):
-
- PUSH_L ( ESI )
- MOV_L ( REGOFF(8, ESP), ECX )
- MOV_L ( REGOFF(12, ESP), ESI )
- MOV_L ( REGOFF(16, ESP), EAX )
- MOV_L ( CONST(4), REGOFF(16, ECX) )
- OR_B ( CONST(15), REGOFF(20, ECX) )
- MOV_L ( REGOFF(8, EAX), EDX )
- MOV_L ( EDX, REGOFF(8, ECX) )
-
-ALIGNTEXT32
+GLOBL GLNAME( gl_3dnow_transform_points3_identity_masked )
+GLNAME( gl_3dnow_transform_points3_identity_masked ):
PUSH_L ( ESI )
PUSH_L ( EDI )
PUSH_L ( EBX )
PUSH_L ( EBP )
- MOV_L ( REGOFF(4, ECX), EDX )
- MOV_L ( ESI, ECX )
- MOV_L ( REGOFF(8, EAX), ESI ) /* count */
- MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */
- MOV_L ( REGOFF(4, EAX), EAX )
- MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */
- MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */
-
- FEMMS
-
- MOVD ( REGIND(ECX), MM0 ) /* | m00 */
- MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */
-
- PSLLQ ( CONST(32), MM7 ) /* m11 | */
- POR ( MM7, MM0 ) /* m11 | m00 */
-
- MOVQ ( REGOFF(32, ECX), MM1 ) /* m21 | m20 */
- MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */
-
- MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */
- CMP_L ( CONST(0), ESI )
- JE ( LLBL(G3TPPM_4) )
+ MOV_L ( ARG_DEST, ECX )
+ MOV_L ( ARG_MATRIX, ESI )
+ MOV_L ( ARG_SOURCE, EAX )
+ MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) )
+ OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), EDX )
+ MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) )
+ MOV_L ( REGOFF(V4F_START, ECX), EDX )
+ MOV_L ( ESI, ECX )
+ MOV_L ( REGOFF(V4F_COUNT, EAX), ESI )
+ MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI )
+ MOV_L ( REGOFF(V4F_START, EAX), EAX )
+ MOV_L ( ARG_CLIP, EBP )
+ MOV_B ( ARG_FLAG, BL )
-ALIGNTEXT32
-LLBL(G3TPPM_2):
-
- TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
- JNZ ( LLBL(G3TPPM_3) /* skip vertex */ )
-
- MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */
- MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */
-
- PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */
- MOVQ ( MM5, MM6 ) /* | x2 */
+ALIGNTEXT16
+LLBL( G3TPIM_2 ):
- PUNPCKLDQ ( MM5, MM5 ) /* x2 | x2 */
- PFMUL ( MM1, MM5 ) /* x2*m21 | x2*m20 */
+ TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */
+ JNZ ( LLBL( G3TPIM_3 ) ) /* skip vertex */
- PFADD ( MM4, MM5 ) /* x1*m11+x2*m21 | x0*m00+x2*m20 */
- MOVQ ( MM5, REGIND(EDX) ) /* write r0, r1 */
+ MOVQ ( REGIND(EAX), MM0 ) /* x1 | x0 */
+ MOVD ( REGOFF(8, EAX), MM1 ) /* | x2 */
- MOVQ ( MM6, MM5 ) /* | x2 */
- PFMUL ( MM2, MM5 ) /* | x2*m22 */
+ MOVQ ( MM0, REGIND(EDX) ) /* r1 | r0 */
+ MOVD ( MM1, REGOFF(8, EDX) ) /* | r2 */
- PFADD ( MM3, MM5 ) /* | x2*m22+m32 */
- PFSUBR ( MM7, MM6 ) /* (LO mm7 == 0) | -x2 */
+LLBL( G3TPIM_3 ):
- MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */
- MOVD ( MM6, REGOFF(12, EDX) ) /* write r3 */
+ ADD_L ( EDI, EAX ) /* next vertex */
+ ADD_L ( CONST(16), EDX ) /* next r */
-LLBL(G3TPPM_3):
- ADD_L ( EDI, EAX ) /* next vertex */
- ADD_L ( CONST(16), EDX ) /* next r */
+ INC_L ( EBP ) /* next clipmask */
+ DEC_L ( ESI ) /* decrement vertex counter */
- INC_L ( EBP ) /* next clipmask */
- DEC_L ( ESI ) /* decrement vertex counter */
+ JNZ ( LLBL( G3TPIM_2 ) ) /* cnt > 0 ? -> process next vertex */
- JA ( LLBL(G3TPPM_2) /* cnt > 0 ? -> process next vertex */ )
+LLBL( G3TPIM_4 ):
-LLBL(G3TPPM_4):
FEMMS
-
POP_L ( EBP )
POP_L ( EBX )
POP_L ( EDI )
POP_L ( ESI )
-
- POP_L ( ESI )
RET
-
-
-
-
-
-
-