diff options
Diffstat (limited to 'xc/extras/Mesa/src/X86/3dnow_xform_masked3.S')
-rw-r--r-- | xc/extras/Mesa/src/X86/3dnow_xform_masked3.S | 916 |
1 files changed, 448 insertions, 468 deletions
diff --git a/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S b/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S index fbc66c34b..45686e03f 100644 --- a/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S +++ b/xc/extras/Mesa/src/X86/3dnow_xform_masked3.S @@ -1,729 +1,709 @@ + +/* + * Mesa 3-D graphics library + * Version: 3.4 + * + * Copyright (C) 1999-2000 Brian Paul All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a + * copy of this software and associated documentation files (the "Software"), + * to deal in the Software without restriction, including without limitation + * the rights to use, copy, modify, merge, publish, distribute, sublicense, + * and/or sell copies of the Software, and to permit persons to whom the + * Software is furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included + * in all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS + * OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL + * BRIAN PAUL BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN + * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ + #include "assyntax.h" +#include "xform_args.h" + + SEG_TEXT - SEG_TEXT +#define FRAME_OFFSET 16 ALIGNTEXT16 -GLOBL GLNAME(gl_3dnow_transform_points3_general_masked) +GLOBL GLNAME( gl_3dnow_transform_points3_general_masked ) GLNAME( gl_3dnow_transform_points3_general_masked ): PUSH_L ( ESI ) - MOV_L ( REGOFF(8, ESP), ECX ) - MOV_L ( REGOFF(12, ESP), ESI ) - MOV_L ( REGOFF(16, ESP), EAX ) - MOV_L ( CONST(4), REGOFF(16, ECX) ) - OR_B ( CONST(15), REGOFF(20, ECX) ) - MOV_L ( REGOFF(8, EAX), EDX ) - MOV_L ( EDX, REGOFF(8, ECX) ) - -ALIGNTEXT32 - - PUSH_L ( ESI ) PUSH_L ( EDI ) PUSH_L ( EBX ) PUSH_L ( EBP ) - MOV_L ( REGOFF(4, ECX), EDX ) - MOV_L ( ESI, ECX ) - MOV_L ( REGOFF(8, EAX), ESI ) /* count */ - MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */ - MOV_L ( REGOFF(4, EAX), EAX ) - MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */ - MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */ + MOV_L ( ARG_DEST, ECX ) + MOV_L ( ARG_MATRIX, ESI ) + MOV_L ( ARG_SOURCE, EAX ) + MOV_L ( CONST(4), REGOFF(V4F_SIZE, ECX) ) + OR_B ( CONST(VEC_SIZE_4), REGOFF(V4F_FLAGS, ECX) ) + MOV_L ( REGOFF(V4F_COUNT, EAX), EDX ) + MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) ) - FEMMS + MOV_L ( REGOFF(V4F_START, ECX), EDX ) + MOV_L ( ESI, ECX ) + MOV_L ( REGOFF(V4F_COUNT, EAX), ESI ) + MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI ) + MOV_L ( REGOFF(V4F_START, EAX), EAX ) + MOV_L ( ARG_CLIP, EBP ) + MOV_B ( ARG_FLAG, BL ) - MOVD ( REGIND(ECX), MM0 ) /* | m00 */ - MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */ + MOVD ( REGIND(ECX), MM0 ) /* | m00 */ + MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */ - PSLLQ ( CONST(32), MM7 ) /* m10 | */ - POR ( MM7, MM0 ) /* m10 | m00 */ + PSLLQ ( CONST(32), MM7 ) /* m10 | */ + POR ( MM7, MM0 ) /* m10 | m00 */ - MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */ - MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ + MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */ + MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ - PSLLQ ( CONST(32), MM7 ) /* m11 | */ - POR ( MM7, MM1 ) /* m11 | m01 */ + PSLLQ ( CONST(32), MM7 ) /* m11 | */ + POR ( MM7, MM1 ) /* m11 | m01 */ - MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */ - MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */ + MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */ + MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */ - CMP_L ( CONST(0), ESI ) - JE ( LLBL(G3TPGM_6) ) + TEST_L ( ESI, ESI ) + JZ ( LLBL( G3TPGM_6 ) ) PUSH_L ( EBP ) PUSH_L ( EAX ) PUSH_L ( EDX ) PUSH_L ( ESI ) +ALIGNTEXT16 +LLBL( G3TPGM_2 ): -ALIGNTEXT32 -LLBL(G3TPGM_2): + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TPGM_3 ) ) /* skip vertex */ - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TPGM_3) /* skip vertex */ ) + MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ + MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ - MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ - MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ + MOVQ ( MM4, MM5 ) /* x1 | x0 */ + PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */ - MOVQ ( MM4, MM5 ) /* x1 | x0 */ - PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */ + PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */ + PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */ - PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */ - PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */ + PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */ + PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */ - PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */ - PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */ + PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */ + PFADD ( MM4, MM6 ) /* r1 | r0 */ - PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */ - PFADD ( MM4, MM6 ) /* r1 | r0 */ + MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */ - MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */ +ALIGNTEXT16 +LLBL( G3TPGM_3 ): -LLBL(G3TPGM_3): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ - JA ( LLBL(G3TPGM_2) /* cnt > 0 ? -> process next vertex */ ) - /* and now the second stripe ... */ + JNZ ( LLBL( G3TPGM_2 ) ) /* cnt > 0 ? -> process next vertex */ - POP_L ( ESI ) /* reset counter & pointers */ + /* and now the second stripe ... */ + POP_L ( ESI ) /* reset counter & pointers */ POP_L ( EDX ) POP_L ( EAX ) POP_L ( EBP ) - MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */ - MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */ + MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */ + MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */ - PSLLQ ( CONST(32), MM7 ) /* m12 | */ - POR ( MM7, MM0 ) /* m12 | m02 */ + PSLLQ ( CONST(32), MM7 ) /* m12 | */ + POR ( MM7, MM0 ) /* m12 | m02 */ - MOVD ( REGOFF(12, ECX), MM1 ) /* | m03 */ - MOVD ( REGOFF(28, ECX), MM7 ) /* | m13 */ + MOVD ( REGOFF(12, ECX), MM1 ) /* | m03 */ + MOVD ( REGOFF(28, ECX), MM7 ) /* | m13 */ - PSLLQ ( CONST(32), MM7 ) /* m13 | */ - POR ( MM7, MM1 ) /* m13 | m03 */ + PSLLQ ( CONST(32), MM7 ) /* m13 | */ + POR ( MM7, MM1 ) /* m13 | m03 */ - MOVQ ( REGOFF(40, ECX), MM2 ) /* m23 | m22 */ - MOVQ ( REGOFF(56, ECX), MM3 ) /* m33 | m32 */ + MOVQ ( REGOFF(40, ECX), MM2 ) /* m23 | m22 */ + MOVQ ( REGOFF(56, ECX), MM3 ) /* m33 | m32 */ +ALIGNTEXT16 +LLBL( G3TPGM_4 ): -ALIGNTEXT32 -LLBL(G3TPGM_4): - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TPGM_5) /* skip vertex */ ) + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TPGM_5 ) ) /* skip vertex */ - MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ - MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ + MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ + MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ - MOVQ ( MM4, MM5 ) /* x1 | x0 */ - PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */ + MOVQ ( MM4, MM5 ) /* x1 | x0 */ + PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */ - PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */ - PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */ + PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */ + PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */ - PFMUL ( MM2, MM6 ) /* x2*m23 | x2*m22 */ - PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */ + PFMUL ( MM2, MM6 ) /* x2*m23 | x2*m22 */ + PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */ - PFADD ( MM3, MM6 ) /* x2*m23+m33 | x2*m22+m32 */ - PFADD ( MM4, MM6 ) /* r3 | r2 */ + PFADD ( MM3, MM6 ) /* x2*m23+m33 | x2*m22+m32 */ + PFADD ( MM4, MM6 ) /* r3 | r2 */ - MOVQ ( MM6, REGOFF(8, EDX) ) /* write r2, r3 */ + MOVQ ( MM6, REGOFF(8, EDX) ) /* write r2, r3 */ -LLBL(G3TPGM_5): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ +ALIGNTEXT16 +LLBL( G3TPGM_5 ): - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ - JA ( LLBL(G3TPGM_4) /* cnt > 0 ? -> process next vertex */ ) + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ -LLBL(G3TPGM_6): + JNZ ( LLBL( G3TPGM_4 ) ) /* cnt > 0 ? -> process next vertex */ - FEMMS +LLBL( G3TPGM_6 ): + FEMMS POP_L ( EBP ) POP_L ( EBX ) POP_L ( EDI ) POP_L ( ESI ) - - POP_L ( ESI ) RET - - ALIGNTEXT16 -GLOBL GLNAME(gl_3dnow_transform_points3_identity_masked) -GLNAME( gl_3dnow_transform_points3_identity_masked ): - - PUSH_L ( ESI ) - MOV_L ( REGOFF(8, ESP), ECX ) - MOV_L ( REGOFF(12, ESP), ESI ) - MOV_L ( REGOFF(16, ESP), EAX ) - MOV_L ( CONST(3), REGOFF(16, ECX) ) - OR_B ( CONST(7), REGOFF(20, ECX) ) - MOV_L ( REGOFF(8, EAX), EDX ) - MOV_L ( EDX, REGOFF(8, ECX) ) - -ALIGNTEXT32 +GLOBL GLNAME( gl_3dnow_transform_points3_perspective_masked ) +GLNAME( gl_3dnow_transform_points3_perspective_masked ): PUSH_L ( ESI ) PUSH_L ( EDI ) PUSH_L ( EBX ) PUSH_L ( EBP ) - MOV_L ( REGOFF(4, ECX), EDX ) + MOV_L ( ARG_DEST, ECX ) + MOV_L ( ARG_MATRIX, ESI ) + MOV_L ( ARG_SOURCE, EAX ) + MOV_L ( CONST(4), REGOFF(V4F_SIZE, ECX) ) + OR_B ( CONST(VEC_SIZE_4), REGOFF(V4F_FLAGS, ECX) ) + MOV_L ( REGOFF(V4F_COUNT, EAX), EDX ) + MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) ) + + MOV_L ( REGOFF(V4F_START, ECX), EDX ) MOV_L ( ESI, ECX ) - MOV_L ( REGOFF(8, EAX), ESI ) /* count */ - MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */ - MOV_L ( REGOFF(4, EAX), EAX ) - MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */ - MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */ + MOV_L ( REGOFF(V4F_COUNT, EAX), ESI ) + MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI ) + MOV_L ( REGOFF(V4F_START, EAX), EAX ) + MOV_L ( ARG_CLIP, EBP ) + MOV_B ( ARG_FLAG, BL ) - FEMMS + MOVD ( REGIND(ECX), MM0 ) /* | m00 */ + MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ + PSLLQ ( CONST(32), MM7 ) /* m11 | */ + POR ( MM7, MM0 ) /* m11 | m00 */ -ALIGNTEXT32 -LLBL(G3TPIM_2): + MOVQ ( REGOFF(32, ECX), MM1 ) /* m21 | m20 */ + MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */ - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TPIM_3) /* skip vertex */ ) + MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */ - MOVQ ( REGIND(EAX), MM0 ) /* x1 | x0 */ - MOVD ( REGOFF(8, EAX), MM1 ) /* | x2 */ + TEST_L ( ESI, ESI ) + JZ ( LLBL( G3TPPM_4 ) ) - MOVQ ( MM0, REGIND(EDX) ) /* r1 | r0 */ - MOVD ( MM1, REGOFF(8, EDX) ) /* | r2 */ +ALIGNTEXT16 +LLBL( G3TPPM_2 ): -LLBL(G3TPIM_3): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TPPM_3 ) ) /* skip vertex */ - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ + MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ + MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ - JA ( LLBL(G3TPIM_2) /* cnt > 0 ? -> process next vertex */ ) + PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */ + MOVQ ( MM5, MM6 ) /* | x2 */ -LLBL(G3TPIM_4): - FEMMS + PUNPCKLDQ ( MM5, MM5 ) /* x2 | x2 */ + PFMUL ( MM1, MM5 ) /* x2*m21 | x2*m20 */ - POP_L ( EBP ) - POP_L ( EBX ) - POP_L ( EDI ) - POP_L ( ESI ) + PFADD ( MM4, MM5 ) /* x1*m11+x2*m21 | x0*m00+x2*m20 */ + MOVQ ( MM5, REGIND(EDX) ) /* write r0, r1 */ - POP_L ( ESI ) - RET + MOVQ ( MM6, MM5 ) /* | x2 */ + PFMUL ( MM2, MM5 ) /* | x2*m22 */ + PFADD ( MM3, MM5 ) /* | x2*m22+m32 */ + PFSUBR ( MM7, MM6 ) /* (LO mm7 == 0) | -x2 */ + MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */ + MOVD ( MM6, REGOFF(12, EDX) ) /* write r3 */ ALIGNTEXT16 -GLOBL GLNAME(gl_3dnow_transform_points3_2d_masked) -GLNAME( gl_3dnow_transform_points3_2d_masked ): +LLBL( G3TPPM_3 ): - PUSH_L ( ESI ) - MOV_L ( REGOFF(8, ESP), ECX ) - MOV_L ( REGOFF(12, ESP), ESI ) - MOV_L ( REGOFF(16, ESP), EAX ) - MOV_L ( CONST(3), REGOFF(16, ECX) ) - OR_B ( CONST(7), REGOFF(20, ECX) ) - MOV_L ( REGOFF(8, EAX), EDX ) - MOV_L ( EDX, REGOFF(8, ECX) ) + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ -ALIGNTEXT32 + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ - PUSH_L ( ESI ) - PUSH_L ( EDI ) - PUSH_L ( EBX ) - PUSH_L ( EBP ) + JNZ ( LLBL( G3TPPM_2 ) ) /* cnt > 0 ? -> process next vertex */ - MOV_L ( REGOFF(4, ECX), EDX ) - MOV_L ( ESI, ECX ) - MOV_L ( REGOFF(8, EAX), ESI ) /* count */ - MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */ - MOV_L ( REGOFF(4, EAX), EAX ) - MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */ - MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */ +LLBL( G3TPPM_4 ): FEMMS + POP_L ( EBP ) + POP_L ( EBX ) + POP_L ( EDI ) + POP_L ( ESI ) + RET - MOVD ( REGIND(ECX), MM0 ) /* | m00 */ - MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */ - PSLLQ ( CONST(32), MM7 ) /* m10 | */ - POR ( MM7, MM0 ) /* m10 | m00 */ - MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */ - MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ - PSLLQ ( CONST(32), MM7 ) /* m11 | */ - POR ( MM7, MM1 ) /* m11 | m01 */ +ALIGNTEXT16 +GLOBL GLNAME( gl_3dnow_transform_points3_3d_masked ) +GLNAME( gl_3dnow_transform_points3_3d_masked ): - MOVQ ( REGOFF(48, ECX), MM2 ) /* m31 | m30 */ - CMP_L ( CONST(0), ESI ) - JE ( LLBL(G3TP2M_4) ) + PUSH_L ( ESI ) + PUSH_L ( EDI ) + PUSH_L ( EBX ) + PUSH_L ( EBP ) + MOV_L ( ARG_DEST, ECX ) + MOV_L ( ARG_MATRIX, ESI ) + MOV_L ( ARG_SOURCE, EAX ) + MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) ) + OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) ) + MOV_L ( REGOFF(V4F_COUNT, EAX), EDX ) + MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) ) -ALIGNTEXT32 -LLBL(G3TP2M_2): + MOV_L ( REGOFF(V4F_START, ECX), EDX ) + MOV_L ( ESI, ECX ) + MOV_L ( REGOFF(V4F_COUNT, EAX), ESI ) + MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI ) + MOV_L ( REGOFF(V4F_START, EAX), EAX ) + MOV_L ( ARG_CLIP, EBP ) + MOV_B ( ARG_FLAG, BL ) - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TP2M_3) /* skip vertex */ ) + MOVD ( REGIND(ECX), MM0 ) /* | m00 */ + MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */ - MOVQ ( REGIND(EAX), MM3 ) /* x1 | x0 */ - MOVQ ( MM3, MM4 ) /* x1 | x0 */ + PSLLQ ( CONST(32), MM7 ) /* m10 | */ + POR ( MM7, MM0 ) /* m10 | m00 */ - MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ - PFMUL ( MM0, MM3 ) /* x1*m10 | x0*m00 */ + MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */ + MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ - PFMUL ( MM1, MM4 ) /* x1*m11 | x0*m01 */ - PFACC ( MM4, MM3 ) /* x0*m00+x1*m10 | x0*m01+x1*m11 */ + PSLLQ ( CONST(32), MM7 ) /* m11 | */ + POR ( MM7, MM1 ) /* m11 | m01 */ - PFADD ( MM2, MM3 ) /* x0*...*m10+m30| x0*m01+x1*m11+m31 */ - MOVQ ( MM3, REGIND(EDX) ) /* write r0, r1 */ + MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */ + MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */ - MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */ + TEST_L ( ESI, ESI ) + JZ ( LLBL( G3TP3M_6 ) ) -LLBL(G3TP2M_3): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ + PUSH_L ( EBP ) + PUSH_L ( EAX ) + PUSH_L ( EDX ) + PUSH_L ( ESI ) - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ +ALIGNTEXT16 +LLBL( G3TP3M_2 ): - JA ( LLBL(G3TP2M_2) /* cnt > 0 ? -> process next vertex */ ) + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TP3M_3 ) ) /* skip vertex */ -LLBL(G3TP2M_4): - FEMMS + MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ + MOVQ ( MM4, MM5 ) /* x1 | x0 */ - POP_L ( EBP ) - POP_L ( EBX ) - POP_L ( EDI ) - POP_L ( ESI ) + PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */ + MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ - POP_L ( ESI ) - RET + PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */ + PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */ + PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */ + PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */ + PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */ + PFADD ( MM4, MM6 ) /* r1 | r0 */ + MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */ ALIGNTEXT16 -GLOBL GLNAME(gl_3dnow_transform_points3_2d_no_rot_masked) -GLNAME( gl_3dnow_transform_points3_2d_no_rot_masked ): +LLBL( G3TP3M_3 ): - PUSH_L ( ESI ) - MOV_L ( REGOFF(8, ESP), ECX ) - MOV_L ( REGOFF(12, ESP), ESI ) - MOV_L ( REGOFF(16, ESP), EAX ) - MOV_L ( CONST(3), REGOFF(16, ECX) ) - OR_B ( CONST(7), REGOFF(20, ECX) ) - MOV_L ( REGOFF(8, EAX), EDX ) - MOV_L ( EDX, REGOFF(8, ECX) ) + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ -ALIGNTEXT32 + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ - PUSH_L ( ESI ) - PUSH_L ( EDI ) - PUSH_L ( EBX ) - PUSH_L ( EBP ) + JNZ ( LLBL( G3TP3M_2 ) ) /* cnt > 0 ? -> process next vertex */ - MOV_L ( REGOFF(4, ECX), EDX ) - MOV_L ( ESI, ECX ) - MOV_L ( REGOFF(8, EAX), ESI ) /* count */ - MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */ - MOV_L ( REGOFF(4, EAX), EAX ) - MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */ - MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */ + /* and now the second stripe ... */ + MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */ - FEMMS + MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */ + PSLLQ ( CONST(32), MM7 ) /* m12 | */ - MOVD ( REGIND(ECX), MM0 ) /* | m00 */ - MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ + POR ( MM7, MM0 ) /* m12 | m02 */ + MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */ - PSLLQ ( CONST(32), MM7 ) /* m11 | */ - POR ( MM7, MM0 ) /* m11 | m00 */ + MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */ - MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */ - CMP_L ( CONST(0), ESI ) - JE ( LLBL(G3TP2NRM_4) ) + POP_L ( ESI ) /* reset counter & pointers */ + POP_L ( EDX ) + POP_L ( EAX ) + POP_L ( EBP ) +ALIGNTEXT16 +LLBL( G3TP3M_4 ): -ALIGNTEXT32 -LLBL(G3TP2NRM_2): + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TP3M_5 ) ) /* skip vertex */ - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TP2NRM_3) /* skip vertex */ ) + MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ + MOVQ ( MM4, MM5 ) /* x1 | x0 */ - MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ - MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ + PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */ + MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ - PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */ - MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */ + PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */ + PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */ - PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */ - MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */ + PFMUL ( MM2, MM6 ) /* | x2*m22 */ + PFADD ( MM3, MM6 ) /* | x2*m22+m32 */ -LLBL(G3TP2NRM_3): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ + PFADD ( MM4, MM6 ) /* | r2 */ + MOVD ( MM6, REGOFF(8, EDX) ) /* write r2 */ - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ +ALIGNTEXT16 +LLBL( G3TP3M_5 ): - JA ( LLBL(G3TP2NRM_2) /* cnt > 0 ? -> process next vertex */ ) + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ -LLBL(G3TP2NRM_4): - FEMMS + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ + + JNZ ( LLBL( G3TP3M_4 ) ) /* cnt > 0 ? -> process next vertex */ +LLBL( G3TP3M_6 ): + + FEMMS POP_L ( EBP ) POP_L ( EBX ) POP_L ( EDI ) POP_L ( ESI ) - - POP_L ( ESI ) RET - ALIGNTEXT16 -GLOBL GLNAME(gl_3dnow_transform_points3_3d_masked) -GLNAME( gl_3dnow_transform_points3_3d_masked ): - - PUSH_L ( ESI ) - MOV_L ( REGOFF(8, ESP), ECX ) - MOV_L ( REGOFF(12, ESP), ESI ) - MOV_L ( REGOFF(16, ESP), EAX ) - MOV_L ( CONST(3), REGOFF(16, ECX) ) - OR_B ( CONST(7), REGOFF(20, ECX) ) - MOV_L ( REGOFF(8, EAX), EDX ) - MOV_L ( EDX, REGOFF(8, ECX) ) - -ALIGNTEXT32 +GLOBL GLNAME( gl_3dnow_transform_points3_3d_no_rot_masked ) +GLNAME( gl_3dnow_transform_points3_3d_no_rot_masked ): PUSH_L ( ESI ) PUSH_L ( EDI ) PUSH_L ( EBX ) PUSH_L ( EBP ) - MOV_L ( REGOFF(4, ECX), EDX ) + MOV_L ( ARG_DEST, ECX ) + MOV_L ( ARG_MATRIX, ESI ) + MOV_L ( ARG_SOURCE, EAX ) + MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) ) + OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) ) + MOV_L ( REGOFF(V4F_COUNT, EAX), EDX ) + MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) ) + + MOV_L ( REGOFF(V4F_START, ECX), EDX ) MOV_L ( ESI, ECX ) - MOV_L ( REGOFF(8, EAX), ESI ) /* count */ - MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */ - MOV_L ( REGOFF(4, EAX), EAX ) - MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */ - MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */ + MOV_L ( REGOFF(V4F_COUNT, EAX), ESI ) + MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI ) + MOV_L ( REGOFF(V4F_START, EAX), EAX ) + MOV_L ( ARG_CLIP, EBP ) + MOV_B ( ARG_FLAG, BL ) - FEMMS + MOVD ( REGIND(ECX), MM0 ) /* | m00 */ + MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ - MOVD ( REGIND(ECX), MM0 ) /* | m00 */ - MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */ + PSLLQ ( CONST(32), MM7 ) /* m11 | */ + POR ( MM7, MM0 ) /* m11 | m00 */ - PSLLQ ( CONST(32), MM7 ) /* m10 | */ - POR ( MM7, MM0 ) /* m10 | m00 */ + MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */ + PUNPCKLDQ ( MM2, MM2 ) /* m22 | m22 */ - MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */ - MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ + MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */ + MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */ - PSLLQ ( CONST(32), MM7 ) /* m11 | */ - POR ( MM7, MM1 ) /* m11 | m01 */ + PUNPCKLDQ ( MM3, MM3 ) /* m32 | m32 */ - MOVQ ( REGOFF(32, ECX), MM2 ) /* m21 | m20 */ - MOVQ ( REGOFF(48, ECX), MM3 ) /* m31 | m30 */ - CMP_L ( CONST(0), ESI ) - JE ( LLBL(G3TP3M_6) ) + TEST_L ( ESI, ESI ) + JZ ( LLBL( G3TP3NRM_4 ) ) - PUSH_L ( EBP ) - PUSH_L ( EAX ) - PUSH_L ( EDX ) - PUSH_L ( ESI ) +ALIGNTEXT16 +LLBL( G3TP3NRM_2 ): + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TP3NRM_3 ) ) /* skip vertex */ -ALIGNTEXT32 -LLBL(G3TP3M_2): + MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ + MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TP3M_3) /* skip vertex */ ) + PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */ + PFMUL ( MM2, MM5 ) /* | x2*m22 */ - MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ - MOVQ ( MM4, MM5 ) /* x1 | x0 */ + PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */ + PFADD ( MM3, MM5 ) /* | x2*m22+m32 */ - PFMUL ( MM0, MM4 ) /* x1*m10 | x0*m00 */ - MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ + MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */ + MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */ - PFMUL ( MM1, MM5 ) /* x1*m11 | x0*m01 */ - PUNPCKLDQ ( MM6, MM6 ) /* x2 | x2 */ +LLBL( G3TP3NRM_3 ): - PFMUL ( MM2, MM6 ) /* x2*m21 | x2*m20 */ - PFACC ( MM5, MM4 ) /* x0*m01+x1*m11 | x0*m00+x1*m10 */ + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ - PFADD ( MM3, MM6 ) /* x2*m21+m31 | x2*m20+m30 */ - PFADD ( MM4, MM6 ) /* r1 | r0 */ + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ - MOVQ ( MM6, REGIND(EDX) ) /* write r0, r1 */ + JNZ ( LLBL( G3TP3NRM_2 ) ) /* cnt > 0 ? -> process next vertex */ -LLBL(G3TP3M_3): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ +LLBL( G3TP3NRM_4 ): - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ + FEMMS + POP_L ( EBP ) + POP_L ( EBX ) + POP_L ( EDI ) + POP_L ( ESI ) + RET - JA ( LLBL(G3TP3M_2) /* cnt > 0 ? -> process next vertex */ ) - /* and now the second stripe ... */ - MOVD ( REGOFF(8, ECX), MM0 ) /* | m02 */ - MOVD ( REGOFF(24, ECX), MM7 ) /* | m12 */ - PSLLQ ( CONST(32), MM7 ) /* m12 | */ - POR ( MM7, MM0 ) /* m12 | m02 */ - MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */ - MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */ - POP_L ( ESI ) /* reset counter & pointers */ +ALIGNTEXT16 +GLOBL GLNAME( gl_3dnow_transform_points3_2d_masked ) +GLNAME( gl_3dnow_transform_points3_2d_masked ): - POP_L ( EDX ) - POP_L ( EAX ) + PUSH_L ( ESI ) + PUSH_L ( EDI ) + PUSH_L ( EBX ) + PUSH_L ( EBP ) - POP_L ( EBP ) + MOV_L ( ARG_DEST, ECX ) + MOV_L ( ARG_MATRIX, ESI ) + MOV_L ( ARG_SOURCE, EAX ) + MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) ) + OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) ) + MOV_L ( REGOFF(V4F_COUNT, EAX), EDX ) + MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) ) + MOV_L ( REGOFF(V4F_START, ECX), EDX ) + MOV_L ( ESI, ECX ) + MOV_L ( REGOFF(V4F_COUNT, EAX), ESI ) + MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI ) + MOV_L ( REGOFF(V4F_START, EAX), EAX ) + MOV_L ( ARG_CLIP, EBP ) + MOV_B ( ARG_FLAG, BL ) -ALIGNTEXT32 -LLBL(G3TP3M_4): + MOVD ( REGIND(ECX), MM0 ) /* | m00 */ + MOVD ( REGOFF(16, ECX), MM7 ) /* | m10 */ - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TP3M_5) /* skip vertex */ ) + PSLLQ ( CONST(32), MM7 ) /* m10 | */ + POR ( MM7, MM0 ) /* m10 | m00 */ - MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ - MOVQ ( MM4, MM5 ) /* x1 | x0 */ + MOVD ( REGOFF(4, ECX), MM1 ) /* | m01 */ + MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ - PFMUL ( MM0, MM4 ) /* x1*m12 | x0*m02 */ - MOVD ( REGOFF(8, EAX), MM6 ) /* | x2 */ + PSLLQ ( CONST(32), MM7 ) /* m11 | */ + POR ( MM7, MM1 ) /* m11 | m01 */ - PFMUL ( MM1, MM5 ) /* x1*m13 | x0*m03 */ - PFACC ( MM5, MM4 ) /* x0*m03+x1*m13 | x0*m02+x1*m12 */ + MOVQ ( REGOFF(48, ECX), MM2 ) /* m31 | m30 */ - PFMUL ( MM2, MM6 ) /* | x2*m22 */ - PFADD ( MM3, MM6 ) /* | x2*m22+m32 */ + TEST_L ( ESI, ESI ) + JZ ( LLBL( G3TP2M_4 ) ) - PFADD ( MM4, MM6 ) /* | r2 */ - MOVD ( MM6, REGOFF(8, EDX) ) /* write r2 */ +ALIGNTEXT16 +LLBL( G3TP2M_2 ): -LLBL(G3TP3M_5): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TP2M_3 ) ) /* skip vertex */ - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ + MOVQ ( REGIND(EAX), MM3 ) /* x1 | x0 */ + MOVQ ( MM3, MM4 ) /* x1 | x0 */ - JA ( LLBL(G3TP3M_4) /* cnt > 0 ? -> process next vertex */ ) + MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ + PFMUL ( MM0, MM3 ) /* x1*m10 | x0*m00 */ -LLBL(G3TP3M_6): - FEMMS + PFMUL ( MM1, MM4 ) /* x1*m11 | x0*m01 */ + PFACC ( MM4, MM3 ) /* x0*m00+x1*m10 | x0*m01+x1*m11 */ + + PFADD ( MM2, MM3 ) /* x0*...*m10+m30 | x0*...*m11+m31 */ + MOVQ ( MM3, REGIND(EDX) ) /* write r0, r1 */ + + MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */ +ALIGNTEXT16 +LLBL( G3TP2M_3 ): + + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ + + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ + + JNZ ( LLBL( G3TP2M_2 ) ) /* cnt > 0 ? -> process next vertex */ + +LLBL( G3TP2M_4 ): + + FEMMS POP_L ( EBP ) POP_L ( EBX ) POP_L ( EDI ) POP_L ( ESI ) - - POP_L ( ESI ) RET - - ALIGNTEXT16 -GLOBL GLNAME(gl_3dnow_transform_points3_3d_no_rot_masked) -GLNAME( gl_3dnow_transform_points3_3d_no_rot_masked ): - - PUSH_L ( ESI ) - MOV_L ( REGOFF(8, ESP), ECX ) - MOV_L ( REGOFF(12, ESP), ESI ) - MOV_L ( REGOFF(16, ESP), EAX ) - MOV_L ( CONST(3), REGOFF(16, ECX) ) - OR_B ( CONST(7), REGOFF(20, ECX) ) - MOV_L ( REGOFF(8, EAX), EDX ) - MOV_L ( EDX, REGOFF(8, ECX) ) - -ALIGNTEXT32 +GLOBL GLNAME( gl_3dnow_transform_points3_2d_no_rot_masked ) +GLNAME( gl_3dnow_transform_points3_2d_no_rot_masked ): PUSH_L ( ESI ) PUSH_L ( EDI ) PUSH_L ( EBX ) PUSH_L ( EBP ) - MOV_L ( REGOFF(4, ECX), EDX ) - MOV_L ( ESI, ECX ) - MOV_L ( REGOFF(8, EAX), ESI ) /* count */ - MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */ - MOV_L ( REGOFF(4, EAX), EAX ) - MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */ - MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */ - - FEMMS + MOV_L ( ARG_DEST, ECX ) + MOV_L ( ARG_MATRIX, ESI ) + MOV_L ( ARG_SOURCE, EAX ) + MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) ) + OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) ) + MOV_L ( REGOFF(V4F_COUNT, EAX), EDX ) + MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) ) - MOVD ( REGIND(ECX), MM0 ) /* | m00 */ - MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ + MOV_L ( REGOFF(V4F_START, ECX), EDX ) + MOV_L ( ESI, ECX ) + MOV_L ( REGOFF(V4F_COUNT, EAX), ESI ) + MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI ) + MOV_L ( REGOFF(V4F_START, EAX), EAX ) + MOV_L ( ARG_CLIP, EBP ) + MOV_B ( ARG_FLAG, BL ) - PSLLQ ( CONST(32), MM7 ) /* m11 | */ - POR ( MM7, MM0 ) /* m11 | m00 */ + MOVD ( REGIND(ECX), MM0 ) /* | m00 */ + MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ - MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */ - PUNPCKLDQ ( MM2, MM2 ) /* m22 | m22 */ + PSLLQ ( CONST(32), MM7 ) /* m11 | */ + POR ( MM7, MM0 ) /* m11 | m00 */ - MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */ - MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */ + MOVQ ( REGOFF(48, ECX), MM1 ) /* m31 | m30 */ - PUNPCKLDQ ( MM3, MM3 ) /* m32 | m32 */ - CMP_L ( CONST(0), ESI ) - JE ( LLBL(G3TP3NRM_4) ) + TEST_L ( ESI, ESI ) + JZ ( LLBL( G3TP2NRM_4 ) ) +ALIGNTEXT16 +LLBL( G3TP2NRM_2 ): -ALIGNTEXT32 -LLBL(G3TP3NRM_2): + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TP2NRM_3 ) ) /* skip vertex */ - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TP3NRM_3) /* skip vertex */ ) + MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ + MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ - MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ - MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ + PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */ + MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 (=x2) */ - PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */ - PFMUL ( MM2, MM5 ) /* | x2*m22 */ + PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */ + MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */ - PFADD ( MM1, MM4 ) /* x1*m11+m31 | x0*m00+m30 */ - PFADD ( MM3, MM5 ) /* | x2*m22+m32 */ +ALIGNTEXT16 +LLBL( G3TP2NRM_3 ): - MOVQ ( MM4, REGIND(EDX) ) /* write r0, r1 */ - MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */ + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ -LLBL(G3TP3NRM_3): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ + JNZ ( LLBL( G3TP2NRM_2 ) ) /* cnt > 0 ? -> process next vertex */ - JA ( LLBL(G3TP3NRM_2) /* cnt > 0 ? -> process next vertex */ ) +LLBL( G3TP2NRM_4 ): -LLBL(G3TP3NRM_4): FEMMS - POP_L ( EBP ) POP_L ( EBX ) POP_L ( EDI ) POP_L ( ESI ) - - POP_L ( ESI ) RET ALIGNTEXT16 -GLOBL GLNAME(gl_3dnow_transform_points3_perspective_masked) -GLNAME( gl_3dnow_transform_points3_perspective_masked ): - - PUSH_L ( ESI ) - MOV_L ( REGOFF(8, ESP), ECX ) - MOV_L ( REGOFF(12, ESP), ESI ) - MOV_L ( REGOFF(16, ESP), EAX ) - MOV_L ( CONST(4), REGOFF(16, ECX) ) - OR_B ( CONST(15), REGOFF(20, ECX) ) - MOV_L ( REGOFF(8, EAX), EDX ) - MOV_L ( EDX, REGOFF(8, ECX) ) - -ALIGNTEXT32 +GLOBL GLNAME( gl_3dnow_transform_points3_identity_masked ) +GLNAME( gl_3dnow_transform_points3_identity_masked ): PUSH_L ( ESI ) PUSH_L ( EDI ) PUSH_L ( EBX ) PUSH_L ( EBP ) - MOV_L ( REGOFF(4, ECX), EDX ) - MOV_L ( ESI, ECX ) - MOV_L ( REGOFF(8, EAX), ESI ) /* count */ - MOV_L ( REGOFF(12, EAX), EDI ) /* input stride */ - MOV_L ( REGOFF(4, EAX), EAX ) - MOV_L ( REGOFF(36, ESP), EBP ) /* clipmask */ - MOV_B ( REGOFF(40, ESP), BL ) /* clip flag */ - - FEMMS - - MOVD ( REGIND(ECX), MM0 ) /* | m00 */ - MOVD ( REGOFF(20, ECX), MM7 ) /* | m11 */ - - PSLLQ ( CONST(32), MM7 ) /* m11 | */ - POR ( MM7, MM0 ) /* m11 | m00 */ - - MOVQ ( REGOFF(32, ECX), MM1 ) /* m21 | m20 */ - MOVD ( REGOFF(40, ECX), MM2 ) /* | m22 */ - - MOVD ( REGOFF(56, ECX), MM3 ) /* | m32 */ - CMP_L ( CONST(0), ESI ) - JE ( LLBL(G3TPPM_4) ) + MOV_L ( ARG_DEST, ECX ) + MOV_L ( ARG_MATRIX, ESI ) + MOV_L ( ARG_SOURCE, EAX ) + MOV_L ( CONST(3), REGOFF(V4F_SIZE, ECX) ) + OR_B ( CONST(VEC_SIZE_3), REGOFF(V4F_FLAGS, ECX) ) + MOV_L ( REGOFF(V4F_COUNT, EAX), EDX ) + MOV_L ( EDX, REGOFF(V4F_COUNT, ECX) ) + MOV_L ( REGOFF(V4F_START, ECX), EDX ) + MOV_L ( ESI, ECX ) + MOV_L ( REGOFF(V4F_COUNT, EAX), ESI ) + MOV_L ( REGOFF(V4F_STRIDE, EAX), EDI ) + MOV_L ( REGOFF(V4F_START, EAX), EAX ) + MOV_L ( ARG_CLIP, EBP ) + MOV_B ( ARG_FLAG, BL ) -ALIGNTEXT32 -LLBL(G3TPPM_2): - - TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ - JNZ ( LLBL(G3TPPM_3) /* skip vertex */ ) - - MOVQ ( REGIND(EAX), MM4 ) /* x1 | x0 */ - MOVD ( REGOFF(8, EAX), MM5 ) /* | x2 */ - - PFMUL ( MM0, MM4 ) /* x1*m11 | x0*m00 */ - MOVQ ( MM5, MM6 ) /* | x2 */ +ALIGNTEXT16 +LLBL( G3TPIM_2 ): - PUNPCKLDQ ( MM5, MM5 ) /* x2 | x2 */ - PFMUL ( MM1, MM5 ) /* x2*m21 | x2*m20 */ + TEST_B ( BL, REGIND(EBP) ) /* mask [i] != clip flag ?? */ + JNZ ( LLBL( G3TPIM_3 ) ) /* skip vertex */ - PFADD ( MM4, MM5 ) /* x1*m11+x2*m21 | x0*m00+x2*m20 */ - MOVQ ( MM5, REGIND(EDX) ) /* write r0, r1 */ + MOVQ ( REGIND(EAX), MM0 ) /* x1 | x0 */ + MOVD ( REGOFF(8, EAX), MM1 ) /* | x2 */ - MOVQ ( MM6, MM5 ) /* | x2 */ - PFMUL ( MM2, MM5 ) /* | x2*m22 */ + MOVQ ( MM0, REGIND(EDX) ) /* r1 | r0 */ + MOVD ( MM1, REGOFF(8, EDX) ) /* | r2 */ - PFADD ( MM3, MM5 ) /* | x2*m22+m32 */ - PFSUBR ( MM7, MM6 ) /* (LO mm7 == 0) | -x2 */ +LLBL( G3TPIM_3 ): - MOVD ( MM5, REGOFF(8, EDX) ) /* write r2 */ - MOVD ( MM6, REGOFF(12, EDX) ) /* write r3 */ + ADD_L ( EDI, EAX ) /* next vertex */ + ADD_L ( CONST(16), EDX ) /* next r */ -LLBL(G3TPPM_3): - ADD_L ( EDI, EAX ) /* next vertex */ - ADD_L ( CONST(16), EDX ) /* next r */ + INC_L ( EBP ) /* next clipmask */ + DEC_L ( ESI ) /* decrement vertex counter */ - INC_L ( EBP ) /* next clipmask */ - DEC_L ( ESI ) /* decrement vertex counter */ + JNZ ( LLBL( G3TPIM_2 ) ) /* cnt > 0 ? -> process next vertex */ - JA ( LLBL(G3TPPM_2) /* cnt > 0 ? -> process next vertex */ ) +LLBL( G3TPIM_4 ): -LLBL(G3TPPM_4): FEMMS - POP_L ( EBP ) POP_L ( EBX ) POP_L ( EDI ) POP_L ( ESI ) - - POP_L ( ESI ) RET - - - - - - - |