1f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/*
2f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Copyright 2003 Tungsten Graphics, inc.
3f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * All Rights Reserved.
4f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org *
5f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Permission is hereby granted, free of charge, to any person obtaining a
6f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * copy of this software and associated documentation files (the "Software"),
7f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * to deal in the Software without restriction, including without limitation
8f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * on the rights to use, copy, modify, merge, publish, distribute, sub
9f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * license, and/or sell copies of the Software, and to permit persons to whom
10f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * the Software is furnished to do so, subject to the following conditions:
11f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org *
12f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * The above copyright notice and this permission notice (including the next
13f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * paragraph) shall be included in all copies or substantial portions of the
14f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Software.
15f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org *
16f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
17f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
18f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT.  IN NO EVENT SHALL
19f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * TUNGSTEN GRAPHICS AND/OR THEIR SUPPLIERS BE LIABLE FOR ANY CLAIM,
20f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
21f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE
22f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * USE OR OTHER DEALINGS IN THE SOFTWARE.
23f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org *
24f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Authors:
25f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org *    Keith Whitwell <keithw@tungstengraphics.com>
26f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */
27f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
28f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/glheader.h"
29f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/context.h"
30f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/colormac.h"
31f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/simple_list.h"
32f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/enums.h"
33f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "swrast/s_chan.h"
34f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "t_context.h"
35f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "t_vertex.h"
36f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
37f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#if defined(USE_SSE_ASM)
38f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
39f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "x86/rtasm/x86sse.h"
40f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "x86/common_x86_asm.h"
41f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
42f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
43f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/**
44f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Number of bytes to allocate for generated SSE functions
45f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */
46f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define MAX_SSE_CODE_SIZE 1024
47f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
48f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
49f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define X    0
50f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define Y    1
51f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define Z    2
52f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define W    3
53f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
54f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
55f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstruct x86_program {
56f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_function func;
57f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
58f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct gl_context *ctx;
59f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   GLboolean inputs_safe;
60f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   GLboolean outputs_safe;
61f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   GLboolean have_sse2;
62f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
63f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg identity;
64f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg chan0;
65f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org};
66f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
67f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
68f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic struct x86_reg get_identity( struct x86_program *p )
69f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
70f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   return p->identity;
71f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
72f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
73f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_4( struct x86_program *p,
74f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
75f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
76f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
77f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movups(&p->func, dest, arg0);
78f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
79f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
80f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_3( struct x86_program *p,
81f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
82f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
83f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
84f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Have to jump through some hoops:
85f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    *
86f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * c 0 0 0
87f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * c 0 0 1
88f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * 0 0 c 1
89f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * a b c 1
90f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
91f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movss(&p->func, dest, x86_make_disp(arg0, 8));
92f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_shufps(&p->func, dest, get_identity(p), SHUF(X,Y,Z,W) );
93f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_shufps(&p->func, dest, dest, SHUF(Y,Z,X,W) );
94f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movlps(&p->func, dest, arg0);
95f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
96f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
97f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_2( struct x86_program *p,
98f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
99f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
100f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
101f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Initialize from identity, then pull in low two words:
102f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
103f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movups(&p->func, dest, get_identity(p));
104f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movlps(&p->func, dest, arg0);
105f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
106f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
107f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_1( struct x86_program *p,
108f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
109f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
110f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
111f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Pull in low word, then swizzle in identity */
112f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movss(&p->func, dest, arg0);
113f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_shufps(&p->func, dest, get_identity(p), SHUF(X,Y,Z,W) );
114f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
115f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
116f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
117f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
118f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load3f_3( struct x86_program *p,
119f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
120f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
121f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
122f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Over-reads by 1 dword - potential SEGV if input is a vertex
123f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * array.
124f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
125f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (p->inputs_safe) {
126f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movups(&p->func, dest, arg0);
127f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
128f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   else {
129f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* c 0 0 0
130f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       * c c c c
131f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       * a b c c
132f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
133f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movss(&p->func, dest, x86_make_disp(arg0, 8));
134f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_shufps(&p->func, dest, dest, SHUF(X,X,X,X));
135f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movlps(&p->func, dest, arg0);
136f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
137f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
138f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
139f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load3f_2( struct x86_program *p,
140f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
141f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
142f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
143f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   emit_load4f_2(p, dest, arg0);
144f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
145f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
146f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load3f_1( struct x86_program *p,
147f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
148f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
149f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
150f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Loading from memory erases the upper bits. */
151f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movss(&p->func, dest, arg0);
152f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
153f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
154f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load2f_2( struct x86_program *p,
155f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
156f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
157f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
158f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movlps(&p->func, dest, arg0);
159f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
160f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
161f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load2f_1( struct x86_program *p,
162f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
163f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
164f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
165f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Loading from memory erases the upper bits. */
166f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movss(&p->func, dest, arg0);
167f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
168f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
169f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load1f_1( struct x86_program *p,
170f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
171f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
172f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
173f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movss(&p->func, dest, arg0);
174f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
175f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
176f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void (*load[4][4])( struct x86_program *p,
177f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
178f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 ) = {
179f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   { emit_load1f_1,
180f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load1f_1,
181f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load1f_1,
182f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load1f_1 },
183f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
184f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   { emit_load2f_1,
185f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load2f_2,
186f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load2f_2,
187f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load2f_2 },
188f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
189f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   { emit_load3f_1,
190f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load3f_2,
191f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load3f_3,
192f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load3f_3 },
193f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
194f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   { emit_load4f_1,
195f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load4f_2,
196f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load4f_3,
197f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org     emit_load4f_4 }
198f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org};
199f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
200f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load( struct x86_program *p,
201f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org		       struct x86_reg dest,
202f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org		       GLuint sz,
203f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org		       struct x86_reg src,
204f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org		       GLuint src_sz)
205f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
206f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   load[sz-1][src_sz-1](p, dest, src);
207f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
208f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
209f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store4f( struct x86_program *p,
210f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			  struct x86_reg dest,
211f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			  struct x86_reg arg0 )
212f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
213f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movups(&p->func, dest, arg0);
214f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
215f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
216f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store3f( struct x86_program *p,
217f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			  struct x86_reg dest,
218f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			  struct x86_reg arg0 )
219f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
220f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (p->outputs_safe) {
221f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* Emit the extra dword anyway.  This may hurt writecombining,
222f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       * may cause other problems.
223f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
224f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movups(&p->func, dest, arg0);
225f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
226f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   else {
227f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* Alternate strategy - emit two, shuffle, emit one.
228f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
229f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movlps(&p->func, dest, arg0);
230f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_shufps(&p->func, arg0, arg0, SHUF(Z,Z,Z,Z) ); /* NOTE! destructive */
231f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movss(&p->func, x86_make_disp(dest,8), arg0);
232f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
233f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
234f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
235f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store2f( struct x86_program *p,
236f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg dest,
237f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			   struct x86_reg arg0 )
238f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
239f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movlps(&p->func, dest, arg0);
240f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
241f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
242f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store1f( struct x86_program *p,
243f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			  struct x86_reg dest,
244f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			  struct x86_reg arg0 )
245f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
246f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movss(&p->func, dest, arg0);
247f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
248f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
249f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
250f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void (*store[4])( struct x86_program *p,
251f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct x86_reg dest,
252f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct x86_reg arg0 ) =
253f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
254f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   emit_store1f,
255f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   emit_store2f,
256f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   emit_store3f,
257f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   emit_store4f
258f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org};
259f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
260f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store( struct x86_program *p,
261f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			struct x86_reg dest,
262f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			GLuint sz,
263f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			struct x86_reg temp )
264f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
265f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
266f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   store[sz-1](p, dest, temp);
267f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
268f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
269f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_pack_store_4ub( struct x86_program *p,
270f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org				 struct x86_reg dest,
271f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org				 struct x86_reg temp )
272f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
273f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Scale by 255.0
274f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
275f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_mulps(&p->func, temp, p->chan0);
276f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
277f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (p->have_sse2) {
278f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse2_cvtps2dq(&p->func, temp, temp);
279f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse2_packssdw(&p->func, temp, temp);
280f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse2_packuswb(&p->func, temp, temp);
281f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movss(&p->func, dest, temp);
282f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
283f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   else {
284f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      struct x86_reg mmx0 = x86_make_reg(file_MMX, 0);
285f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      struct x86_reg mmx1 = x86_make_reg(file_MMX, 1);
286f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_cvtps2pi(&p->func, mmx0, temp);
287f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movhlps(&p->func, temp, temp);
288f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_cvtps2pi(&p->func, mmx1, temp);
289f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      mmx_packssdw(&p->func, mmx0, mmx1);
290f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      mmx_packuswb(&p->func, mmx0, mmx0);
291f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      mmx_movd(&p->func, dest, mmx0);
292f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
293f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
294f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
295f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic GLint get_offset( const void *a, const void *b )
296f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
297f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   return (const char *)b - (const char *)a;
298f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
299f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
300f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/* Not much happens here.  Eventually use this function to try and
301f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * avoid saving/reloading the source pointers each vertex (if some of
302f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * them can fit in registers).
303f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */
304f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void get_src_ptr( struct x86_program *p,
305f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct x86_reg srcREG,
306f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct x86_reg vtxREG,
307f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct tnl_clipspace_attr *a )
308f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
309f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct tnl_clipspace *vtx = GET_VERTEX_STATE(p->ctx);
310f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg ptr_to_src = x86_make_disp(vtxREG, get_offset(vtx, &a->inputptr));
311f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
312f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Load current a[j].inputptr
313f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
314f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_mov(&p->func, srcREG, ptr_to_src);
315f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
316f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
317f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void update_src_ptr( struct x86_program *p,
318f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct x86_reg srcREG,
319f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct x86_reg vtxREG,
320f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org			 struct tnl_clipspace_attr *a )
321f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
322f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (a->inputstride) {
323f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      struct tnl_clipspace *vtx = GET_VERTEX_STATE(p->ctx);
324f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      struct x86_reg ptr_to_src = x86_make_disp(vtxREG, get_offset(vtx, &a->inputptr));
325f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
326f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* add a[j].inputstride (hardcoded value - could just as easily
327f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       * pull the stride value from memory each time).
328f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
329f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      x86_lea(&p->func, srcREG, x86_make_disp(srcREG, a->inputstride));
330f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
331f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* save new value of a[j].inputptr
332f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
333f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      x86_mov(&p->func, ptr_to_src, srcREG);
334f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
335f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
336f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
337f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
338f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/* Lots of hardcoding
339f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org *
340f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * EAX -- pointer to current output vertex
341f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * ECX -- pointer to current attribute
342f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org *
343f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */
344f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic GLboolean build_vertex_emit( struct x86_program *p )
345f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
346f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct gl_context *ctx = p->ctx;
347f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   TNLcontext *tnl = TNL_CONTEXT(ctx);
348f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct tnl_clipspace *vtx = GET_VERTEX_STATE(ctx);
349f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   GLuint j = 0;
350f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
351f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg vertexEAX = x86_make_reg(file_REG32, reg_AX);
352f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg srcECX = x86_make_reg(file_REG32, reg_CX);
353f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg countEBP = x86_make_reg(file_REG32, reg_BP);
354f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg vtxESI = x86_make_reg(file_REG32, reg_SI);
355f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg temp = x86_make_reg(file_XMM, 0);
356f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg vp0 = x86_make_reg(file_XMM, 1);
357f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg vp1 = x86_make_reg(file_XMM, 2);
358f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_reg temp2 = x86_make_reg(file_XMM, 3);
359f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   GLubyte *fixup, *label;
360f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
361f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Push a few regs?
362f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
363f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_push(&p->func, countEBP);
364f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_push(&p->func, vtxESI);
365f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
366f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
367f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Get vertex count, compare to zero
368f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
369f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_xor(&p->func, srcECX, srcECX);
370f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_mov(&p->func, countEBP, x86_fn_arg(&p->func, 2));
371f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_cmp(&p->func, countEBP, srcECX);
372f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   fixup = x86_jcc_forward(&p->func, cc_E);
373f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
374f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Initialize destination register.
375f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
376f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_mov(&p->func, vertexEAX, x86_fn_arg(&p->func, 3));
377f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
378f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Dereference ctx to get tnl, then vtx:
379f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
380f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_mov(&p->func, vtxESI, x86_fn_arg(&p->func, 1));
381f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_mov(&p->func, vtxESI, x86_make_disp(vtxESI, get_offset(ctx, &ctx->swtnl_context)));
382f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   vtxESI = x86_make_disp(vtxESI, get_offset(tnl, &tnl->clipspace));
383f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
384f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
385f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Possibly load vp0, vp1 for viewport calcs:
386f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
387f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (vtx->need_viewport) {
388f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movups(&p->func, vp0, x86_make_disp(vtxESI, get_offset(vtx, &vtx->vp_scale[0])));
389f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      sse_movups(&p->func, vp1, x86_make_disp(vtxESI, get_offset(vtx, &vtx->vp_xlate[0])));
390f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
391f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
392f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* always load, needed or not:
393f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
394f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movups(&p->func, p->chan0, x86_make_disp(vtxESI, get_offset(vtx, &vtx->chan_scale[0])));
395f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   sse_movups(&p->func, p->identity, x86_make_disp(vtxESI, get_offset(vtx, &vtx->identity[0])));
396f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
397f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Note address for loop jump */
398f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   label = x86_get_label(&p->func);
399f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
400f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Emit code for each of the attributes.  Currently routes
401f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * everything through SSE registers, even when it might be more
402f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * efficient to stick with regular old x86.  No optimization or
403f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * other tricks - enough new ground to cover here just getting
404f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    * things working.
405f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
406f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   while (j < vtx->attr_count) {
407f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      struct tnl_clipspace_attr *a = &vtx->attr[j];
408f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      struct x86_reg dest = x86_make_disp(vertexEAX, a->vertoffset);
409f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
410f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* Now, load an XMM reg from src, perhaps transform, then save.
411f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       * Could be shortcircuited in specific cases:
412f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
413f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      switch (a->format) {
414f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_1F:
415f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
416f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 1, x86_deref(srcECX), a->inputsize);
417f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_store(p, dest, 1, temp);
418f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
419f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
420f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_2F:
421f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
422f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 2, x86_deref(srcECX), a->inputsize);
423f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_store(p, dest, 2, temp);
424f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
425f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
426f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_3F:
427f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 /* Potentially the worst case - hardcode 2+1 copying:
428f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	  */
429f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 if (0) {
430f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, a);
431f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize);
432f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_store(p, dest, 3, temp);
433f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, a);
434f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
435f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 else {
436f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, a);
437f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp, 2, x86_deref(srcECX), a->inputsize);
438f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_store(p, dest, 2, temp);
439f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    if (a->inputsize > 2) {
440f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	       emit_load(p, temp, 1, x86_make_disp(srcECX, 8), 1);
441f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	       emit_store(p, x86_make_disp(dest,8), 1, temp);
442f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    }
443f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    else {
444f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	       sse_movss(&p->func, x86_make_disp(dest,8), get_identity(p));
445f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    }
446f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, a);
447f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
448f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
449f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_4F:
450f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
451f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
452f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_store(p, dest, 4, temp);
453f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
454f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
455f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_2F_VIEWPORT:
456f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
457f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 2, x86_deref(srcECX), a->inputsize);
458f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_mulps(&p->func, temp, vp0);
459f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_addps(&p->func, temp, vp1);
460f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_store(p, dest, 2, temp);
461f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
462f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
463f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_3F_VIEWPORT:
464f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
465f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize);
466f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_mulps(&p->func, temp, vp0);
467f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_addps(&p->func, temp, vp1);
468f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_store(p, dest, 3, temp);
469f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
470f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
471f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_4F_VIEWPORT:
472f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
473f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
474f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_mulps(&p->func, temp, vp0);
475f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_addps(&p->func, temp, vp1);
476f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_store(p, dest, 4, temp);
477f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
478f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
479f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_3F_XYW:
480f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
481f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
482f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_shufps(&p->func, temp, temp, SHUF(X,Y,W,Z));
483f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_store(p, dest, 3, temp);
484f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
485f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
486f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
487f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_1UB_1F:
488f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 /* Test for PAD3 + 1UB:
489f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	  */
490f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 if (j > 0 &&
491f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	     a[-1].vertoffset + a[-1].vertattrsize <= a->vertoffset - 3)
492f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 {
493f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, a);
494f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp, 1, x86_deref(srcECX), a->inputsize);
495f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    sse_shufps(&p->func, temp, temp, SHUF(X,X,X,X));
496f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_pack_store_4ub(p, x86_make_disp(dest, -3), temp); /* overkill! */
497f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, a);
498f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
499f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 else {
500f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    printf("Can't emit 1ub %x %x %d\n", a->vertoffset, a[-1].vertoffset, a[-1].vertattrsize );
501f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    return GL_FALSE;
502f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
503f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
504f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_3UB_3F_RGB:
505f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_3UB_3F_BGR:
506f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 /* Test for 3UB + PAD1:
507f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	  */
508f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 if (j == vtx->attr_count - 1 ||
509f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	     a[1].vertoffset >= a->vertoffset + 4) {
510f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, a);
511f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize);
512f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    if (a->format == EMIT_3UB_3F_BGR)
513f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	       sse_shufps(&p->func, temp, temp, SHUF(Z,Y,X,W));
514f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_pack_store_4ub(p, dest, temp);
515f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, a);
516f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
517f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 /* Test for 3UB + 1UB:
518f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	  */
519f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 else if (j < vtx->attr_count - 1 &&
520f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org		  a[1].format == EMIT_1UB_1F &&
521f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org		  a[1].vertoffset == a->vertoffset + 3) {
522f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, a);
523f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize);
524f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, a);
525f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
526f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    /* Make room for incoming value:
527f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	     */
528f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    sse_shufps(&p->func, temp, temp, SHUF(W,X,Y,Z));
529f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
530f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, &a[1]);
531f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp2, 1, x86_deref(srcECX), a[1].inputsize);
532f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    sse_movss(&p->func, temp, temp2);
533f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, &a[1]);
534f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
535f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    /* Rearrange and possibly do BGR conversion:
536f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	     */
537f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    if (a->format == EMIT_3UB_3F_BGR)
538f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	       sse_shufps(&p->func, temp, temp, SHUF(W,Z,Y,X));
539f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    else
540f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	       sse_shufps(&p->func, temp, temp, SHUF(Y,Z,W,X));
541f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
542f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_pack_store_4ub(p, dest, temp);
543f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    j++;		/* NOTE: two attrs consumed */
544f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
545f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 else {
546f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    printf("Can't emit 3ub\n");
547f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    return GL_FALSE;	/* add this later */
548f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
549f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
550f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
551f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_4UB_4F_RGBA:
552f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
553f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
554f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_pack_store_4ub(p, dest, temp);
555f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
556f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
557f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_4UB_4F_BGRA:
558f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
559f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
560f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_shufps(&p->func, temp, temp, SHUF(Z,Y,X,W));
561f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_pack_store_4ub(p, dest, temp);
562f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
563f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
564f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_4UB_4F_ARGB:
565f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
566f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
567f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_shufps(&p->func, temp, temp, SHUF(W,X,Y,Z));
568f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_pack_store_4ub(p, dest, temp);
569f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
570f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
571f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_4UB_4F_ABGR:
572f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 get_src_ptr(p, srcECX, vtxESI, a);
573f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
574f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 sse_shufps(&p->func, temp, temp, SHUF(W,Z,Y,X));
575f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 emit_pack_store_4ub(p, dest, temp);
576f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 update_src_ptr(p, srcECX, vtxESI, a);
577f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
578f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      case EMIT_4CHAN_4F_RGBA:
579f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 switch (CHAN_TYPE) {
580f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 case GL_UNSIGNED_BYTE:
581f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, a);
582f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
583f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_pack_store_4ub(p, dest, temp);
584f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, a);
585f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    break;
586f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 case GL_FLOAT:
587f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    get_src_ptr(p, srcECX, vtxESI, a);
588f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize);
589f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    emit_store(p, dest, 4, temp);
590f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    update_src_ptr(p, srcECX, vtxESI, a);
591f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    break;
592f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 case GL_UNSIGNED_SHORT:
593f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 default:
594f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    printf("unknown CHAN_TYPE %s\n", _mesa_lookup_enum_by_nr(CHAN_TYPE));
595f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	    return GL_FALSE;
596f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 }
597f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 break;
598f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      default:
599f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 printf("unknown a[%d].format %d\n", j, a->format);
600f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org	 return GL_FALSE;	/* catch any new opcodes */
601f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      }
602f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
603f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* Increment j by at least 1 - may have been incremented above also:
604f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
605f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      j++;
606f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
607f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
608f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Next vertex:
609f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
610f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_lea(&p->func, vertexEAX, x86_make_disp(vertexEAX, vtx->vertex_size));
611f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
612f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* decr count, loop if not zero
613f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
614f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_dec(&p->func, countEBP);
615f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_test(&p->func, countEBP, countEBP);
616f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_jcc(&p->func, cc_NZ, label);
617f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
618f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Exit mmx state?
619f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
620f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (p->func.need_emms)
621f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      mmx_emms(&p->func);
622f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
623f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Land forward jump here:
624f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
625f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_fixup_fwd_jump(&p->func, fixup);
626f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
627f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Pop regs and return
628f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org    */
629f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_pop(&p->func, x86_get_base_reg(vtxESI));
630f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_pop(&p->func, countEBP);
631f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   x86_ret(&p->func);
632f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
633f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   assert(!vtx->emit);
634f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   vtx->emit = (tnl_emit_func)x86_get_func(&p->func);
635f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
636f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   assert( (char *) p->func.csr - (char *) p->func.store <= MAX_SSE_CODE_SIZE );
637f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   return GL_TRUE;
638f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
639f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
640f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
641f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
642f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgvoid _tnl_generate_sse_emit( struct gl_context *ctx )
643f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
644f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct tnl_clipspace *vtx = GET_VERTEX_STATE(ctx);
645f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   struct x86_program p;
646f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
647f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (!cpu_has_xmm) {
648f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      vtx->codegen_emit = NULL;
649f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      return;
650f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
651f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
652f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   memset(&p, 0, sizeof(p));
653f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
654f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   p.ctx = ctx;
655f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   p.inputs_safe = 0;		/* for now */
656f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   p.outputs_safe = 0;		/* for now */
657f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   p.have_sse2 = cpu_has_xmm2;
658f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   p.identity = x86_make_reg(file_XMM, 6);
659f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   p.chan0 = x86_make_reg(file_XMM, 7);
660f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
661f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (!x86_init_func_size(&p.func, MAX_SSE_CODE_SIZE)) {
662f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      vtx->emit = NULL;
663f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      return;
664f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
665f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
666f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   if (build_vertex_emit(&p)) {
667f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      _tnl_register_fastpath( vtx, GL_TRUE );
668f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
669f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   else {
670f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      /* Note the failure so that we don't keep trying to codegen an
671f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       * impossible state:
672f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org       */
673f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      _tnl_register_fastpath( vtx, GL_FALSE );
674f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org      x86_release_func(&p.func);
675f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   }
676f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
677f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
678f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#else
679f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
680f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgvoid _tnl_generate_sse_emit( struct gl_context *ctx )
681f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{
682f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org   /* Dummy version for when USE_SSE_ASM not defined */
683f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}
684f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org
685f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#endif
686