1f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/* 2f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Copyright 2003 Tungsten Graphics, inc. 3f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * All Rights Reserved. 4f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 5f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Permission is hereby granted, free of charge, to any person obtaining a 6f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * copy of this software and associated documentation files (the "Software"), 7f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * to deal in the Software without restriction, including without limitation 8f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * on the rights to use, copy, modify, merge, publish, distribute, sub 9f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * license, and/or sell copies of the Software, and to permit persons to whom 10f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * the Software is furnished to do so, subject to the following conditions: 11f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 12f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * The above copyright notice and this permission notice (including the next 13f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * paragraph) shall be included in all copies or substantial portions of the 14f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Software. 15f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 16f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR 17f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, 18f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT. IN NO EVENT SHALL 19f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * TUNGSTEN GRAPHICS AND/OR THEIR SUPPLIERS BE LIABLE FOR ANY CLAIM, 20f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR 21f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE 22f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * USE OR OTHER DEALINGS IN THE SOFTWARE. 23f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 24f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Authors: 25f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Keith Whitwell <keithw@tungstengraphics.com> 26f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 27f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 28f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/glheader.h" 29f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/context.h" 30f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/colormac.h" 31f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/simple_list.h" 32f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "main/enums.h" 33f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "swrast/s_chan.h" 34f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "t_context.h" 35f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "t_vertex.h" 36f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 37f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#if defined(USE_SSE_ASM) 38f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 39f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "x86/rtasm/x86sse.h" 40f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#include "x86/common_x86_asm.h" 41f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 42f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 43f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/** 44f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Number of bytes to allocate for generated SSE functions 45f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 46f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define MAX_SSE_CODE_SIZE 1024 47f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 48f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 49f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define X 0 50f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define Y 1 51f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define Z 2 52f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#define W 3 53f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 54f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 55f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstruct x86_program { 56f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_function func; 57f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 58f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct gl_context *ctx; 59f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLboolean inputs_safe; 60f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLboolean outputs_safe; 61f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLboolean have_sse2; 62f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 63f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg identity; 64f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg chan0; 65f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}; 66f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 67f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 68f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic struct x86_reg get_identity( struct x86_program *p ) 69f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 70f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return p->identity; 71f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 72f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 73f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_4( struct x86_program *p, 74f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 75f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 76f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 77f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, dest, arg0); 78f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 79f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 80f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_3( struct x86_program *p, 81f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 82f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 83f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 84f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Have to jump through some hoops: 85f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 86f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * c 0 0 0 87f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * c 0 0 1 88f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 0 0 c 1 89f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * a b c 1 90f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 91f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, x86_make_disp(arg0, 8)); 92f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, dest, get_identity(p), SHUF(X,Y,Z,W) ); 93f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, dest, dest, SHUF(Y,Z,X,W) ); 94f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movlps(&p->func, dest, arg0); 95f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 96f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 97f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_2( struct x86_program *p, 98f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 99f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 100f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 101f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Initialize from identity, then pull in low two words: 102f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 103f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, dest, get_identity(p)); 104f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movlps(&p->func, dest, arg0); 105f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 106f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 107f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load4f_1( struct x86_program *p, 108f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 109f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 110f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 111f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Pull in low word, then swizzle in identity */ 112f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, arg0); 113f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, dest, get_identity(p), SHUF(X,Y,Z,W) ); 114f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 115f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 116f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 117f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 118f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load3f_3( struct x86_program *p, 119f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 120f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 121f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 122f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Over-reads by 1 dword - potential SEGV if input is a vertex 123f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * array. 124f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 125f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (p->inputs_safe) { 126f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, dest, arg0); 127f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 128f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 129f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* c 0 0 0 130f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * c c c c 131f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * a b c c 132f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 133f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, x86_make_disp(arg0, 8)); 134f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, dest, dest, SHUF(X,X,X,X)); 135f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movlps(&p->func, dest, arg0); 136f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 137f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 138f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 139f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load3f_2( struct x86_program *p, 140f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 141f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 142f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 143f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load4f_2(p, dest, arg0); 144f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 145f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 146f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load3f_1( struct x86_program *p, 147f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 148f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 149f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 150f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Loading from memory erases the upper bits. */ 151f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, arg0); 152f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 153f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 154f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load2f_2( struct x86_program *p, 155f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 156f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 157f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 158f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movlps(&p->func, dest, arg0); 159f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 160f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 161f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load2f_1( struct x86_program *p, 162f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 163f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 164f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 165f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Loading from memory erases the upper bits. */ 166f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, arg0); 167f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 168f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 169f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load1f_1( struct x86_program *p, 170f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 171f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 172f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 173f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, arg0); 174f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 175f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 176f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void (*load[4][4])( struct x86_program *p, 177f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 178f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) = { 179f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org { emit_load1f_1, 180f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load1f_1, 181f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load1f_1, 182f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load1f_1 }, 183f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 184f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org { emit_load2f_1, 185f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load2f_2, 186f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load2f_2, 187f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load2f_2 }, 188f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 189f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org { emit_load3f_1, 190f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load3f_2, 191f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load3f_3, 192f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load3f_3 }, 193f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 194f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org { emit_load4f_1, 195f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load4f_2, 196f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load4f_3, 197f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load4f_4 } 198f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}; 199f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 200f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_load( struct x86_program *p, 201f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 202f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLuint sz, 203f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg src, 204f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLuint src_sz) 205f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 206f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org load[sz-1][src_sz-1](p, dest, src); 207f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 208f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 209f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store4f( struct x86_program *p, 210f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 211f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 212f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 213f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, dest, arg0); 214f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 215f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 216f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store3f( struct x86_program *p, 217f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 218f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 219f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 220f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (p->outputs_safe) { 221f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Emit the extra dword anyway. This may hurt writecombining, 222f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * may cause other problems. 223f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 224f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, dest, arg0); 225f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 226f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 227f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Alternate strategy - emit two, shuffle, emit one. 228f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 229f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movlps(&p->func, dest, arg0); 230f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, arg0, arg0, SHUF(Z,Z,Z,Z) ); /* NOTE! destructive */ 231f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, x86_make_disp(dest,8), arg0); 232f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 233f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 234f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 235f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store2f( struct x86_program *p, 236f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 237f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 238f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 239f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movlps(&p->func, dest, arg0); 240f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 241f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 242f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store1f( struct x86_program *p, 243f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 244f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) 245f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 246f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, arg0); 247f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 248f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 249f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 250f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void (*store[4])( struct x86_program *p, 251f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 252f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg arg0 ) = 253f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 254f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store1f, 255f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store2f, 256f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store3f, 257f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store4f 258f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org}; 259f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 260f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_store( struct x86_program *p, 261f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 262f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLuint sz, 263f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg temp ) 264f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 265f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 266f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org store[sz-1](p, dest, temp); 267f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 268f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 269f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void emit_pack_store_4ub( struct x86_program *p, 270f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest, 271f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg temp ) 272f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 273f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Scale by 255.0 274f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 275f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_mulps(&p->func, temp, p->chan0); 276f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 277f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (p->have_sse2) { 278f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse2_cvtps2dq(&p->func, temp, temp); 279f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse2_packssdw(&p->func, temp, temp); 280f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse2_packuswb(&p->func, temp, temp); 281f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, dest, temp); 282f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 283f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 284f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg mmx0 = x86_make_reg(file_MMX, 0); 285f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg mmx1 = x86_make_reg(file_MMX, 1); 286f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_cvtps2pi(&p->func, mmx0, temp); 287f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movhlps(&p->func, temp, temp); 288f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_cvtps2pi(&p->func, mmx1, temp); 289f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org mmx_packssdw(&p->func, mmx0, mmx1); 290f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org mmx_packuswb(&p->func, mmx0, mmx0); 291f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org mmx_movd(&p->func, dest, mmx0); 292f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 293f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 294f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 295f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic GLint get_offset( const void *a, const void *b ) 296f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 297f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return (const char *)b - (const char *)a; 298f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 299f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 300f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/* Not much happens here. Eventually use this function to try and 301f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * avoid saving/reloading the source pointers each vertex (if some of 302f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * them can fit in registers). 303f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 304f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void get_src_ptr( struct x86_program *p, 305f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg srcREG, 306f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg vtxREG, 307f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct tnl_clipspace_attr *a ) 308f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 309f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct tnl_clipspace *vtx = GET_VERTEX_STATE(p->ctx); 310f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg ptr_to_src = x86_make_disp(vtxREG, get_offset(vtx, &a->inputptr)); 311f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 312f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Load current a[j].inputptr 313f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 314f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_mov(&p->func, srcREG, ptr_to_src); 315f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 316f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 317f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic void update_src_ptr( struct x86_program *p, 318f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg srcREG, 319f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg vtxREG, 320f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct tnl_clipspace_attr *a ) 321f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 322f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (a->inputstride) { 323f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct tnl_clipspace *vtx = GET_VERTEX_STATE(p->ctx); 324f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg ptr_to_src = x86_make_disp(vtxREG, get_offset(vtx, &a->inputptr)); 325f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 326f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* add a[j].inputstride (hardcoded value - could just as easily 327f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * pull the stride value from memory each time). 328f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 329f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_lea(&p->func, srcREG, x86_make_disp(srcREG, a->inputstride)); 330f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 331f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* save new value of a[j].inputptr 332f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 333f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_mov(&p->func, ptr_to_src, srcREG); 334f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 335f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 336f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 337f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 338f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org/* Lots of hardcoding 339f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 340f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * EAX -- pointer to current output vertex 341f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * ECX -- pointer to current attribute 342f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * 343f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 344f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgstatic GLboolean build_vertex_emit( struct x86_program *p ) 345f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 346f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct gl_context *ctx = p->ctx; 347f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org TNLcontext *tnl = TNL_CONTEXT(ctx); 348f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct tnl_clipspace *vtx = GET_VERTEX_STATE(ctx); 349f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLuint j = 0; 350f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 351f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg vertexEAX = x86_make_reg(file_REG32, reg_AX); 352f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg srcECX = x86_make_reg(file_REG32, reg_CX); 353f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg countEBP = x86_make_reg(file_REG32, reg_BP); 354f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg vtxESI = x86_make_reg(file_REG32, reg_SI); 355f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg temp = x86_make_reg(file_XMM, 0); 356f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg vp0 = x86_make_reg(file_XMM, 1); 357f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg vp1 = x86_make_reg(file_XMM, 2); 358f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg temp2 = x86_make_reg(file_XMM, 3); 359f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org GLubyte *fixup, *label; 360f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 361f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Push a few regs? 362f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 363f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_push(&p->func, countEBP); 364f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_push(&p->func, vtxESI); 365f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 366f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 367f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Get vertex count, compare to zero 368f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 369f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_xor(&p->func, srcECX, srcECX); 370f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_mov(&p->func, countEBP, x86_fn_arg(&p->func, 2)); 371f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_cmp(&p->func, countEBP, srcECX); 372f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org fixup = x86_jcc_forward(&p->func, cc_E); 373f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 374f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Initialize destination register. 375f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 376f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_mov(&p->func, vertexEAX, x86_fn_arg(&p->func, 3)); 377f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 378f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Dereference ctx to get tnl, then vtx: 379f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 380f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_mov(&p->func, vtxESI, x86_fn_arg(&p->func, 1)); 381f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_mov(&p->func, vtxESI, x86_make_disp(vtxESI, get_offset(ctx, &ctx->swtnl_context))); 382f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org vtxESI = x86_make_disp(vtxESI, get_offset(tnl, &tnl->clipspace)); 383f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 384f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 385f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Possibly load vp0, vp1 for viewport calcs: 386f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 387f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (vtx->need_viewport) { 388f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, vp0, x86_make_disp(vtxESI, get_offset(vtx, &vtx->vp_scale[0]))); 389f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, vp1, x86_make_disp(vtxESI, get_offset(vtx, &vtx->vp_xlate[0]))); 390f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 391f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 392f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* always load, needed or not: 393f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 394f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, p->chan0, x86_make_disp(vtxESI, get_offset(vtx, &vtx->chan_scale[0]))); 395f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movups(&p->func, p->identity, x86_make_disp(vtxESI, get_offset(vtx, &vtx->identity[0]))); 396f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 397f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Note address for loop jump */ 398f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org label = x86_get_label(&p->func); 399f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 400f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Emit code for each of the attributes. Currently routes 401f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * everything through SSE registers, even when it might be more 402f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * efficient to stick with regular old x86. No optimization or 403f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * other tricks - enough new ground to cover here just getting 404f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * things working. 405f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 406f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org while (j < vtx->attr_count) { 407f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct tnl_clipspace_attr *a = &vtx->attr[j]; 408f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_reg dest = x86_make_disp(vertexEAX, a->vertoffset); 409f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 410f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Now, load an XMM reg from src, perhaps transform, then save. 411f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * Could be shortcircuited in specific cases: 412f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 413f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org switch (a->format) { 414f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_1F: 415f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 416f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 1, x86_deref(srcECX), a->inputsize); 417f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 1, temp); 418f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 419f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 420f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_2F: 421f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 422f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 2, x86_deref(srcECX), a->inputsize); 423f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 2, temp); 424f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 425f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 426f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_3F: 427f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Potentially the worst case - hardcode 2+1 copying: 428f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 429f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (0) { 430f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 431f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize); 432f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 3, temp); 433f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 434f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 435f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 436f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 437f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 2, x86_deref(srcECX), a->inputsize); 438f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 2, temp); 439f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (a->inputsize > 2) { 440f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 1, x86_make_disp(srcECX, 8), 1); 441f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, x86_make_disp(dest,8), 1, temp); 442f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 443f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 444f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, x86_make_disp(dest,8), get_identity(p)); 445f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 446f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 447f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 448f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 449f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_4F: 450f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 451f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 452f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 4, temp); 453f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 454f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 455f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_2F_VIEWPORT: 456f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 457f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 2, x86_deref(srcECX), a->inputsize); 458f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_mulps(&p->func, temp, vp0); 459f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_addps(&p->func, temp, vp1); 460f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 2, temp); 461f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 462f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 463f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_3F_VIEWPORT: 464f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 465f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize); 466f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_mulps(&p->func, temp, vp0); 467f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_addps(&p->func, temp, vp1); 468f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 3, temp); 469f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 470f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 471f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_4F_VIEWPORT: 472f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 473f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 474f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_mulps(&p->func, temp, vp0); 475f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_addps(&p->func, temp, vp1); 476f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 4, temp); 477f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 478f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 479f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_3F_XYW: 480f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 481f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 482f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(X,Y,W,Z)); 483f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 3, temp); 484f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 485f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 486f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 487f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_1UB_1F: 488f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Test for PAD3 + 1UB: 489f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 490f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (j > 0 && 491f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org a[-1].vertoffset + a[-1].vertattrsize <= a->vertoffset - 3) 492f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org { 493f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 494f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 1, x86_deref(srcECX), a->inputsize); 495f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(X,X,X,X)); 496f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, x86_make_disp(dest, -3), temp); /* overkill! */ 497f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 498f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 499f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 500f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org printf("Can't emit 1ub %x %x %d\n", a->vertoffset, a[-1].vertoffset, a[-1].vertattrsize ); 501f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return GL_FALSE; 502f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 503f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 504f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_3UB_3F_RGB: 505f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_3UB_3F_BGR: 506f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Test for 3UB + PAD1: 507f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 508f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (j == vtx->attr_count - 1 || 509f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org a[1].vertoffset >= a->vertoffset + 4) { 510f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 511f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize); 512f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (a->format == EMIT_3UB_3F_BGR) 513f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(Z,Y,X,W)); 514f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, dest, temp); 515f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 516f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 517f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Test for 3UB + 1UB: 518f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 519f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else if (j < vtx->attr_count - 1 && 520f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org a[1].format == EMIT_1UB_1F && 521f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org a[1].vertoffset == a->vertoffset + 3) { 522f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 523f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 3, x86_deref(srcECX), a->inputsize); 524f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 525f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 526f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Make room for incoming value: 527f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 528f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(W,X,Y,Z)); 529f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 530f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, &a[1]); 531f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp2, 1, x86_deref(srcECX), a[1].inputsize); 532f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_movss(&p->func, temp, temp2); 533f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, &a[1]); 534f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 535f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Rearrange and possibly do BGR conversion: 536f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 537f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (a->format == EMIT_3UB_3F_BGR) 538f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(W,Z,Y,X)); 539f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else 540f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(Y,Z,W,X)); 541f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 542f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, dest, temp); 543f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org j++; /* NOTE: two attrs consumed */ 544f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 545f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 546f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org printf("Can't emit 3ub\n"); 547f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return GL_FALSE; /* add this later */ 548f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 549f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 550f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 551f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_4UB_4F_RGBA: 552f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 553f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 554f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, dest, temp); 555f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 556f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 557f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_4UB_4F_BGRA: 558f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 559f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 560f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(Z,Y,X,W)); 561f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, dest, temp); 562f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 563f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 564f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_4UB_4F_ARGB: 565f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 566f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 567f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(W,X,Y,Z)); 568f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, dest, temp); 569f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 570f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 571f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_4UB_4F_ABGR: 572f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 573f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 574f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org sse_shufps(&p->func, temp, temp, SHUF(W,Z,Y,X)); 575f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, dest, temp); 576f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 577f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 578f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case EMIT_4CHAN_4F_RGBA: 579f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org switch (CHAN_TYPE) { 580f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case GL_UNSIGNED_BYTE: 581f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 582f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 583f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_pack_store_4ub(p, dest, temp); 584f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 585f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 586f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case GL_FLOAT: 587f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org get_src_ptr(p, srcECX, vtxESI, a); 588f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_load(p, temp, 4, x86_deref(srcECX), a->inputsize); 589f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org emit_store(p, dest, 4, temp); 590f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org update_src_ptr(p, srcECX, vtxESI, a); 591f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 592f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org case GL_UNSIGNED_SHORT: 593f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org default: 594f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org printf("unknown CHAN_TYPE %s\n", _mesa_lookup_enum_by_nr(CHAN_TYPE)); 595f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return GL_FALSE; 596f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 597f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org break; 598f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org default: 599f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org printf("unknown a[%d].format %d\n", j, a->format); 600f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return GL_FALSE; /* catch any new opcodes */ 601f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 602f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 603f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Increment j by at least 1 - may have been incremented above also: 604f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 605f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org j++; 606f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 607f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 608f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Next vertex: 609f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 610f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_lea(&p->func, vertexEAX, x86_make_disp(vertexEAX, vtx->vertex_size)); 611f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 612f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* decr count, loop if not zero 613f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 614f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_dec(&p->func, countEBP); 615f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_test(&p->func, countEBP, countEBP); 616f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_jcc(&p->func, cc_NZ, label); 617f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 618f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Exit mmx state? 619f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 620f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (p->func.need_emms) 621f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org mmx_emms(&p->func); 622f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 623f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Land forward jump here: 624f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 625f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_fixup_fwd_jump(&p->func, fixup); 626f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 627f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Pop regs and return 628f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 629f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_pop(&p->func, x86_get_base_reg(vtxESI)); 630f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_pop(&p->func, countEBP); 631f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_ret(&p->func); 632f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 633f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org assert(!vtx->emit); 634f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org vtx->emit = (tnl_emit_func)x86_get_func(&p->func); 635f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 636f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org assert( (char *) p->func.csr - (char *) p->func.store <= MAX_SSE_CODE_SIZE ); 637f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return GL_TRUE; 638f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 639f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 640f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 641f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 642f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgvoid _tnl_generate_sse_emit( struct gl_context *ctx ) 643f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 644f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct tnl_clipspace *vtx = GET_VERTEX_STATE(ctx); 645f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org struct x86_program p; 646f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 647f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (!cpu_has_xmm) { 648f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org vtx->codegen_emit = NULL; 649f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return; 650f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 651f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 652f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org memset(&p, 0, sizeof(p)); 653f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 654f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org p.ctx = ctx; 655f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org p.inputs_safe = 0; /* for now */ 656f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org p.outputs_safe = 0; /* for now */ 657f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org p.have_sse2 = cpu_has_xmm2; 658f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org p.identity = x86_make_reg(file_XMM, 6); 659f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org p.chan0 = x86_make_reg(file_XMM, 7); 660f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 661f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (!x86_init_func_size(&p.func, MAX_SSE_CODE_SIZE)) { 662f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org vtx->emit = NULL; 663f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org return; 664f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 665f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 666f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org if (build_vertex_emit(&p)) { 667f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org _tnl_register_fastpath( vtx, GL_TRUE ); 668f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 669f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org else { 670f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Note the failure so that we don't keep trying to codegen an 671f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org * impossible state: 672f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org */ 673f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org _tnl_register_fastpath( vtx, GL_FALSE ); 674f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org x86_release_func(&p.func); 675f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org } 676f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 677f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 678f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#else 679f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 680f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.orgvoid _tnl_generate_sse_emit( struct gl_context *ctx ) 681f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org{ 682f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org /* Dummy version for when USE_SSE_ASM not defined */ 683f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org} 684f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org 685f2ba7591b1407a7ee9209f842c50696914dc2dedkbr@chromium.org#endif 686