intel: Use specified alignment for writes into the upload buffer

[mesa.git] / src / mesa / drivers / dri / i965 / brw_clip_util.c
diff --git a/src/mesa/drivers/dri/i965/brw_clip_util.c b/src/mesa/drivers/dri/i965/brw_clip_util.c

index e09efc07ed47521da7b5c57d2ea06a5b9b4048d4..d2ac1235e46fa6a3f88288a699093013b54480b5 100644 (file)
--- a/src/mesa/drivers/dri/i965/brw_clip_util.c
+++ b/src/mesa/drivers/dri/i965/brw_clip_util.c
@@ -33,14 +33,13 @@
  #include "main/glheader.h"
  #include "main/macros.h"
  #include "main/enums.h"
-#include "shader/program.h"
+#include "program/program.h"
  
  #include "intel_batchbuffer.h"
  
  #include "brw_defines.h"
  #include "brw_context.h"
  #include "brw_eu.h"
-#include "brw_util.h"
  #include "brw_clip.h"
  
  
@@ -142,19 +141,16 @@ void brw_clip_interp_vertex( struct brw_clip_compile *c,
      */
     /*
      * After CLIP stage, only first 256 bits of the VUE are read
-    * back on IGDNG, so needn't change it
+    * back on Ironlake, so needn't change it
      */
     brw_copy_indirect_to_indirect(p, dest_ptr, v0_ptr, 1);
        
     /* Iterate over each attribute (could be done in pairs?)
      */
     for (i = 0; i < c->nr_attrs; i++) {
-      GLuint delta = i*16 + 32;
+      GLuint delta = c->offset[c->idx_to_attr[i]];
  
-      if (BRW_IS_IGDNG(p->brw))
-          delta = i * 16 + 32 * 3;
-
-      if (delta == c->offset[VERT_RESULT_EDGE]) {
+      if (c->idx_to_attr[i] == VERT_RESULT_EDGE) {
          if (force_edgeflag) 
             brw_MOV(p, deref_4f(dest_ptr, delta), brw_imm_f(1));
          else
@@ -183,10 +179,7 @@ void brw_clip_interp_vertex( struct brw_clip_compile *c,
     }
  
     if (i & 1) {
-      GLuint delta = i*16 + 32;
-
-      if (BRW_IS_IGDNG(p->brw))
-          delta = i * 16 + 32 * 3;
+      GLuint delta = c->offset[c->idx_to_attr[c->nr_attrs - 1]] + ATTR_SIZE;
  
        brw_MOV(p, deref_4f(dest_ptr, delta), brw_imm_f(0));
     }
@@ -199,11 +192,6 @@ void brw_clip_interp_vertex( struct brw_clip_compile *c,
     brw_clip_project_vertex(c, dest_ptr );
  }
  
-
-
-
-#define MAX_MRF 16
-
  void brw_clip_emit_vue(struct brw_clip_compile *c, 
                        struct brw_indirect vert,
                        GLboolean allocate,
@@ -211,25 +199,14 @@ void brw_clip_emit_vue(struct brw_clip_compile *c,
                        GLuint header)
  {
     struct brw_compile *p = &c->func;
-   GLuint start = c->last_mrf;
+
+   brw_clip_ff_sync(c);
  
     assert(!(allocate && eot));
-   
-   /* Cycle through mrf regs - probably futile as we have to wait for
-    * the allocation response anyway.  Also, the order this function
-    * is invoked doesn't correspond to the order the instructions will
-    * be executed, so it won't have any effect in many cases.
-    */
-#if 0
-   if (start + c->nr_regs + 1 >= MAX_MRF)
-      start = 0;
  
-   c->last_mrf = start + c->nr_regs + 1;
-#endif
-       
     /* Copy the vertex from vertn into m1..mN+1:
      */
-   brw_copy_from_indirect(p, brw_message_reg(start+1), vert, c->nr_regs);
+   brw_copy_from_indirect(p, brw_message_reg(1), vert, c->nr_regs);
  
     /* Overwrite PrimType and PrimStart in the message header, for
      * each vertex in turn:
@@ -245,7 +222,7 @@ void brw_clip_emit_vue(struct brw_clip_compile *c,
      */
     brw_urb_WRITE(p, 
                  allocate ? c->reg.R0 : retype(brw_null_reg(), BRW_REGISTER_TYPE_UD),
-                start,
+                0,
                  c->reg.R0,
                  allocate,
                  1,             /* used */
@@ -263,6 +240,7 @@ void brw_clip_kill_thread(struct brw_clip_compile *c)
  {
     struct brw_compile *p = &c->func;
  
+   brw_clip_ff_sync(c);
     /* Send an empty message to kill the thread and release any
      * allocated urb entry:
      */
@@ -356,17 +334,37 @@ void brw_clip_init_clipmask( struct brw_clip_compile *c )
  
  void brw_clip_ff_sync(struct brw_clip_compile *c)
  {
+    struct intel_context *intel = &c->func.brw->intel;
+
+    if (intel->needs_ff_sync) {
+        struct brw_compile *p = &c->func;
+        struct brw_instruction *need_ff_sync;
+
+        brw_set_conditionalmod(p, BRW_CONDITIONAL_Z);
+        brw_AND(p, brw_null_reg(), c->reg.ff_sync, brw_imm_ud(0x1));
+        need_ff_sync = brw_IF(p, BRW_EXECUTE_1);
+        {
+            brw_OR(p, c->reg.ff_sync, c->reg.ff_sync, brw_imm_ud(0x1));
+            brw_ff_sync(p,
+                       c->reg.R0,
+                       0,
+                       c->reg.R0,
+                       1, /* allocate */
+                       1, /* response length */
+                       0 /* eot */);
+        }
+        brw_ENDIF(p, need_ff_sync);
+        brw_set_predicate_control(p, BRW_PREDICATE_NONE);
+    }
+}
+
+void brw_clip_init_ff_sync(struct brw_clip_compile *c)
+{
+    struct intel_context *intel = &c->func.brw->intel;
+
+    if (intel->needs_ff_sync) {
         struct brw_compile *p = &c->func;
-       brw_ff_sync(p, 
-                               c->reg.R0,
-                               0,
-                               c->reg.R0,
-                               1,      
-                               1,              /* used */
-                               1,      /* msg length */
-                               1,              /* response length */
-                               0,              /* eot */
-                               1,              /* write compelete */
-                               0,              /* urb offset */
-                               BRW_URB_SWIZZLE_NONE);
+        
+        brw_MOV(p, c->reg.ff_sync, brw_imm_ud(0));
+    }
  }