Skip to main content
GameDev.net gamedev.net
🔒 Locked

VBO / Immediate Benchmark

Started by TheAverageUser Jun 16, 2010 at 10:06 AM 8 replies 3.4k views
Original Post
TheAverageUser
TheAverageUser
Hey there,

I repeatedly here people complaining about how sloooow immediate mode is but, using it myself, I always thought "Can't be too bad for 2D, worked good so far". Due to some recent performance issues (The first in ~three years) I wrote a small benchmark to test excactly how slow immediate mode is.

I didn't believe it: Even in the worst case I could think of using VBOs, they are still 4 times faster than immediate mode?! Please, anyone, take a look at my code and tell me if it looks good. I never used VBOs before and don't want to rework my projects whole gfx engine and see that I did something terribly wrong during profiling.

The benchmark app is written in C# using the Tao framework. I just used a fixed vertex format assuming per-vertex 2d coords, three sets of 2d texcoords and a 4-float-alpha value.

Test cases are:

Immediate mode: ~ 10 ms
Immediate mode "Singe Surface" (single glBegin / glEnd): ~ 9,5 ms
Static VBO: ~1,2 ms
Streamed VBO used like immediate (store 1 quad, draw 1 quad, n times each frame): ~1,2 ms
Streamed VBO (Store once / Draw once each frame): ~0,8 ms
Streamed VBO double-buffered (Store to A / Draw B / switch both each frame): ~0,8 ms

using System;using System.Collections.Generic;using System.Text;using System.Runtime.InteropServices;using Tao.Glfw;using Tao.OpenGl;namespace OpenGL_Test{	public class Program	{		public	const	int	NumPrimitives	= 1000;		public	const	int	Iterations		= 120;		public static void Main(string[] args)		{			Glfw.glfwInit();			Glfw.glfwOpenWindow(640, 480, 8, 8, 8, 8, 32, 0, Glfw.GLFW_WINDOW);			Glfw.glfwSwapInterval(0); // Vsync off			Gl.glOrtho(0, 640, 480, 0, 0, 1);			DoTest(new Immedate());			DoTest(new ImmedateSS());			DoTest(new VBOStatic());			DoTest(new VBOStreamCh());			DoTest(new VBOStream());			DoTest(new VBOStreamDB());			Console.ReadLine();			Glfw.glfwTerminate();		}		public static void DoTest(ITest test)		{			test.Init(NumPrimitives);			double frameTimeSum = 0.0d;			int frameNum = 0;			double currentTime;			double frameTime;			for (int i = 0; i < Iterations; i++)			{				currentTime = Glfw.glfwGetTime();				Gl.glClear(Gl.GL_COLOR | Gl.GL_DEPTH);				test.Draw(NumPrimitives);				Gl.glFinish();				Glfw.glfwSwapBuffers();				Glfw.glfwPollEvents();				frameTime = Glfw.glfwGetTime() - currentTime;				frameTimeSum += frameTime;				frameNum++;			}			Console.WriteLine(				"{0}:\t{1:F} us", 				test.GetType().Name, 				1000000.0d * frameTimeSum / frameNum);		}	}	public interface ITest	{		void Draw(int n);		void Init(int n);	}	[StructLayout(LayoutKind.Sequential)]	public struct Vertex	{		public static readonly IntPtr	VertexOffsetPtr	= (IntPtr)(sizeof(float) * 0);		public static readonly IntPtr	Tex0OffsetPtr	= (IntPtr)(sizeof(float) * 2);		public static readonly IntPtr	Tex1OffsetPtr	= (IntPtr)(sizeof(float) * 4);		public static readonly IntPtr	Tex2OffsetPtr	= (IntPtr)(sizeof(float) * 6);		public static readonly IntPtr	ColorOffsetPtr	= (IntPtr)(sizeof(float) * 8);		public static readonly int		Size			= (sizeof(float) * 16);		public float x, y;		public float s0, t0;		public float s1, t1;		public float s2, t2;		public float r, g, b, a;		public float pad0, pad1, pad2, pad3;		public Vertex(float x, float y, float r, float g, float b, float a, float s0, float t0, float s1, float t1, float s2, float t2)		{			this.x = x;			this.y = y;			this.s0 = s0;			this.t0 = t0;			this.s1 = s1;			this.t1 = t1;			this.s2 = s2;			this.t2 = t2;			this.r = r;			this.g = g;			this.b = b;			this.a = a;			this.pad0 = 0;			this.pad1 = 0;			this.pad2 = 0;			this.pad3 = 0;		}		public Vertex(float x, float y, float r, float g, float b, float a, float s0, float t0, float s1, float t1) : this(x, y, r, g, b, a, s0, t0, s1, t1, 0.0f, 0.0f) {}		public Vertex(float x, float y, float r, float g, float b, float a, float s0, float t0) : this(x, y, r, g, b, a, s0, t0, 0.0f, 0.0f, 0.0f, 0.0f) {}		public Vertex(float x, float y, float r, float g, float b, float a) : this(x, y, r, g, b, a, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f) {}		public Vertex(float x, float y) : this(x, y, 1.0f, 1.0f, 1.0f, 1.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f) {}	}	public class Immedate : ITest	{		public void Draw(int n)		{			for (int i = 0; i < n; i++)			{				Gl.glBegin(Gl.GL_QUADS);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(100.0f, 100.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(200.0f, 100.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(200.0f, 200.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(100.0f, 200.0f);				Gl.glEnd();			}		}		public void Init(int n)		{		}	}	public class ImmedateSS : ITest	{		public void Draw(int n)		{			Gl.glBegin(Gl.GL_QUADS);			for (int i = 0; i < n; i++)			{				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(100.0f, 100.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(200.0f, 100.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(200.0f, 200.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE0, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE1, 0.0f, 0.0f);				Gl.glMultiTexCoord2f(Gl.GL_TEXTURE2, 0.0f, 0.0f);				Gl.glColor4f(1.0f, 1.0f, 1.0f, 1.0f);				Gl.glVertex2f(100.0f, 200.0f);			}			Gl.glEnd();		}		public void Init(int n)		{		}	}	public class VBOStatic : ITest	{		protected	int	glIdVBO	= 0;		public void Draw(int n)		{			Gl.glBindBufferARB(Gl.GL_ARRAY_BUFFER, glIdVBO);			Gl.glEnableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glVertexPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.VertexOffsetPtr);			Gl.glEnableClientState(Gl.GL_COLOR_ARRAY);			Gl.glColorPointer(4, Gl.GL_FLOAT, Vertex.Size, Vertex.ColorOffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex0OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex1OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex2OffsetPtr);			Gl.glDrawArrays(Gl.GL_QUADS, 0, 4 * n);			Gl.glDisableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glDisableClientState(Gl.GL_COLOR_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glBindBufferARB(Gl.GL_ARRAY_BUFFER, 0);		}		public void Init(int n)		{			Random rnd = new Random();			Vertex[] vertices = new Vertex[4 * n];			for (int i = 0; i < n; i++)			{				vertices[i * 4 + 0] = new Vertex(100.0f, 100.0f);				vertices[i * 4 + 1] = new Vertex(200.0f, 100.0f);				vertices[i * 4 + 2] = new Vertex(200.0f, 200.0f);				vertices[i * 4 + 3] = new Vertex(100.0f, 200.0f);			}			Gl.glGenBuffers(1, out glIdVBO);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO);			Gl.glBufferData(Gl.GL_ARRAY_BUFFER, (IntPtr)(vertices.Length * Vertex.Size), vertices, Gl.GL_STATIC_DRAW);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, 0);		}	}	public class VBOStream : ITest	{		protected	int	glIdVBO	= 0;		public void Draw(int n)		{			Vertex[] vertices = new Vertex[4 * n];			for (int i = 0; i < n; i++)			{				vertices[i * 4 + 0] = new Vertex(100.0f, 100.0f);				vertices[i * 4 + 1] = new Vertex(200.0f, 100.0f);				vertices[i * 4 + 2] = new Vertex(200.0f, 200.0f);				vertices[i * 4 + 3] = new Vertex(100.0f, 200.0f);			}			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO);			Gl.glBufferData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), null, Gl.GL_STREAM_DRAW);			Gl.glBufferSubData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), (IntPtr)(vertices.Length * Vertex.Size), vertices);			Gl.glEnableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glVertexPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.VertexOffsetPtr);			Gl.glEnableClientState(Gl.GL_COLOR_ARRAY);			Gl.glColorPointer(4, Gl.GL_FLOAT, Vertex.Size, Vertex.ColorOffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex0OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex1OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex2OffsetPtr);			Gl.glDrawArrays(Gl.GL_QUADS, 0, 4 * n);			Gl.glDisableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glDisableClientState(Gl.GL_COLOR_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glBindBufferARB(Gl.GL_ARRAY_BUFFER, 0);		}		public void Init(int n)		{			Gl.glGenBuffers(1, out glIdVBO);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO);			Gl.glBufferData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), null, Gl.GL_STREAM_DRAW);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, 0);		}	}	public class VBOStreamDB : ITest	{		protected	int	glIdVBO1	= 0;		protected	int	glIdVBO2	= 0;		public void Draw(int n)		{			Vertex[] vertices = new Vertex[4 * n];			for (int i = 0; i < n; i++)			{				vertices[i * 4 + 0] = new Vertex(100.0f, 100.0f);				vertices[i * 4 + 1] = new Vertex(200.0f, 100.0f);				vertices[i * 4 + 2] = new Vertex(200.0f, 200.0f);				vertices[i * 4 + 3] = new Vertex(100.0f, 200.0f);			}			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO1);			Gl.glBufferData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), null, Gl.GL_STREAM_DRAW);			Gl.glBufferSubData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), (IntPtr)(vertices.Length * Vertex.Size), vertices);						Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO2);			Gl.glEnableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glVertexPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.VertexOffsetPtr);			Gl.glEnableClientState(Gl.GL_COLOR_ARRAY);			Gl.glColorPointer(4, Gl.GL_FLOAT, Vertex.Size, Vertex.ColorOffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex0OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex1OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex2OffsetPtr);			Gl.glDrawArrays(Gl.GL_QUADS, 0, 4 * n);			Gl.glDisableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glDisableClientState(Gl.GL_COLOR_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, 0);			int swap = glIdVBO1;			glIdVBO1 = glIdVBO2;			glIdVBO2 = swap;		}		public void Init(int n)		{			Gl.glGenBuffers(1, out glIdVBO1);			Gl.glGenBuffers(1, out glIdVBO2);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO1);			Gl.glBufferData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), null, Gl.GL_STREAM_DRAW);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO2);			Gl.glBufferData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), null, Gl.GL_STREAM_DRAW);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, 0);		}	}	public class VBOStreamCh : ITest	{		protected	int	glIdVBO	= 0;		public void Draw(int n)		{			Vertex[] vertices = new Vertex[4];			vertices[0] = new Vertex(100.0f, 100.0f);			vertices[1] = new Vertex(200.0f, 100.0f);			vertices[2] = new Vertex(200.0f, 200.0f);			vertices[3] = new Vertex(100.0f, 200.0f);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO);			Gl.glEnableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glVertexPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.VertexOffsetPtr);			Gl.glEnableClientState(Gl.GL_COLOR_ARRAY);			Gl.glColorPointer(4, Gl.GL_FLOAT, Vertex.Size, Vertex.ColorOffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex0OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex1OffsetPtr);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glEnableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glTexCoordPointer(2, Gl.GL_FLOAT, Vertex.Size, Vertex.Tex2OffsetPtr);			for (int i = 0; i < n; i++)			{				Gl.glBufferSubData(Gl.GL_ARRAY_BUFFER, (IntPtr)(i * 4 * Vertex.Size), (IntPtr)(4 * Vertex.Size), vertices);				Gl.glDrawArrays(Gl.GL_QUADS, 4 * i, 4);			}			Gl.glDisableClientState(Gl.GL_VERTEX_ARRAY);			Gl.glDisableClientState(Gl.GL_COLOR_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE0);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE1);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glClientActiveTextureARB(Gl.GL_TEXTURE2);			Gl.glDisableClientState(Gl.GL_TEXTURE_COORD_ARRAY);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, 0);		}		public void Init(int n)		{			Gl.glGenBuffers(1, out glIdVBO);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, glIdVBO);			Gl.glBufferData(Gl.GL_ARRAY_BUFFER, (IntPtr)(0), null, Gl.GL_STREAM_DRAW);			Gl.glBindBuffer(Gl.GL_ARRAY_BUFFER, 0);		}	}}
outRider
outRider
Immediate mode in and of itself is not too bad for dynamic vertices. On most hardware you can just inline the data into the command buffer and if it's never used again it's comparable to and sometimes even faster than buffers. On Nvidia it definitely does very well, you can shovel vertices, indices, pixels and a bunch of other stuff into the command buffer.

But all that doesn't matter because that's the hardware's POV of immediate mode, not your POV. Most of your overhead is likely from function calls, object contruction, etc. I imagine the C#/Tao versions of GL immediate mode are even more expensive than the native versions so it would probably be very difficult to see any benefits of immediate mode in this environment. You're probably seeing object allocation cost, C# dispatch cost and a bunch of other unrelated things rather than an accurate measurement of immediates vs buffers.
21st Century Moose
21st Century Moose
To get a more valid benchmark you really need to be submitting more data. Much more data. Start with maybe 100,000 polys per frame and ramp it up to 1,000,000.

NVIDIA hardware is extremely good at optimising immediate mode. I suspect it might be doing some serious batching in the driver before submitting. At the same time, you can optimise immediate mode quite well yourself by e.g. not doing anything silly like having a glBegin (GL_TRIANGLES);/glEnd (); pair inside a loop. You can get pretty damn close to VBO performance (especially if you need to refresh the data every frame) using some sensible techniques.

Where VBOs really shine is if your app is bandwidth-bound. By storing your verts directly in video RAM you totally avoid the need to resubmit them every frame, thus freeing up bandwidth for other tasks such as dynamic texture updating. Of course you'll likely end up being either shader or fillrate (or even CPU) bound as a result of that, not to mention that you'll have less video RAM available for textures (not as big a problem as it used to be).

Where results can be misleading is if the degree to which you're bandwidth bound is not sufficiently higher than the degree to which you're bound by other factors. You might re-engineer an app and only see a 0.5% improvement, or you might run twice as fast (or more). It depends.

You also won't get much improvement if you need to do many state changes between small batches of polys. If for example you have 20,000 polys and they use 50 textures you can batch things up very well; if they use 5000 textures it's not so good.

Theoretical worst case is that you're not bandwidth bound to begin with, you submit a single triangle per draw call owing to everything needing different state, and you lose a huge chunk of video RAM into the bargain.

In other words you need to know your app, what it does, and how much of an improvement you're likely to get before making the decision to switch over.
Direct3D has need of instancing, but we do not. We have plenty of glVertexAttrib calls. 
TheAverageUser
TheAverageUser
Yeah, I just discovered that myself.. spent 6 hours in bringing VBOs properly to my project code, supporting particle systems using them... and it actually ran slower than immediate mode (on Ati)... what a frustration. Rolled all back now. Damn you, VBOs!

I'm doing a 2d game without any atlas textures and manual z sorting due to alpha, so there's nearly no opportunity of batching at all. Should've known better before.

Still, in some scenes, the FPS (normally at 220 ~) drop down to 50. There are huge amounts of objects in the scene, particles, shaders active for dynamic lighting.. I'm figuring and figuring what might be the cause and recently thought, it was vertex bandwith - which doesn't seem to be the case, judging by what you say. I also tried deactivating any shader usage and setting the resolution to less than half, but the performance impact (which is clearly on the GPU side) remains. What is left as possible cause if I wiped out vertex bandwith, pixel fillrate and shaders?
21st Century Moose
21st Century Moose
Dammit, I could have warned you about particles. They're about the only thing left in immediate mode in my D3D project, and for a very good reason, so the behaviour you've noticed seems to be neither anamolous nor API-specific (although I intend revisiting them at some point).

What's left to check? Either you're CPU-bound or you're doing something that's stalling the pipeline I reckon. Most likely the latter given how low you're dropping. Any readbacks from the GPU in there? They're one of the most likely causes.
Direct3D has need of instancing, but we do not. We have plenty of glVertexAttrib calls. 
TheAverageUser
TheAverageUser
Nope, no readbacks. Due to the manual sorting issue I have because of 2D graphics where nearly every sprite has alpha, I have to do many state changes including switching shaders, looking up uniform adresses and setting their values, etc.
There are approximately 400 textured sprites rendered, sized from 10 pixels to 500 pixels diameter. As I immediate-batch sprites in constant z layers, I get around with approximately 250 to 300 "batches".

What comes into my mind right now: Those state changes I have to do between most draw calls most likely include texture changes, sometimes even multitexture-changes (change three bound textures at once). Is that likely to be a performance issue in 2d games or in general?
21st Century Moose
21st Century Moose
Texture changes are quite fast these days; the main slowdown from them comes from breaking your batching. Maybe look at implementing a texture atlas? This will reduce the number of changes and allow you to use bigger batches.

Any dynamic texture updates in there? Anything else that might break CPU/GPU parallelism? 400 textured sprites is not a lot, but if you have a lot of overdraw from that it's going to hurt fillrate like crazy, especially if they overlap and especially if your driver isn't doing (or can't do, owing to alpha blending) early z-rejection.
Direct3D has need of instancing, but we do not. We have plenty of glVertexAttrib calls. 
TheAverageUser
TheAverageUser
All my sprites have alpha, so doing any z-rejection is completely pointless. As logic result, blending is active non-stop. I also tried additionally activating alpha testing to early-out zero-alpha pixels but that didn't give me ANY difference. o_O

Atlas textures.. hmm.. might be an idea. Though, I do not know if this would be worth the effort of re-structuring a lot of graphics code.

However, when I reduce screen resolution to less than half, it doesn't really run faster, so I'm not sure if its really fill rate limited?
21st Century Moose
21st Century Moose
Unless it's still fillrate limited even at half resolution. Up to 400 passes over a potentially large area of the framebuffer is really quite a lot.
Direct3D has need of instancing, but we do not. We have plenty of glVertexAttrib calls. 
TheAverageUser
TheAverageUser
I tested it with a 200x150 resolution (windowed) - the games native resolution is 1024x768. While gaining around 30 to 50 fps in general, the performance drop in a particular scene with many objects was still there unchanged, so I guess pixel fillrate is really out of scope for optimizing.

Is there any tool I can use to determine the bottleneck?


Edit:
Just tried the demo of glDeBugger and discovered I'd forgotten to optimize shader uniform variable lookup and had a lot of glGet-Calls per frame. After fixing this by caching the locations and not getting them anew every call, I got a nice 10 to 20 FPS boost in that particular scene; the FPS now only drop to 70 - 80 instead of 50 - 60. Feels a lot better, but I still think there's a lot to optimize.

Unfortunately, glDebugger isn't very helpful in its trial version to determine bottlenecks because its "best" functions are deactivated. Are there other tools I could use?

[Edited by - Fetze on June 17, 2010 11:48:49 AM]

Topic Locked

This topic has been locked by a moderator. New replies are not allowed.

Sign in to reply to this topic.