Add texture batching with TDD coverage at enqueue and prepare seams.

Opt-in batch_group regrouping cuts texture binds; prepare_draw_batches and draw_list tests lock the contract so end_frame cannot silently drop regroup.
This commit is contained in:
2026-08-14 22:15:08 -07:00
parent 0d4cdbd8ca
commit 25dd3aa920
3 changed files with 414 additions and 17 deletions
+78 -12
View File
@@ -16,8 +16,15 @@ SPRITE_VERTS_SIZE :: SPRITE_VERT_COUNT * size_of(Vertex)
VERTEX_BUFFER_SIZE :: MAX_SPRITES * SPRITE_VERTS_SIZE VERTEX_BUFFER_SIZE :: MAX_SPRITES * SPRITE_VERTS_SIZE
Queued_Sprite :: struct { Queued_Sprite :: struct {
texture: ^sdl.GPUTexture,
verts: [SPRITE_VERT_COUNT]Vertex,
batch_group: u32, // 0 = strict order; nonzero = caller permits regrouping
}
Draw_Batch :: struct {
start: int,
count: int,
texture: ^sdl.GPUTexture, texture: ^sdl.GPUTexture,
verts: [SPRITE_VERT_COUNT]Vertex,
} }
App :: struct { App :: struct {
@@ -65,7 +72,7 @@ init :: proc(app: ^App, title: cstring, width, height: i32) -> bool {
} }
requested: sdl.GPUShaderFormat = {.SPIRV, .DXIL, .MSL} requested: sdl.GPUShaderFormat = {.SPIRV, .DXIL, .MSL}
app.device = sdl.CreateGPUDevice(requested, true, nil) app.device = sdl.CreateGPUDevice(requested, false, nil)
if app.device == nil { if app.device == nil {
fmt.eprintfln("CreateGPUDevice failed: %s", sdl.GetError()) fmt.eprintfln("CreateGPUDevice failed: %s", sdl.GetError())
return false return false
@@ -76,6 +83,10 @@ init :: proc(app: ^App, title: cstring, width, height: i32) -> bool {
return false return false
} }
if !sdl.SetGPUAllowedFramesInFlight(app.device, 3) {
fmt.eprintfln("SetGPUAllowedFramesInFlight failed: %s", sdl.GetError())
}
// Prefer uncapped present for profiling; fall back if unsupported. // Prefer uncapped present for profiling; fall back if unsupported.
present := sdl.GPUPresentMode.VSYNC present := sdl.GPUPresentMode.VSYNC
if sdl.WindowSupportsGPUPresentMode(app.device, app.window, .IMMEDIATE) { if sdl.WindowSupportsGPUPresentMode(app.device, app.window, .IMMEDIATE) {
@@ -242,7 +253,11 @@ end_frame :: proc(app: ^App) {
} }
n := len(app.draw_list) n := len(app.draw_list)
batches: [dynamic]Draw_Batch
defer delete(batches)
if n > 0 { if n > 0 {
prepare_draw_batches(app.draw_list[:], &batches)
map_ptr := sdl.MapGPUTransferBuffer(app.device, app.transfer_buffer, false) map_ptr := sdl.MapGPUTransferBuffer(app.device, app.transfer_buffer, false)
if map_ptr == nil { if map_ptr == nil {
fmt.eprintfln("MapGPUTransferBuffer failed: %s", sdl.GetError()) fmt.eprintfln("MapGPUTransferBuffer failed: %s", sdl.GetError())
@@ -286,26 +301,20 @@ end_frame :: proc(app: ^App) {
app.render_pass = sdl.BeginGPURenderPass(cmd, &color_info, 1, nil) app.render_pass = sdl.BeginGPURenderPass(cmd, &color_info, 1, nil)
sdl.BindGPUGraphicsPipeline(app.render_pass, app.pipeline) sdl.BindGPUGraphicsPipeline(app.render_pass, app.pipeline)
i := 0 for batch in batches {
for i < n {
run := texture_run_len(app.draw_list[:], i)
q0 := app.draw_list[i]
sampler_binding := sdl.GPUTextureSamplerBinding { sampler_binding := sdl.GPUTextureSamplerBinding {
texture = q0.texture, texture = batch.texture,
sampler = app.sampler, sampler = app.sampler,
} }
sdl.BindGPUFragmentSamplers(app.render_pass, 0, &sampler_binding, 1) sdl.BindGPUFragmentSamplers(app.render_pass, 0, &sampler_binding, 1)
vb_binding := sdl.GPUBufferBinding { vb_binding := sdl.GPUBufferBinding {
buffer = app.vertex_buffer, buffer = app.vertex_buffer,
offset = u32(i * SPRITE_VERTS_SIZE), offset = u32(batch.start * SPRITE_VERTS_SIZE),
} }
sdl.BindGPUVertexBuffers(app.render_pass, 0, &vb_binding, 1) sdl.BindGPUVertexBuffers(app.render_pass, 0, &vb_binding, 1)
sdl.DrawGPUPrimitives(app.render_pass, u32(run * SPRITE_VERT_COUNT), 1, 0, 0) sdl.DrawGPUPrimitives(app.render_pass, u32(batch.count * SPRITE_VERT_COUNT), 1, 0, 0)
i += run
} }
sdl.EndGPURenderPass(app.render_pass) sdl.EndGPURenderPass(app.render_pass)
app.render_pass = nil app.render_pass = nil
@@ -454,3 +463,60 @@ texture_run_len :: proc(list: []Queued_Sprite, start: int) -> int {
return n return n
} }
texture_run_count :: proc(list: []Queued_Sprite) -> int {
if len(list) == 0 do return 0
count := 0
i := 0
for i < len(list) {
run := texture_run_len(list, i)
count += 1
i += run
}
return count
}
prepare_draw_batches :: proc(list: []Queued_Sprite, batches: ^[dynamic]Draw_Batch) {
clear(batches)
group_texture_runs(list)
i := 0
for i < len(list) {
run := texture_run_len(list, i)
append(batches, Draw_Batch{start = i, count = run, texture = list[i].texture})
i += run
}
}
// Within each contiguous nonzero batch_group, stably sort by texture so
// consecutive same-texture sprites become one draw. Group 0 and group
// boundaries are never crossed.
group_texture_runs :: proc(list: []Queued_Sprite) {
start := 0
for start < len(list) {
group := list[start].batch_group
if group == 0 {
start += 1
continue
}
end := start + 1
for end < len(list) && list[end].batch_group == group {
end += 1
}
// Stable insertion sort is sufficient while MAX_SPRITES is 128.
for i in start + 1 ..< end {
item := list[i]
j := i
for j > start {
if uintptr(list[j - 1].texture) <= uintptr(item.texture) {
break
}
list[j] = list[j - 1]
j -= 1
}
list[j] = item
}
start = end
}
}
+302
View File
@@ -7,6 +7,18 @@ fake_tex :: proc(id: uintptr) -> ^sdl.GPUTexture {
return cast(^sdl.GPUTexture)id return cast(^sdl.GPUTexture)id
} }
queued :: proc(tex_id: uintptr, group: u32, marker: f32) -> Queued_Sprite {
q: Queued_Sprite
q.texture = fake_tex(tex_id)
q.batch_group = group
q.verts[0].pos = {marker, 0}
return q
}
marker_of :: proc(q: Queued_Sprite) -> f32 {
return q.verts[0].pos.x
}
@(test) @(test)
texture_run_len_empty_or_oob :: proc(t: ^testing.T) { texture_run_len_empty_or_oob :: proc(t: ^testing.T) {
testing.expect_value(t, texture_run_len(nil, 0), 0) testing.expect_value(t, texture_run_len(nil, 0), 0)
@@ -62,3 +74,293 @@ texture_run_len_all_different :: proc(t: ^testing.T) {
testing.expect_value(t, texture_run_len(list, 1), 1) testing.expect_value(t, texture_run_len(list, 1), 1)
testing.expect_value(t, texture_run_len(list, 2), 1) testing.expect_value(t, texture_run_len(list, 2), 1)
} }
@(test)
group_texture_runs_strict_order_unchanged :: proc(t: ^testing.T) {
// Group 0 alternates textures; sorting must not reorder (alpha order).
list := []Queued_Sprite {
queued(2, 0, 1),
queued(1, 0, 2),
queued(2, 0, 3),
queued(1, 0, 4),
}
before_runs := texture_run_count(list)
group_texture_runs(list)
testing.expect_value(t, texture_run_count(list), before_runs)
testing.expect_value(t, marker_of(list[0]), f32(1))
testing.expect_value(t, marker_of(list[1]), f32(2))
testing.expect_value(t, marker_of(list[2]), f32(3))
testing.expect_value(t, marker_of(list[3]), f32(4))
}
@(test)
group_texture_runs_reduces_runs_inside_group :: proc(t: ^testing.T) {
list := []Queued_Sprite {
queued(2, 1, 1),
queued(1, 1, 2),
queued(2, 1, 3),
queued(1, 1, 4),
}
testing.expect_value(t, texture_run_count(list), 4)
group_texture_runs(list)
testing.expect_value(t, texture_run_count(list), 2)
testing.expect(t, list[0].texture == fake_tex(1), "lower texture pointer first")
testing.expect(t, list[1].texture == fake_tex(1), "same texture run")
testing.expect(t, list[2].texture == fake_tex(2), "second texture run")
testing.expect(t, list[3].texture == fake_tex(2), "second texture run cont")
}
@(test)
group_texture_runs_stable_same_texture :: proc(t: ^testing.T) {
list := []Queued_Sprite {
queued(2, 1, 10),
queued(1, 1, 20),
queued(2, 1, 30),
queued(1, 1, 40),
}
group_texture_runs(list)
// Same-texture relative order preserved (stable sort).
testing.expect_value(t, marker_of(list[0]), f32(20))
testing.expect_value(t, marker_of(list[1]), f32(40))
testing.expect_value(t, marker_of(list[2]), f32(10))
testing.expect_value(t, marker_of(list[3]), f32(30))
}
@(test)
group_texture_runs_respects_group_boundaries :: proc(t: ^testing.T) {
list := []Queued_Sprite {
queued(2, 1, 1),
queued(1, 1, 2),
queued(2, 0, 3), // strict barrier
queued(1, 2, 4),
queued(2, 2, 5),
}
group_texture_runs(list)
testing.expect_value(t, marker_of(list[2]), f32(3)) // barrier stays put
testing.expect(t, list[0].texture == fake_tex(1))
testing.expect(t, list[1].texture == fake_tex(2))
testing.expect(t, list[3].texture == fake_tex(1))
testing.expect(t, list[4].texture == fake_tex(2))
testing.expect_value(t, list[0].batch_group, u32(1))
testing.expect_value(t, list[1].batch_group, u32(1))
testing.expect_value(t, list[2].batch_group, u32(0))
testing.expect_value(t, list[3].batch_group, u32(2))
testing.expect_value(t, list[4].batch_group, u32(2))
}
@(test)
group_texture_runs_does_not_merge_across_different_groups :: proc(t: ^testing.T) {
// Adjacent nonzero groups with different ids must not merge runs across.
list := []Queued_Sprite {
queued(1, 1, 1),
queued(2, 1, 2),
queued(1, 2, 3),
queued(2, 2, 4),
}
group_texture_runs(list)
testing.expect_value(t, texture_run_count(list), 4)
testing.expect_value(t, list[0].batch_group, u32(1))
testing.expect_value(t, list[1].batch_group, u32(1))
testing.expect_value(t, list[2].batch_group, u32(2))
testing.expect_value(t, list[3].batch_group, u32(2))
}
@(test)
group_texture_runs_empty :: proc(t: ^testing.T) {
list := []Queued_Sprite{}
testing.expect_value(t, texture_run_count(list), 0)
group_texture_runs(list)
testing.expect_value(t, texture_run_count(list), 0)
}
@(test)
group_texture_runs_already_optimal :: proc(t: ^testing.T) {
list := []Queued_Sprite {
queued(1, 1, 10),
queued(1, 1, 20),
queued(2, 1, 30),
queued(2, 1, 40),
}
testing.expect_value(t, texture_run_count(list), 2)
group_texture_runs(list)
testing.expect_value(t, texture_run_count(list), 2)
testing.expect_value(t, marker_of(list[0]), f32(10))
testing.expect_value(t, marker_of(list[1]), f32(20))
testing.expect_value(t, marker_of(list[2]), f32(30))
testing.expect_value(t, marker_of(list[3]), f32(40))
}
@(test)
group_texture_runs_noncontiguous_same_group_id :: proc(t: ^testing.T) {
// Same nonzero id split by group 0: each window regroups alone.
list := []Queued_Sprite {
queued(2, 1, 1),
queued(1, 1, 2),
queued(2, 0, 3),
queued(2, 1, 4),
queued(1, 1, 5),
}
group_texture_runs(list)
testing.expect_value(t, marker_of(list[2]), f32(3))
testing.expect(t, list[0].texture == fake_tex(1))
testing.expect(t, list[1].texture == fake_tex(2))
testing.expect_value(t, list[2].batch_group, u32(0))
testing.expect(t, list[3].texture == fake_tex(1))
testing.expect(t, list[4].texture == fake_tex(2))
testing.expect_value(t, list[0].batch_group, u32(1))
testing.expect_value(t, list[1].batch_group, u32(1))
testing.expect_value(t, list[3].batch_group, u32(1))
testing.expect_value(t, list[4].batch_group, u32(1))
}
make_test_draw_app :: proc() -> App {
app: App
app.cmd = cast(^sdl.GPUCommandBuffer)uintptr(1)
app.swapchain_texture = fake_tex(99)
app.swapchain_w = 800
app.swapchain_h = 600
app.camera = camera_default()
app.draw_list = make([dynamic]Queued_Sprite)
return app
}
destroy_test_draw_app :: proc(app: ^App) {
if app == nil do return
delete(app.draw_list)
app^ = {}
}
make_test_draw_character :: proc() -> Character_Data {
data: Character_Data
data.texture = fake_tex(42)
data.width = 100
data.height = 100
data.def.pivot = {0.5, 1.0}
data.def.clips = make(map[string]Clip_Def)
frames := make([]Frame_Def, 1)
frames[0] = Frame_Def {
rect = {0, 0, 10, 10},
source_size = {10, 10},
trim_offset = {0, 0},
}
data.def.clips["idle"] = Clip_Def {
loop = true,
fps = 10,
frames = frames,
}
return data
}
destroy_test_draw_character :: proc(data: ^Character_Data) {
if data == nil do return
keys := make([dynamic]string, context.temp_allocator)
for key, clip in data.def.clips {
delete(clip.frames)
append(&keys, key)
}
for key in keys {
delete_key(&data.def.clips, key)
}
delete(data.def.clips)
data^ = {}
}
@(test)
draw_sprite_stamps_batch_group_zero :: proc(t: ^testing.T) {
app := make_test_draw_app()
defer destroy_test_draw_app(&app)
data := make_test_draw_character()
defer destroy_test_draw_character(&data)
sprite := spawn_sprite(&data, {100, 200}, "idle", 0)
draw_sprite(&app, &sprite)
testing.expect_value(t, len(app.draw_list), 1)
testing.expect_value(t, app.draw_list[0].batch_group, u32(0))
testing.expect(t, app.draw_list[0].texture == data.texture)
}
@(test)
draw_sprite_batched_stamps_batch_group :: proc(t: ^testing.T) {
app := make_test_draw_app()
defer destroy_test_draw_app(&app)
data := make_test_draw_character()
defer destroy_test_draw_character(&data)
sprite := spawn_sprite(&data, {100, 200}, "idle", 0)
draw_sprite_batched(&app, &sprite, 7)
testing.expect_value(t, len(app.draw_list), 1)
testing.expect_value(t, app.draw_list[0].batch_group, u32(7))
testing.expect(t, app.draw_list[0].texture == data.texture)
}
@(test)
draw_sprite_batched_guards_leave_list_unchanged :: proc(t: ^testing.T) {
app := make_test_draw_app()
defer destroy_test_draw_app(&app)
data := make_test_draw_character()
defer destroy_test_draw_character(&data)
sprite := spawn_sprite(&data, {100, 200}, "idle", 0)
app.cmd = nil
draw_sprite_batched(&app, &sprite, 1)
testing.expect_value(t, len(app.draw_list), 0)
app.cmd = cast(^sdl.GPUCommandBuffer)uintptr(1)
app.swapchain_texture = nil
draw_sprite_batched(&app, &sprite, 1)
testing.expect_value(t, len(app.draw_list), 0)
app.swapchain_texture = fake_tex(99)
draw_sprite_batched(&app, nil, 1)
testing.expect_value(t, len(app.draw_list), 0)
no_data := sprite
no_data.data = nil
draw_sprite_batched(&app, &no_data, 1)
testing.expect_value(t, len(app.draw_list), 0)
no_tex := sprite
tex_data := data
tex_data.texture = nil
no_tex.data = &tex_data
draw_sprite_batched(&app, &no_tex, 1)
testing.expect_value(t, len(app.draw_list), 0)
}
@(test)
prepare_draw_batches_regroups_then_plans_runs :: proc(t: ^testing.T) {
list := []Queued_Sprite {
queued(2, 1, 1),
queued(1, 1, 2),
queued(2, 1, 3),
queued(1, 1, 4),
queued(3, 0, 5),
queued(1, 0, 6),
}
batches := make([dynamic]Draw_Batch)
defer delete(batches)
prepare_draw_batches(list, &batches)
testing.expect_value(t, len(batches), 4)
testing.expect_value(t, batches[0].start, 0)
testing.expect_value(t, batches[0].count, 2)
testing.expect(t, batches[0].texture == fake_tex(1))
testing.expect_value(t, batches[1].start, 2)
testing.expect_value(t, batches[1].count, 2)
testing.expect(t, batches[1].texture == fake_tex(2))
testing.expect_value(t, batches[2].start, 4)
testing.expect_value(t, batches[2].count, 1)
testing.expect(t, batches[2].texture == fake_tex(3))
testing.expect_value(t, batches[3].start, 5)
testing.expect_value(t, batches[3].count, 1)
testing.expect(t, batches[3].texture == fake_tex(1))
// Group 0 submission order preserved.
testing.expect_value(t, marker_of(list[4]), f32(5))
testing.expect_value(t, marker_of(list[5]), f32(6))
testing.expect_value(t, list[4].batch_group, u32(0))
testing.expect_value(t, list[5].batch_group, u32(0))
}
+34 -5
View File
@@ -82,7 +82,31 @@ to_clip :: proc(px, py, sw, sh: f32) -> [2]f32 {
} }
} }
// Axis-aligned quad: two unique x and y values, so scale once and reuse.
sprite_quad_to_clip :: proc(x0, y0, x1, y1, sw, sh: f32) -> [4]Vec2 {
sx := 2.0 / sw
sy := 2.0 / sh
left := x0 * sx - 1
right := x1 * sx - 1
top := 1 - y0 * sy
bottom := 1 - y1 * sy
return {
{left, top},
{right, top},
{right, bottom},
{left, bottom},
}
}
draw_sprite :: proc(app: ^App, sprite: ^Sprite) { draw_sprite :: proc(app: ^App, sprite: ^Sprite) {
draw_sprite_batched(app, sprite, 0)
}
// Nonzero batch_group lets end_frame regroup consecutive same-group sprites by
// texture. Group 0 keeps exact submission order for correct alpha overlap.
draw_sprite_batched :: proc(app: ^App, sprite: ^Sprite, batch_group: u32) {
if app.cmd == nil || app.swapchain_texture == nil { if app.cmd == nil || app.swapchain_texture == nil {
return return
} }
@@ -124,10 +148,8 @@ draw_sprite :: proc(app: ^App, sprite: ^Sprite) {
sw := f32(app.swapchain_w) sw := f32(app.swapchain_w)
sh := f32(app.swapchain_h) sh := f32(app.swapchain_h)
p0 := to_clip(x0_px, y0_px, sw, sh) points := sprite_quad_to_clip(x0_px, y0_px, x1_px, y1_px, sw, sh)
p1 := to_clip(x1_px, y0_px, sw, sh) p0, p1, p2, p3 := points[0], points[1], points[2], points[3]
p2 := to_clip(x1_px, y1_px, sw, sh)
p3 := to_clip(x0_px, y1_px, sw, sh)
tex_w := f32(sprite.data.width) tex_w := f32(sprite.data.width)
tex_h := f32(sprite.data.height) tex_h := f32(sprite.data.height)
@@ -142,7 +164,14 @@ draw_sprite :: proc(app: ^App, sprite: ^Sprite) {
{pos = p3, uv = {u0, v1}}, {pos = p3, uv = {u0, v1}},
} }
append(&app.draw_list, Queued_Sprite{texture = sprite.data.texture, verts = verts}) append(
&app.draw_list,
Queued_Sprite {
texture = sprite.data.texture,
verts = verts,
batch_group = batch_group,
},
)
} }
set_sprite_clip :: proc(sprite: ^Sprite, clip: string) { set_sprite_clip :: proc(sprite: ^Sprite, clip: string) {