diff --git a/engine/app.odin b/engine/app.odin index 7bc961a..412e67b 100644 --- a/engine/app.odin +++ b/engine/app.odin @@ -16,8 +16,15 @@ SPRITE_VERTS_SIZE :: SPRITE_VERT_COUNT * size_of(Vertex) VERTEX_BUFFER_SIZE :: MAX_SPRITES * SPRITE_VERTS_SIZE Queued_Sprite :: struct { + texture: ^sdl.GPUTexture, + verts: [SPRITE_VERT_COUNT]Vertex, + batch_group: u32, // 0 = strict order; nonzero = caller permits regrouping +} + +Draw_Batch :: struct { + start: int, + count: int, texture: ^sdl.GPUTexture, - verts: [SPRITE_VERT_COUNT]Vertex, } App :: struct { @@ -65,7 +72,7 @@ init :: proc(app: ^App, title: cstring, width, height: i32) -> bool { } requested: sdl.GPUShaderFormat = {.SPIRV, .DXIL, .MSL} - app.device = sdl.CreateGPUDevice(requested, true, nil) + app.device = sdl.CreateGPUDevice(requested, false, nil) if app.device == nil { fmt.eprintfln("CreateGPUDevice failed: %s", sdl.GetError()) return false @@ -76,6 +83,10 @@ init :: proc(app: ^App, title: cstring, width, height: i32) -> bool { return false } + if !sdl.SetGPUAllowedFramesInFlight(app.device, 3) { + fmt.eprintfln("SetGPUAllowedFramesInFlight failed: %s", sdl.GetError()) + } + // Prefer uncapped present for profiling; fall back if unsupported. present := sdl.GPUPresentMode.VSYNC if sdl.WindowSupportsGPUPresentMode(app.device, app.window, .IMMEDIATE) { @@ -242,7 +253,11 @@ end_frame :: proc(app: ^App) { } n := len(app.draw_list) + batches: [dynamic]Draw_Batch + defer delete(batches) if n > 0 { + prepare_draw_batches(app.draw_list[:], &batches) + map_ptr := sdl.MapGPUTransferBuffer(app.device, app.transfer_buffer, false) if map_ptr == nil { fmt.eprintfln("MapGPUTransferBuffer failed: %s", sdl.GetError()) @@ -286,26 +301,20 @@ end_frame :: proc(app: ^App) { app.render_pass = sdl.BeginGPURenderPass(cmd, &color_info, 1, nil) sdl.BindGPUGraphicsPipeline(app.render_pass, app.pipeline) - i := 0 - for i < n { - run := texture_run_len(app.draw_list[:], i) - q0 := app.draw_list[i] - + for batch in batches { sampler_binding := sdl.GPUTextureSamplerBinding { - texture = q0.texture, + texture = batch.texture, sampler = app.sampler, } sdl.BindGPUFragmentSamplers(app.render_pass, 0, &sampler_binding, 1) vb_binding := sdl.GPUBufferBinding { buffer = app.vertex_buffer, - offset = u32(i * SPRITE_VERTS_SIZE), + offset = u32(batch.start * SPRITE_VERTS_SIZE), } sdl.BindGPUVertexBuffers(app.render_pass, 0, &vb_binding, 1) - sdl.DrawGPUPrimitives(app.render_pass, u32(run * SPRITE_VERT_COUNT), 1, 0, 0) - - i += run + sdl.DrawGPUPrimitives(app.render_pass, u32(batch.count * SPRITE_VERT_COUNT), 1, 0, 0) } sdl.EndGPURenderPass(app.render_pass) app.render_pass = nil @@ -454,3 +463,60 @@ texture_run_len :: proc(list: []Queued_Sprite, start: int) -> int { return n } + +texture_run_count :: proc(list: []Queued_Sprite) -> int { + if len(list) == 0 do return 0 + count := 0 + i := 0 + for i < len(list) { + run := texture_run_len(list, i) + count += 1 + i += run + } + return count +} + +prepare_draw_batches :: proc(list: []Queued_Sprite, batches: ^[dynamic]Draw_Batch) { + clear(batches) + group_texture_runs(list) + i := 0 + for i < len(list) { + run := texture_run_len(list, i) + append(batches, Draw_Batch{start = i, count = run, texture = list[i].texture}) + i += run + } +} + +// Within each contiguous nonzero batch_group, stably sort by texture so +// consecutive same-texture sprites become one draw. Group 0 and group +// boundaries are never crossed. +group_texture_runs :: proc(list: []Queued_Sprite) { + start := 0 + for start < len(list) { + group := list[start].batch_group + if group == 0 { + start += 1 + continue + } + + end := start + 1 + for end < len(list) && list[end].batch_group == group { + end += 1 + } + + // Stable insertion sort is sufficient while MAX_SPRITES is 128. + for i in start + 1 ..< end { + item := list[i] + j := i + for j > start { + if uintptr(list[j - 1].texture) <= uintptr(item.texture) { + break + } + list[j] = list[j - 1] + j -= 1 + } + list[j] = item + } + start = end + } +} diff --git a/engine/batching_test.odin b/engine/batching_test.odin index 3aaeabf..a378054 100644 --- a/engine/batching_test.odin +++ b/engine/batching_test.odin @@ -7,6 +7,18 @@ fake_tex :: proc(id: uintptr) -> ^sdl.GPUTexture { return cast(^sdl.GPUTexture)id } +queued :: proc(tex_id: uintptr, group: u32, marker: f32) -> Queued_Sprite { + q: Queued_Sprite + q.texture = fake_tex(tex_id) + q.batch_group = group + q.verts[0].pos = {marker, 0} + return q +} + +marker_of :: proc(q: Queued_Sprite) -> f32 { + return q.verts[0].pos.x +} + @(test) texture_run_len_empty_or_oob :: proc(t: ^testing.T) { testing.expect_value(t, texture_run_len(nil, 0), 0) @@ -62,3 +74,293 @@ texture_run_len_all_different :: proc(t: ^testing.T) { testing.expect_value(t, texture_run_len(list, 1), 1) testing.expect_value(t, texture_run_len(list, 2), 1) } + +@(test) +group_texture_runs_strict_order_unchanged :: proc(t: ^testing.T) { + // Group 0 alternates textures; sorting must not reorder (alpha order). + list := []Queued_Sprite { + queued(2, 0, 1), + queued(1, 0, 2), + queued(2, 0, 3), + queued(1, 0, 4), + } + before_runs := texture_run_count(list) + group_texture_runs(list) + testing.expect_value(t, texture_run_count(list), before_runs) + testing.expect_value(t, marker_of(list[0]), f32(1)) + testing.expect_value(t, marker_of(list[1]), f32(2)) + testing.expect_value(t, marker_of(list[2]), f32(3)) + testing.expect_value(t, marker_of(list[3]), f32(4)) +} + +@(test) +group_texture_runs_reduces_runs_inside_group :: proc(t: ^testing.T) { + list := []Queued_Sprite { + queued(2, 1, 1), + queued(1, 1, 2), + queued(2, 1, 3), + queued(1, 1, 4), + } + testing.expect_value(t, texture_run_count(list), 4) + group_texture_runs(list) + testing.expect_value(t, texture_run_count(list), 2) + testing.expect(t, list[0].texture == fake_tex(1), "lower texture pointer first") + testing.expect(t, list[1].texture == fake_tex(1), "same texture run") + testing.expect(t, list[2].texture == fake_tex(2), "second texture run") + testing.expect(t, list[3].texture == fake_tex(2), "second texture run cont") +} + +@(test) +group_texture_runs_stable_same_texture :: proc(t: ^testing.T) { + list := []Queued_Sprite { + queued(2, 1, 10), + queued(1, 1, 20), + queued(2, 1, 30), + queued(1, 1, 40), + } + group_texture_runs(list) + // Same-texture relative order preserved (stable sort). + testing.expect_value(t, marker_of(list[0]), f32(20)) + testing.expect_value(t, marker_of(list[1]), f32(40)) + testing.expect_value(t, marker_of(list[2]), f32(10)) + testing.expect_value(t, marker_of(list[3]), f32(30)) +} + +@(test) +group_texture_runs_respects_group_boundaries :: proc(t: ^testing.T) { + list := []Queued_Sprite { + queued(2, 1, 1), + queued(1, 1, 2), + queued(2, 0, 3), // strict barrier + queued(1, 2, 4), + queued(2, 2, 5), + } + group_texture_runs(list) + testing.expect_value(t, marker_of(list[2]), f32(3)) // barrier stays put + testing.expect(t, list[0].texture == fake_tex(1)) + testing.expect(t, list[1].texture == fake_tex(2)) + testing.expect(t, list[3].texture == fake_tex(1)) + testing.expect(t, list[4].texture == fake_tex(2)) + testing.expect_value(t, list[0].batch_group, u32(1)) + testing.expect_value(t, list[1].batch_group, u32(1)) + testing.expect_value(t, list[2].batch_group, u32(0)) + testing.expect_value(t, list[3].batch_group, u32(2)) + testing.expect_value(t, list[4].batch_group, u32(2)) +} + +@(test) +group_texture_runs_does_not_merge_across_different_groups :: proc(t: ^testing.T) { + // Adjacent nonzero groups with different ids must not merge runs across. + list := []Queued_Sprite { + queued(1, 1, 1), + queued(2, 1, 2), + queued(1, 2, 3), + queued(2, 2, 4), + } + group_texture_runs(list) + testing.expect_value(t, texture_run_count(list), 4) + testing.expect_value(t, list[0].batch_group, u32(1)) + testing.expect_value(t, list[1].batch_group, u32(1)) + testing.expect_value(t, list[2].batch_group, u32(2)) + testing.expect_value(t, list[3].batch_group, u32(2)) +} + +@(test) +group_texture_runs_empty :: proc(t: ^testing.T) { + list := []Queued_Sprite{} + testing.expect_value(t, texture_run_count(list), 0) + group_texture_runs(list) + testing.expect_value(t, texture_run_count(list), 0) +} + +@(test) +group_texture_runs_already_optimal :: proc(t: ^testing.T) { + list := []Queued_Sprite { + queued(1, 1, 10), + queued(1, 1, 20), + queued(2, 1, 30), + queued(2, 1, 40), + } + testing.expect_value(t, texture_run_count(list), 2) + group_texture_runs(list) + testing.expect_value(t, texture_run_count(list), 2) + testing.expect_value(t, marker_of(list[0]), f32(10)) + testing.expect_value(t, marker_of(list[1]), f32(20)) + testing.expect_value(t, marker_of(list[2]), f32(30)) + testing.expect_value(t, marker_of(list[3]), f32(40)) +} + +@(test) +group_texture_runs_noncontiguous_same_group_id :: proc(t: ^testing.T) { + // Same nonzero id split by group 0: each window regroups alone. + list := []Queued_Sprite { + queued(2, 1, 1), + queued(1, 1, 2), + queued(2, 0, 3), + queued(2, 1, 4), + queued(1, 1, 5), + } + group_texture_runs(list) + testing.expect_value(t, marker_of(list[2]), f32(3)) + testing.expect(t, list[0].texture == fake_tex(1)) + testing.expect(t, list[1].texture == fake_tex(2)) + testing.expect_value(t, list[2].batch_group, u32(0)) + testing.expect(t, list[3].texture == fake_tex(1)) + testing.expect(t, list[4].texture == fake_tex(2)) + testing.expect_value(t, list[0].batch_group, u32(1)) + testing.expect_value(t, list[1].batch_group, u32(1)) + testing.expect_value(t, list[3].batch_group, u32(1)) + testing.expect_value(t, list[4].batch_group, u32(1)) +} + +make_test_draw_app :: proc() -> App { + app: App + app.cmd = cast(^sdl.GPUCommandBuffer)uintptr(1) + app.swapchain_texture = fake_tex(99) + app.swapchain_w = 800 + app.swapchain_h = 600 + app.camera = camera_default() + app.draw_list = make([dynamic]Queued_Sprite) + return app +} + +destroy_test_draw_app :: proc(app: ^App) { + if app == nil do return + delete(app.draw_list) + app^ = {} +} + +make_test_draw_character :: proc() -> Character_Data { + data: Character_Data + data.texture = fake_tex(42) + data.width = 100 + data.height = 100 + data.def.pivot = {0.5, 1.0} + data.def.clips = make(map[string]Clip_Def) + frames := make([]Frame_Def, 1) + frames[0] = Frame_Def { + rect = {0, 0, 10, 10}, + source_size = {10, 10}, + trim_offset = {0, 0}, + } + data.def.clips["idle"] = Clip_Def { + loop = true, + fps = 10, + frames = frames, + } + return data +} + +destroy_test_draw_character :: proc(data: ^Character_Data) { + if data == nil do return + keys := make([dynamic]string, context.temp_allocator) + for key, clip in data.def.clips { + delete(clip.frames) + append(&keys, key) + } + for key in keys { + delete_key(&data.def.clips, key) + } + delete(data.def.clips) + data^ = {} +} + +@(test) +draw_sprite_stamps_batch_group_zero :: proc(t: ^testing.T) { + app := make_test_draw_app() + defer destroy_test_draw_app(&app) + data := make_test_draw_character() + defer destroy_test_draw_character(&data) + + sprite := spawn_sprite(&data, {100, 200}, "idle", 0) + draw_sprite(&app, &sprite) + + testing.expect_value(t, len(app.draw_list), 1) + testing.expect_value(t, app.draw_list[0].batch_group, u32(0)) + testing.expect(t, app.draw_list[0].texture == data.texture) +} + +@(test) +draw_sprite_batched_stamps_batch_group :: proc(t: ^testing.T) { + app := make_test_draw_app() + defer destroy_test_draw_app(&app) + data := make_test_draw_character() + defer destroy_test_draw_character(&data) + + sprite := spawn_sprite(&data, {100, 200}, "idle", 0) + draw_sprite_batched(&app, &sprite, 7) + + testing.expect_value(t, len(app.draw_list), 1) + testing.expect_value(t, app.draw_list[0].batch_group, u32(7)) + testing.expect(t, app.draw_list[0].texture == data.texture) +} + +@(test) +draw_sprite_batched_guards_leave_list_unchanged :: proc(t: ^testing.T) { + app := make_test_draw_app() + defer destroy_test_draw_app(&app) + data := make_test_draw_character() + defer destroy_test_draw_character(&data) + sprite := spawn_sprite(&data, {100, 200}, "idle", 0) + + app.cmd = nil + draw_sprite_batched(&app, &sprite, 1) + testing.expect_value(t, len(app.draw_list), 0) + app.cmd = cast(^sdl.GPUCommandBuffer)uintptr(1) + + app.swapchain_texture = nil + draw_sprite_batched(&app, &sprite, 1) + testing.expect_value(t, len(app.draw_list), 0) + app.swapchain_texture = fake_tex(99) + + draw_sprite_batched(&app, nil, 1) + testing.expect_value(t, len(app.draw_list), 0) + + no_data := sprite + no_data.data = nil + draw_sprite_batched(&app, &no_data, 1) + testing.expect_value(t, len(app.draw_list), 0) + + no_tex := sprite + tex_data := data + tex_data.texture = nil + no_tex.data = &tex_data + draw_sprite_batched(&app, &no_tex, 1) + testing.expect_value(t, len(app.draw_list), 0) +} + +@(test) +prepare_draw_batches_regroups_then_plans_runs :: proc(t: ^testing.T) { + list := []Queued_Sprite { + queued(2, 1, 1), + queued(1, 1, 2), + queued(2, 1, 3), + queued(1, 1, 4), + queued(3, 0, 5), + queued(1, 0, 6), + } + batches := make([dynamic]Draw_Batch) + defer delete(batches) + + prepare_draw_batches(list, &batches) + + testing.expect_value(t, len(batches), 4) + testing.expect_value(t, batches[0].start, 0) + testing.expect_value(t, batches[0].count, 2) + testing.expect(t, batches[0].texture == fake_tex(1)) + testing.expect_value(t, batches[1].start, 2) + testing.expect_value(t, batches[1].count, 2) + testing.expect(t, batches[1].texture == fake_tex(2)) + testing.expect_value(t, batches[2].start, 4) + testing.expect_value(t, batches[2].count, 1) + testing.expect(t, batches[2].texture == fake_tex(3)) + testing.expect_value(t, batches[3].start, 5) + testing.expect_value(t, batches[3].count, 1) + testing.expect(t, batches[3].texture == fake_tex(1)) + + // Group 0 submission order preserved. + testing.expect_value(t, marker_of(list[4]), f32(5)) + testing.expect_value(t, marker_of(list[5]), f32(6)) + testing.expect_value(t, list[4].batch_group, u32(0)) + testing.expect_value(t, list[5].batch_group, u32(0)) +} diff --git a/engine/sprite.odin b/engine/sprite.odin index 12a9dba..103ef79 100644 --- a/engine/sprite.odin +++ b/engine/sprite.odin @@ -82,7 +82,31 @@ to_clip :: proc(px, py, sw, sh: f32) -> [2]f32 { } } +// Axis-aligned quad: two unique x and y values, so scale once and reuse. +sprite_quad_to_clip :: proc(x0, y0, x1, y1, sw, sh: f32) -> [4]Vec2 { + sx := 2.0 / sw + sy := 2.0 / sh + + left := x0 * sx - 1 + right := x1 * sx - 1 + top := 1 - y0 * sy + bottom := 1 - y1 * sy + + return { + {left, top}, + {right, top}, + {right, bottom}, + {left, bottom}, + } +} + draw_sprite :: proc(app: ^App, sprite: ^Sprite) { + draw_sprite_batched(app, sprite, 0) +} + +// Nonzero batch_group lets end_frame regroup consecutive same-group sprites by +// texture. Group 0 keeps exact submission order for correct alpha overlap. +draw_sprite_batched :: proc(app: ^App, sprite: ^Sprite, batch_group: u32) { if app.cmd == nil || app.swapchain_texture == nil { return } @@ -124,10 +148,8 @@ draw_sprite :: proc(app: ^App, sprite: ^Sprite) { sw := f32(app.swapchain_w) sh := f32(app.swapchain_h) - p0 := to_clip(x0_px, y0_px, sw, sh) - p1 := to_clip(x1_px, y0_px, sw, sh) - p2 := to_clip(x1_px, y1_px, sw, sh) - p3 := to_clip(x0_px, y1_px, sw, sh) + points := sprite_quad_to_clip(x0_px, y0_px, x1_px, y1_px, sw, sh) + p0, p1, p2, p3 := points[0], points[1], points[2], points[3] tex_w := f32(sprite.data.width) tex_h := f32(sprite.data.height) @@ -142,7 +164,14 @@ draw_sprite :: proc(app: ^App, sprite: ^Sprite) { {pos = p3, uv = {u0, v1}}, } - append(&app.draw_list, Queued_Sprite{texture = sprite.data.texture, verts = verts}) + append( + &app.draw_list, + Queued_Sprite { + texture = sprite.data.texture, + verts = verts, + batch_group = batch_group, + }, + ) } set_sprite_clip :: proc(sprite: ^Sprite, clip: string) {