-
Notifications
You must be signed in to change notification settings - Fork 53
Expand file tree
/
Copy pathupload_queue.hpp
More file actions
112 lines (97 loc) · 4.88 KB
/
Copy pathupload_queue.hpp
File metadata and controls
112 lines (97 loc) · 4.88 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
#pragma once
#include <NoGraphicsAPI/NoGraphicsAPI.hpp>
namespace gpu
{
struct UploadQueueStats
{
uint64 capacity = 0;
uint64 bytes_in_use = 0;
uint64 peak_bytes = 0;
uint64 submissions = 0;
uint64 waits = 0;
uint32 pending_batches = 0;
uint32 pending_operations = 0;
};
// Externally synchronize this uploader and its selected queue. Destination and callback resource lifetimes remain caller-owned.
// Order texture initialization and prior resource use on that queue before enqueuing uploads, including any waits on other queues.
class UploadQueue
{
public:
static constexpr uint32 operation_limit = 4096;
static constexpr uint64 alignment = 16;
// Capacity must be a positive multiple of 16. queue_index must be less than DeviceCaps::queue_count.
// max_pending_batches must be nonzero; normally match the number of frames in flight.
explicit UploadQueue(Device* device, uint64 capacity = 64ull * 1024 * 1024, uint32 queue_index = 0, uint32 max_pending_batches = 2) noexcept;
~UploadQueue() noexcept;
// Flushes and waits for uploads. Finish external submissions using returned timeline points before destruction or move assignment.
void destroy() noexcept;
UploadQueue(const UploadQueue&) = delete;
UploadQueue& operator=(const UploadQueue&) = delete;
UploadQueue(UploadQueue&& other) noexcept;
UploadQueue& operator=(UploadQueue&& other) noexcept;
// Sources are copied immediately. Staging and batch storage never grow; reuse waits for completion when needed.
// Flush between overlapping transfer writes. Buffer copies follow copy_memory's requirements; large sources are split to fit.
void upload_buffer(GpuRange destination, ByteSpan source) noexcept;
// Source covers one complete region, including pitch padding, and must fit capacity. The region follows copy_memory_to_texture's requirements.
void upload_texture(Texture* destination, ByteSpan source, const TextureCopyDesc& copy = {}) noexcept;
// Reserves byte_size (1..capacity), then invokes callback(commands, staging) synchronously with 16-byte-aligned CPU/GPU addresses.
// Requires a general or compute queue. Fill staging and record compute work; the reservation remains occupied until submission completes.
// The callback must not mutate the uploader or end/submit/retain its command buffer. Staging cannot be used by later submissions.
// Bind required pipeline/heaps and record dependencies between dispatches inside the callback.
template<typename Callback>
void upload_with_compute(uint64 byte_size, Callback&& callback) noexcept
{
const GpuCpuRange<byte> staging = begin_compute(byte_size);
static_assert(noexcept(callback(state_.commands, staging)));
callback(state_.commands, staging);
end_compute();
}
// The device's per-command-buffer timestamp limit applies to each batch. Keep CPU destinations valid until the batch is reclaimed.
void write_timestamp(uint64* cpu_destination) noexcept;
// Returns the latest upload completion point, borrowing this uploader's semaphore. Other queues can consume it through SubmitDesc::waits.
TimelinePoint flush() noexcept;
void wait() noexcept;
// Retrieves timestamps and reclaims completed batches without waiting or flushing pending uploads. wait() also reclaims all batches.
void reclaim() noexcept;
// Occupied bytes and pending batches reflect retirements observed by the most recent operation.
[[nodiscard]] UploadQueueStats stats() const noexcept;
private:
friend void upload_texture(UploadQueue& queue, Texture* destination, const TextureDesc& description,
ByteSpan source, const TextureCopyDesc& copy) noexcept;
struct Batch
{
CommandPool* pool = nullptr;
uint64 end = 0;
uint64 value = 0;
};
struct State
{
Device* device = nullptr;
GpuHeap heap{};
TimelinePoint completion{};
CommandBuffer* commands = nullptr;
uint64 head = 0;
uint64 tail = 0;
Batch* batches = nullptr;
uint32 max_pending_batches = 0;
uint32 queue_index = 0;
Stage upload_stages = Stage::transfer;
Access upload_access = Access::transfer_read | Access::transfer_write;
Access queue_access = Access::transfer_read | Access::transfer_write;
uint32x3 texture_granularity = {.x = 1, .y = 1, .z = 1};
uint32 retirement_first = 0;
uint32 retirement_count = 0;
uint32 operation_count = 0;
bool in_callback = false;
uint64 submissions = 0;
uint64 waits = 0;
uint64 peak_bytes = 0;
};
CommandBuffer* begin() noexcept;
void wait_oldest() noexcept;
GpuCpuRange<byte> reserve(uint64 byte_size) noexcept;
GpuCpuRange<byte> begin_compute(uint64 byte_size) noexcept;
void end_compute() noexcept;
State state_{};
};
} // namespace gpu