Repository navigation
Expand file tree
/
Copy pathqueue.cpp
More file actions
1519 lines (1399 loc) · 63.3 KB
/
Copy pathqueue.cpp
File metadata and controls
1519 lines (1399 loc) · 63.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
// Copyright © 2019-2023
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
// http://www.apache.org/licenses/LICENSE-2.0
#include "vortex2_internal.h"
#include <VX_types.h>
#include <algorithm>
#include <array>
#include <cstring>
namespace vx {
// ============================================================================
// Construction / destruction
// ============================================================================
Queue::Queue(Device* dev, const vx_queue_info_t& info)
: device_(dev),
priority_(static_cast<uint32_t>(info.priority)),
flags_(info.flags) {
device_->retain();
device_->register_queue(this);
worker_ = std::thread([this]{ this->worker_loop(); });
}
Queue::~Queue() {
// Drain + stop the worker. Push a shutdown flag and wake the worker;
// it will finish any commands already in the FIFO and then return.
{
std::lock_guard<std::mutex> g(cmd_mu_);
shutdown_ = true;
}
cmd_cv_.notify_all();
if (worker_.joinable()) worker_.join();
if (device_) {
device_->unregister_queue(this);
device_->release();
}
}
vx_result_t Queue::create(Device* dev, const vx_queue_info_t* info,
Queue** out) {
if (!dev || !out) return VX_ERR_INVALID_VALUE;
vx_queue_info_t default_info = {};
default_info.struct_size = sizeof(default_info);
default_info.priority = VX_QUEUE_PRIORITY_NORMAL;
default_info.flags = 0;
if (!info) info = &default_info;
if (info->struct_size < sizeof(vx_queue_info_t)) return VX_ERR_INVALID_INFO;
*out = new Queue(dev, *info);
return VX_SUCCESS;
}
// ============================================================================
// Worker loop — processes commands strictly in FIFO order.
//
// Each command may have a wait-list of events that must complete before its
// work runs. The waits happen on the worker thread, so an enqueue gated on
// an unsignaled user event does not block the caller. In-order queue
// semantics are preserved because there is exactly one worker per Queue.
// ============================================================================
void Queue::worker_loop() {
while (true) {
Command cmd;
{
std::unique_lock<std::mutex> lk(cmd_mu_);
cmd_cv_.wait(lk, [&]{ return shutdown_ || !commands_.empty(); });
if (commands_.empty()) return; // shutdown with empty queue
cmd = std::move(commands_.front());
commands_.pop_front();
}
// Wait for each external dependency. wait() blocks the worker but
// not the caller; if a wait fails (event errored), short-circuit
// the command's work and propagate the failure into completion.
vx_result_t r = VX_SUCCESS;
for (Event* dep : cmd.waits) {
if (r == VX_SUCCESS) r = dep->wait(VX_TIMEOUT_INFINITE);
dep->release();
}
uint64_t submit_ns = now_ns();
uint64_t start_ns = submit_ns;
uint64_t end_ns = submit_ns;
{
std::lock_guard<std::mutex> g(cmd_mu_);
if (r == VX_SUCCESS && async_error_ != VX_SUCCESS) r = async_error_;
}
if (r == VX_SUCCESS && cmd.work) {
r = cmd.work(&start_ns, &end_ns);
}
if (r != VX_SUCCESS) {
std::lock_guard<std::mutex> g(cmd_mu_);
if (async_error_ == VX_SUCCESS) async_error_ = r;
}
if (cmd.completion) {
if (profiling_enabled()) {
cmd.completion->set_profile(cmd.queued_ns, submit_ns,
start_ns, end_ns);
}
cmd.completion->complete(r);
cmd.completion->release();
}
}
}
// ============================================================================
// enqueue() — common builder: capture waits, allocate completion event,
// stuff the command into the FIFO, notify the worker.
// ============================================================================
vx_result_t Queue::enqueue(Command&& cmd, uint32_t nw, const vx_event_h* w,
vx_event_h* out) {
if (nw != 0 && !w) return VX_ERR_INVALID_VALUE;
// Retain each wait event so the caller can release them immediately
// after enqueue returns. The worker releases them in turn after each
// wait completes.
cmd.waits.reserve(nw);
for (uint32_t i = 0; i < nw; ++i) {
if (!w[i]) return VX_ERR_INVALID_HANDLE;
Event* e = to_event(w[i]);
e->retain();
cmd.waits.push_back(e);
}
// Completion event — created in QUEUED state. The worker will mark it
// COMPLETE (or set ERROR status) once cmd.work runs. We hand the
// caller one ref and the worker holds one ref.
Event* completion = nullptr;
auto r = Event::create(device_, &completion);
if (r != VX_SUCCESS) {
for (Event* e : cmd.waits) e->release();
return r;
}
completion->retain(); // for the worker
cmd.completion = completion;
if (out) *out = to_handle(completion);
else completion->release(); // caller doesn't want it — drop caller's ref
{
std::lock_guard<std::mutex> g(cmd_mu_);
commands_.push_back(std::move(cmd));
}
cmd_cv_.notify_one();
return VX_SUCCESS;
}
// ============================================================================
// flush / finish
// ============================================================================
vx_result_t Queue::flush() {
// The worker is already woken on each enqueue, so this is effectively
// a no-op sync point for higher layers.
cmd_cv_.notify_one();
return VX_SUCCESS;
}
vx_result_t Queue::finish(uint64_t timeout_ns) {
// Enqueue a sentinel barrier and wait for its completion event. This
// is the in-order-queue contract: after finish returns, every
// previously enqueued command has completed (the barrier sits behind
// them in FIFO order).
vx_event_h ev = nullptr;
auto r = this->enqueue_barrier(0, nullptr, &ev);
if (r != VX_SUCCESS) return r;
r = to_event(ev)->wait(timeout_ns);
to_event(ev)->release();
if (r == VX_ERR_TIMEOUT) return r;
// Report the first failure since the last finish, then let the queue run
// again (the commands behind it completed with that error, unexecuted).
std::lock_guard<std::mutex> g(cmd_mu_);
if (async_error_ != VX_SUCCESS) {
r = async_error_;
async_error_ = VX_SUCCESS;
}
return r;
}
// ============================================================================
// Enqueue primitives — each wraps a Platform call into a Command lambda.
// ============================================================================
vx_result_t Queue::enqueue_write(Buffer* dst, uint64_t off, const void* host,
uint64_t sz, uint32_t nw,
const vx_event_h* w, vx_event_h* out) {
if (!dst || (!host && sz != 0)) return VX_ERR_INVALID_VALUE;
if (off + sz > dst->size()) return VX_ERR_INVALID_VALUE;
// Retain dst for the worker's lifetime — caller may release the buffer
// immediately after enqueue returns (matches OpenCL retain semantics).
dst->retain();
Command cmd;
cmd.queued_ns = now_ns();
cmd.work = [this, dst, off, host, sz](uint64_t* s, uint64_t* e) {
vx_result_t r;
{
*s = now_ns();
std::lock_guard<std::mutex> g(enqueue_mu_);
// Host->device through the CP's DMA engine (CMD_MEM_WRITE).
r = device_->cp_submit_mem_write(dst->dev_address() + off,
host, sz);
*e = now_ns();
}
dst->release();
return r;
};
auto r = this->enqueue(std::move(cmd), nw, w, out);
if (r != VX_SUCCESS) dst->release();
return r;
}
vx_result_t Queue::enqueue_read(void* host, Buffer* src, uint64_t so,
uint64_t sz, uint32_t nw,
const vx_event_h* w, vx_event_h* out) {
if (!src || (!host && sz != 0)) return VX_ERR_INVALID_VALUE;
if (so + sz > src->size()) return VX_ERR_INVALID_VALUE;
src->retain();
Command cmd;
cmd.queued_ns = now_ns();
cmd.work = [this, host, src, so, sz](uint64_t* s, uint64_t* e) {
vx_result_t r;
{
*s = now_ns();
std::lock_guard<std::mutex> g(enqueue_mu_);
// Device->host through the CP's DMA engine (CMD_MEM_READ).
r = device_->cp_submit_mem_read(host,
src->dev_address() + so, sz);
*e = now_ns();
}
src->release();
return r;
};
auto r = this->enqueue(std::move(cmd), nw, w, out);
if (r != VX_SUCCESS) src->release();
return r;
}
vx_result_t Queue::enqueue_copy(Buffer* dst, uint64_t do_, Buffer* src,
uint64_t so, uint64_t sz, uint32_t nw,
const vx_event_h* w, vx_event_h* out) {
if (!dst || !src) return VX_ERR_INVALID_VALUE;
if (do_ + sz > dst->size()) return VX_ERR_INVALID_VALUE;
if (so + sz > src->size()) return VX_ERR_INVALID_VALUE;
dst->retain();
src->retain();
Command cmd;
cmd.queued_ns = now_ns();
cmd.work = [this, dst, do_, src, so, sz](uint64_t* s, uint64_t* e) {
vx_result_t r;
{
*s = now_ns();
std::lock_guard<std::mutex> g(enqueue_mu_);
// Device->device through the CP's DMA engine (CMD_MEM_COPY).
r = device_->cp_submit_mem_copy(dst->dev_address() + do_,
src->dev_address() + so, sz);
*e = now_ns();
}
src->release();
dst->release();
return r;
};
auto r = this->enqueue(std::move(cmd), nw, w, out);
if (r != VX_SUCCESS) { src->release(); dst->release(); }
return r;
}
vx_result_t Queue::enqueue_launch(const vx_launch_info_t* info,
uint32_t nw, const vx_event_h* w,
vx_event_h* out) {
if (!info) return VX_ERR_INVALID_VALUE;
if (info->struct_size < sizeof(vx_launch_info_t))
return VX_ERR_INVALID_INFO;
if (info->ndim > 3) return VX_ERR_INVALID_VALUE;
if (info->args_size != 0 && !info->args_host) return VX_ERR_INVALID_VALUE;
// info->kernel is a vx_kernel_h (vx_module_get_kernel). NULL is the legacy
// escape hatch — the caller pre-programmed the PC DCRs itself. Retain the
// kernel for the worker's lifetime so the underlying module image
// (Kernel → Module → image Buffer) stays alive until the launch retires,
// even if the caller releases the kernel immediately. The launch PCs are
// derived from this handle inside the work lambda, so it is the only
// kernel-state we capture.
Kernel* kernel = (info->kernel != nullptr) ? to_kernel(info->kernel) : nullptr;
if (kernel) kernel->retain();
// Copy the args block now so the caller can free/reuse `info` (and the
// memory it points at) the instant enqueue returns. An empty blob is the
// legacy escape hatch — the caller pre-programmed the ARG DCRs itself.
std::vector<uint8_t> args_blob;
if (info->args_host && info->args_size > 0) {
const uint8_t* p = static_cast<const uint8_t*>(info->args_host);
args_blob.assign(p, p + info->args_size);
}
// Capture the launch descriptor by value into the work lambda so the
// caller can free/reuse `info` immediately after enqueue returns.
// ndim==0 is the legacy escape hatch — grid/block DCRs are left to the
// host's prior vx_dcr_write calls (matches legacy vx_start semantics);
// kernel==NULL and args_host==NULL are the analogous PC / ARG hatches.
const uint32_t ndim = info->ndim;
const uint32_t lmem_size = info->lmem_size;
std::array<uint32_t, 3> grid_in = {1, 1, 1};
std::array<uint32_t, 3> block_in = {1, 1, 1};
for (uint32_t i = 0; i < ndim; ++i) {
grid_in [i] = info->grid_dim [i];
block_in[i] = info->block_dim[i];
}
// Cluster shape (CTAs guaranteed co-resident on a core). Normalise
// zero entries to 1 (no grouping). Validate that grid_dim is a whole
// multiple of cluster_dim along each in-use axis.
std::array<uint32_t, 3> lg_in = {1, 1, 1};
for (uint32_t i = 0; i < ndim; ++i) {
uint32_t lg = info->cluster_dim[i];
if (lg == 0) lg = 1;
if (grid_in[i] % lg != 0) {
if (kernel) kernel->release();
return VX_ERR_INVALID_VALUE;
}
lg_in[i] = lg;
}
// A CTA maps to a single core, so its block cannot exceed the core's thread
// capacity (NUM_WARPS × NUM_THREADS). Reject an oversized block here — the
// maxThreadsPerBlock contract — instead of letting the KMU block-size
// descriptor silently wrap to a bogus value. A shared-memory request larger
// than the core's local memory is rejected for the same reason: the
// dispatcher would admit the CTA into a slot too small to hold it and the
// kernel would read and write past it.
if (ndim > 0) {
uint64_t nt = 0, nw = 0, lmem_cap = 0;
auto r = device_->query_caps(VX_CAPS_NUM_THREADS, &nt);
if (r == VX_SUCCESS) {
r = device_->query_caps(VX_CAPS_NUM_WARPS, &nw);
}
if (r == VX_SUCCESS) {
r = device_->query_caps(VX_CAPS_LOCAL_MEM_SIZE, &lmem_cap);
}
if (r != VX_SUCCESS) {
if (kernel) kernel->release();
return r;
}
uint32_t block_size = 1;
for (uint32_t i = 0; i < ndim; ++i) {
block_size *= block_in[i];
}
if (block_size > (uint32_t)(nt * nw)) {
if (kernel) kernel->release();
return VX_ERR_INVALID_VALUE;
}
if (lmem_size > lmem_cap) {
if (kernel) kernel->release();
return VX_ERR_INVALID_VALUE;
}
}
Command cmd;
cmd.queued_ns = now_ns();
cmd.work = [this, kernel, ndim, lmem_size, grid_in, block_in, lg_in,
args_blob = std::move(args_blob)](uint64_t* s, uint64_t* e) {
// Launch PCs, derived from the retained kernel handle. has_kernel is
// false on the legacy escape hatch (caller pre-programmed the DCRs).
// kernel_pc → KERNEL_ENTRY: the selected kernel's function entry,
// read back per-CTA via VX_CSR_CTA_ENTRY.
// program_pc → STARTUP_ADDR: the program image base where every warp
// begins executing (__vx_cta_entry is linked there).
// Both are XLEN-wide (64-bit on build64) and split across the
// paired *_ADDR0/1 / *_ENTRY0/1 DCRs below.
const bool has_kernel = (kernel != nullptr);
const uint64_t kernel_pc = has_kernel ? kernel->pc() : 0;
const uint64_t program_pc = has_kernel ? kernel->module()->base_address() : 0;
// ---- Compute the full KMU descriptor (block_size, warp_step).
uint64_t num_threads = 0, num_warps = 0;
if (ndim > 0) {
auto r = device_->query_caps(VX_CAPS_NUM_THREADS, &num_threads);
if (r != VX_SUCCESS) {
if (kernel) kernel->release();
*s = *e = now_ns(); return r;
}
r = device_->query_caps(VX_CAPS_NUM_WARPS, &num_warps);
if (r != VX_SUCCESS) {
if (kernel) kernel->release();
*s = *e = now_ns(); return r;
}
}
uint32_t eff_block[3] = {1, 1, 1};
for (uint32_t i = 0; i < ndim; ++i) eff_block[i] = block_in[i];
uint32_t block_size = 1;
for (uint32_t i = 0; i < ndim; ++i) block_size *= eff_block[i];
const uint32_t tpw = (uint32_t)num_threads;
const uint32_t ws_x = (ndim >= 1 && eff_block[0]) ?
tpw % eff_block[0] : 0;
const uint32_t ws_y = (ndim >= 2 && eff_block[1]) ?
(tpw / eff_block[0]) % eff_block[1] : 0;
const uint32_t ws_z = (ndim >= 3 && eff_block[2]) ?
(tpw / (eff_block[0] * eff_block[1]))
% eff_block[2] : 0;
// ---- Stage the kernel-args blob into a device scratch slot.
// Empty blob → caller pre-programmed the ARG DCRs (legacy path).
uint64_t args_addr = 0;
bool args_pooled = false;
bool args_staged = !args_blob.empty();
if (args_staged) {
auto r = device_->args_slot_acquire(args_blob.size(),
&args_addr, &args_pooled);
if (r != VX_SUCCESS) {
if (kernel) kernel->release();
*s = *e = now_ns(); return r;
}
r = device_->dev_write(args_addr, args_blob.data(),
args_blob.size());
if (r != VX_SUCCESS) {
device_->args_slot_release(args_addr, args_pooled);
if (kernel) kernel->release();
*s = *e = now_ns();
return r;
}
}
vx_result_t r;
{
std::lock_guard<std::mutex> g(enqueue_mu_);
// Program the KMU DCRs via CMD_DCR_WRITE descriptors through
// the CP ring. has_kernel / args_staged false → caller
// pre-programmed those DCRs (legacy escape hatch).
#define WR(addr, val) do { \
auto _r = device_->cp_submit_dcr_write((addr), (uint32_t)(val)); \
if (_r != VX_SUCCESS) { \
if (args_staged) \
device_->args_slot_release(args_addr, args_pooled); \
if (kernel) kernel->release(); \
*s = *e = now_ns(); \
return _r; \
} \
} while (0)
if (has_kernel) {
WR(VX_DCR_KMU_STARTUP_ADDR0, program_pc & 0xffffffffu);
WR(VX_DCR_KMU_STARTUP_ADDR1, program_pc >> 32);
WR(VX_DCR_KMU_KERNEL_ENTRY0, kernel_pc & 0xffffffffu);
WR(VX_DCR_KMU_KERNEL_ENTRY1, kernel_pc >> 32);
}
if (args_staged) {
WR(VX_DCR_KMU_STARTUP_ARG0, args_addr & 0xffffffffu);
WR(VX_DCR_KMU_STARTUP_ARG1, args_addr >> 32);
}
if (ndim > 0) {
WR(VX_DCR_KMU_BLOCK_DIM_X, eff_block[0]);
WR(VX_DCR_KMU_BLOCK_DIM_Y, eff_block[1]);
WR(VX_DCR_KMU_BLOCK_DIM_Z, eff_block[2]);
WR(VX_DCR_KMU_GRID_DIM_X, grid_in[0]);
WR(VX_DCR_KMU_GRID_DIM_Y, ndim >= 2 ? grid_in[1] : 1);
WR(VX_DCR_KMU_GRID_DIM_Z, ndim >= 3 ? grid_in[2] : 1);
WR(VX_DCR_KMU_LMEM_SIZE, lmem_size);
WR(VX_DCR_KMU_BLOCK_SIZE, block_size);
WR(VX_DCR_KMU_WARP_STEP_X, ws_x);
WR(VX_DCR_KMU_WARP_STEP_Y, ws_y);
WR(VX_DCR_KMU_WARP_STEP_Z, ws_z);
WR(VX_DCR_KMU_CLUSTER_DIM_X, lg_in[0]);
WR(VX_DCR_KMU_CLUSTER_DIM_Y, lg_in[1]);
WR(VX_DCR_KMU_CLUSTER_DIM_Z, lg_in[2]);
}
#undef WR
*s = now_ns();
// cp_submit_launch posts CMD_LAUNCH and polls Q_SEQNUM until
// the engine retires (the engine retires only after Vortex
// signals done, so Q_SEQNUM advance means the kernel
// finished).
r = device_->cp_submit_launch();
*e = now_ns();
}
// Launch retired — the kernel has consumed its args. Return the
// scratch slot to the pool for the next launch.
if (args_staged)
device_->args_slot_release(args_addr, args_pooled);
if (kernel) kernel->release();
return r;
};
auto r = this->enqueue(std::move(cmd), nw, w, out);
if (r != VX_SUCCESS && kernel) kernel->release();
return r;
}
namespace {
// CP wire opcodes for draw-descriptor steps — mirror the CommandProcessor
// opcode enum (sim/common/cmd_processor.h) and device.cpp.
constexpr uint8_t CP_OP_DCR_WRITE = 0x04;
constexpr uint8_t CP_OP_CACHE_FLUSH = 0x0A;
constexpr uint8_t CP_OP_LAUNCH_QMD = 0x0B;
// Fixed per-step stride in a draw descriptor (28-byte cmd-record prefix).
// Must equal CommandProcessor::DRAW_STEP_BYTES.
constexpr int CP_DRAW_STEP_BYTES = 28;
// One captured command in an ordered batch / draw bundle. DCR writes carry
// (addr,value); launches carry the same state vx_enqueue_launch captures
// (retained kernel + copied args + grid/block), resolved to KMU DCRs on the
// worker thread.
struct CmdRec {
bool is_launch = false;
uint32_t dcr_addr = 0;
uint32_t dcr_value = 0;
Kernel* kernel = nullptr; // retained when is_launch
std::vector<uint8_t> args;
uint32_t ndim = 0;
uint32_t lmem_size = 0;
std::array<uint32_t,3> grid = {1, 1, 1};
std::array<uint32_t,3> block = {1, 1, 1};
std::array<uint32_t,3> cluster = {1, 1, 1};
};
// Per-launch device-scratch staging: args blob + QMD launch descriptor.
struct CmdStaged {
uint64_t args_addr = 0; bool args_pooled = false; bool args_active = false;
uint64_t qmd_addr = 0; bool qmd_pooled = false; bool qmd_active = false;
// Host copy of the packed QMD ([count, count x (dcr_addr, value)]) for
// CPs without the CMD_LAUNCH_QMD decoder (see cmd_submit_launch).
std::vector<uint32_t> qmd_words;
};
// Validate `commands`, copy into `recs`, retaining launched kernels (recorded
// in `retained`). On any validation error the retained kernels are released and
// the error is returned; on success the caller owns `retained`.
vx_result_t cmd_build_recs(const vx_command_t* commands, uint32_t count,
std::vector<CmdRec>& recs,
std::vector<Kernel*>& retained) {
recs.reserve(count);
auto fail = [&](vx_result_t e) -> vx_result_t {
for (Kernel* k : retained) k->release();
retained.clear();
return e;
};
for (uint32_t i = 0; i < count; ++i) {
const vx_command_t& c = commands[i];
CmdRec r;
if (c.type == VX_COMMAND_DCR_WRITE) {
r.dcr_addr = c.data.dcr.addr;
r.dcr_value = c.data.dcr.value;
} else if (c.type == VX_COMMAND_LAUNCH) {
const vx_launch_info_t* info = c.data.launch;
if (!info) return fail(VX_ERR_INVALID_VALUE);
if (info->struct_size < sizeof(vx_launch_info_t))
return fail(VX_ERR_INVALID_INFO);
if (info->ndim > 3) return fail(VX_ERR_INVALID_VALUE);
if (info->args_size != 0 && !info->args_host)
return fail(VX_ERR_INVALID_VALUE);
r.is_launch = true;
r.kernel = (info->kernel != nullptr) ? to_kernel(info->kernel) : nullptr;
if (r.kernel) { r.kernel->retain(); retained.push_back(r.kernel); }
if (info->args_host && info->args_size > 0) {
const uint8_t* p = static_cast<const uint8_t*>(info->args_host);
r.args.assign(p, p + info->args_size);
}
r.ndim = info->ndim;
r.lmem_size = info->lmem_size;
for (uint32_t d = 0; d < info->ndim; ++d) {
r.grid [d] = info->grid_dim [d];
r.block[d] = info->block_dim[d];
}
for (uint32_t d = 0; d < info->ndim; ++d) {
uint32_t lg = info->cluster_dim[d];
if (lg == 0) lg = 1;
if (r.grid[d] % lg != 0) return fail(VX_ERR_INVALID_VALUE);
r.cluster[d] = lg;
}
} else {
return fail(VX_ERR_INVALID_VALUE);
}
recs.push_back(std::move(r));
}
return VX_SUCCESS;
}
// Stage each launch's args AND its QMD launch descriptor into device scratch.
// Both are CP DMAs that must complete BEFORE the ring batch / OP_DRAW is
// submitted (the launch's CMD_LAUNCH_QMD reads the QMD from memory at drain
// time). The QMD is the KMU descriptor as a {count,(dcr_addr,value)...} list
// the CP replays — one launch command in place of ~18 CMD_DCR_WRITEs. Returns
// the first error; the caller releases any active slots in `staged`. `tpw` is
// threads-per-warp (NUM_THREADS) for the warp_step derivation.
vx_result_t cmd_stage_qmds(Device* device_, const std::vector<CmdRec>& recs,
std::vector<CmdStaged>& staged, uint32_t tpw) {
vx_result_t r = VX_SUCCESS;
for (size_t i = 0; i < recs.size() && r == VX_SUCCESS; ++i) {
const CmdRec& rec = recs[i];
if (!rec.is_launch) continue;
CmdStaged& st = staged[i];
// 1) Args blob.
if (!rec.args.empty()) {
r = device_->args_slot_acquire(rec.args.size(),
&st.args_addr, &st.args_pooled);
if (r != VX_SUCCESS) break;
st.args_active = true;
r = device_->dev_write(st.args_addr, rec.args.data(), rec.args.size());
if (r != VX_SUCCESS) break;
}
// 2) KMU descriptor values (same derivation as vx_enqueue_launch).
const bool has_kernel = (rec.kernel != nullptr);
const uint64_t kernel_pc = has_kernel ? rec.kernel->pc() : 0;
const uint64_t program_pc = has_kernel
? rec.kernel->module()->base_address() : 0;
uint32_t eff_block[3] = {1, 1, 1};
for (uint32_t d = 0; d < rec.ndim; ++d) eff_block[d] = rec.block[d];
uint32_t block_size = 1;
for (uint32_t d = 0; d < rec.ndim; ++d) block_size *= eff_block[d];
const uint32_t ws_x = (rec.ndim >= 1 && eff_block[0]) ?
tpw % eff_block[0] : 0;
const uint32_t ws_y = (rec.ndim >= 2 && eff_block[1]) ?
(tpw / eff_block[0]) % eff_block[1] : 0;
const uint32_t ws_z = (rec.ndim >= 3 && eff_block[2]) ?
(tpw / (eff_block[0] * eff_block[1]))
% eff_block[2] : 0;
// 3) Pack the QMD: [count, then count × (dcr_addr, value)].
std::vector<uint32_t> qmd;
qmd.push_back(0); // count placeholder
auto put = [&](uint32_t addr, uint32_t val) {
qmd.push_back(addr); qmd.push_back(val);
};
if (has_kernel) {
put(VX_DCR_KMU_STARTUP_ADDR0, uint32_t(program_pc & 0xffffffffu));
put(VX_DCR_KMU_STARTUP_ADDR1, uint32_t(program_pc >> 32));
put(VX_DCR_KMU_KERNEL_ENTRY0, uint32_t(kernel_pc & 0xffffffffu));
put(VX_DCR_KMU_KERNEL_ENTRY1, uint32_t(kernel_pc >> 32));
}
if (st.args_active) {
put(VX_DCR_KMU_STARTUP_ARG0, uint32_t(st.args_addr & 0xffffffffu));
put(VX_DCR_KMU_STARTUP_ARG1, uint32_t(st.args_addr >> 32));
}
if (rec.ndim > 0) {
put(VX_DCR_KMU_BLOCK_DIM_X, eff_block[0]);
put(VX_DCR_KMU_BLOCK_DIM_Y, eff_block[1]);
put(VX_DCR_KMU_BLOCK_DIM_Z, eff_block[2]);
put(VX_DCR_KMU_GRID_DIM_X, rec.grid[0]);
put(VX_DCR_KMU_GRID_DIM_Y, rec.ndim >= 2 ? rec.grid[1] : 1);
put(VX_DCR_KMU_GRID_DIM_Z, rec.ndim >= 3 ? rec.grid[2] : 1);
put(VX_DCR_KMU_LMEM_SIZE, rec.lmem_size);
put(VX_DCR_KMU_BLOCK_SIZE, block_size);
put(VX_DCR_KMU_WARP_STEP_X, ws_x);
put(VX_DCR_KMU_WARP_STEP_Y, ws_y);
put(VX_DCR_KMU_WARP_STEP_Z, ws_z);
put(VX_DCR_KMU_CLUSTER_DIM_X, rec.cluster[0]);
put(VX_DCR_KMU_CLUSTER_DIM_Y, rec.cluster[1]);
put(VX_DCR_KMU_CLUSTER_DIM_Z, rec.cluster[2]);
}
qmd[0] = uint32_t((qmd.size() - 1) / 2); // pair count
// 4) Stage the QMD into device memory when the CP decodes
// CMD_LAUNCH_QMD; otherwise keep only the host copy that
// cmd_submit_launch replays as plain ring commands.
st.qmd_words = std::move(qmd);
if (device_->cp_supports_qmd()) {
const uint64_t qmd_bytes = st.qmd_words.size() * sizeof(uint32_t);
r = device_->args_slot_acquire(qmd_bytes, &st.qmd_addr, &st.qmd_pooled);
if (r != VX_SUCCESS) break;
st.qmd_active = true;
r = device_->dev_write(st.qmd_addr, st.qmd_words.data(), qmd_bytes);
if (r != VX_SUCCESS) break;
}
}
return r;
}
// Submit one staged launch: CMD_LAUNCH_QMD when the CP decodes it, else
// replay the descriptor pairs as CMD_DCR_WRITEs followed by a plain
// CMD_LAUNCH (same trailing cache-flush discipline in cp_submit_launch).
vx_result_t cmd_submit_launch(Device* device_, const CmdStaged& st) {
if (device_->cp_supports_qmd()) {
return device_->cp_submit_launch_qmd(st.qmd_addr);
}
const auto& qmd = st.qmd_words;
const uint32_t count = qmd.empty() ? 0 : qmd[0];
for (uint32_t k = 0; k < count; ++k) {
auto r = device_->cp_submit_dcr_write(qmd[1 + 2 * k], qmd[2 + 2 * k]);
if (r != VX_SUCCESS) return r;
}
return device_->cp_submit_launch();
}
} // namespace
vx_result_t Queue::enqueue_commands(const vx_command_t* commands,
uint32_t count, uint32_t nw,
const vx_event_h* w, vx_event_h* out) {
if (!commands || count == 0) return VX_ERR_INVALID_VALUE;
std::vector<CmdRec> recs;
std::vector<Kernel*> retained; // released if enqueue() fails (work won't run)
auto rb = cmd_build_recs(commands, count, recs, retained);
if (rb != VX_SUCCESS) return rb;
Command cmd;
cmd.queued_ns = now_ns();
cmd.work = [this, recs = std::move(recs)](uint64_t* s, uint64_t* e)
-> vx_result_t {
// Device occupancy for the KMU warp_step computation (same as
// enqueue_launch); resolved once for the whole batch.
uint64_t num_threads = 0, num_warps = 0;
auto rq = device_->query_caps(VX_CAPS_NUM_THREADS, &num_threads);
if (rq == VX_SUCCESS)
rq = device_->query_caps(VX_CAPS_NUM_WARPS, &num_warps);
if (rq != VX_SUCCESS) {
for (const CmdRec& rec : recs) if (rec.kernel) rec.kernel->release();
*s = *e = now_ns();
return rq;
}
// Step 1 — stage each launch's args + QMD into device scratch.
std::vector<CmdStaged> staged(recs.size());
const uint32_t tpw = (uint32_t)num_threads; (void)num_warps;
vx_result_t r = cmd_stage_qmds(device_, recs, staged, tpw);
// Step 2 — emit the whole sequence as one ring batch: each launch is a
// single CMD_LAUNCH_QMD (the CP replays the staged descriptor); FF/state
// DCR writes pass through directly. One doorbell, one poll.
if (r == VX_SUCCESS) {
std::lock_guard<std::mutex> g(enqueue_mu_);
*s = now_ns();
device_->cp_batch_begin();
for (size_t i = 0; i < recs.size(); ++i) {
const CmdRec& rec = recs[i];
r = rec.is_launch
? cmd_submit_launch(device_, staged[i])
: device_->cp_submit_dcr_write(rec.dcr_addr, rec.dcr_value);
if (r != VX_SUCCESS) break;
}
// Always close the batch (commit + poll + drain) even on a mid-batch
// error so the partial sequence retires and the lock releases.
auto re = device_->cp_batch_end();
if (r == VX_SUCCESS) r = re;
*e = now_ns();
} else {
*s = *e = now_ns();
}
// Batch drained → release the args + QMD scratch slots and the kernels.
for (auto& st : staged) {
if (st.args_active) device_->args_slot_release(st.args_addr, st.args_pooled);
if (st.qmd_active) device_->args_slot_release(st.qmd_addr, st.qmd_pooled);
}
for (const CmdRec& rec : recs) if (rec.kernel) rec.kernel->release();
return r;
};
auto r = this->enqueue(std::move(cmd), nw, w, out);
if (r != VX_SUCCESS) { // work lambda won't run → release kernels
for (Kernel* k : retained) k->release();
return r;
}
return r;
}
vx_result_t Queue::enqueue_draw(const vx_command_t* commands,
uint32_t count, uint32_t nw,
const vx_event_h* w, vx_event_h* out) {
if (!commands || count == 0) return VX_ERR_INVALID_VALUE;
std::vector<CmdRec> recs;
std::vector<Kernel*> retained;
auto rb = cmd_build_recs(commands, count, recs, retained);
if (rb != VX_SUCCESS) return rb;
Command cmd;
cmd.queued_ns = now_ns();
cmd.work = [this, recs = std::move(recs)](uint64_t* s, uint64_t* e)
-> vx_result_t {
// Resolve occupancy (warp_step) + core count (per-stage cache flush).
uint64_t num_threads = 0, num_warps = 0, num_cores = 0;
auto rq = device_->query_caps(VX_CAPS_NUM_THREADS, &num_threads);
if (rq == VX_SUCCESS)
rq = device_->query_caps(VX_CAPS_NUM_WARPS, &num_warps);
if (rq == VX_SUCCESS)
rq = device_->query_caps(VX_CAPS_NUM_CORES, &num_cores);
if (rq != VX_SUCCESS) {
for (const CmdRec& rec : recs) if (rec.kernel) rec.kernel->release();
*s = *e = now_ns();
return rq;
}
// Step 1 — stage each launch's args + QMD into device scratch.
std::vector<CmdStaged> staged(recs.size());
const uint32_t tpw = (uint32_t)num_threads; (void)num_warps;
vx_result_t r = cmd_stage_qmds(device_, recs, staged, tpw);
// Step 2 — build the resident draw descriptor: [u32 num_steps] then
// 28-byte cmd-record steps. A launch becomes a CMD_LAUNCH_QMD step
// followed by a CMD_CACHE_FLUSH step — the same sequence the ring batch
// streams (cp_submit_launch_qmd appends a flush after every launch), so
// OP_DRAW executes a byte-identical program. The whole draw is then a
// single OP_DRAW ring command the CP expands on-device.
//
// Only when the CP advertises OP_DRAW decode (CP DEV_CAPS bit 25). On a
// CP without it (e.g. an RTL CP whose OP_DRAW mirror is not yet
// synth-validated), fall back to streaming the same launches + DCRs as a
// ring batch — functionally identical, just N ring commands instead of 1.
// Draw bundles embed CP_OP_LAUNCH_QMD steps, so OP_DRAW also requires
// the QMD decoder (and its staged descriptors).
const bool use_op_draw = device_->cp_supports_draw()
&& device_->cp_supports_qmd();
uint64_t desc_addr = 0; bool desc_pooled = false; bool desc_active = false;
if (r == VX_SUCCESS && use_op_draw) {
std::vector<uint8_t> desc(4, 0); // header placeholder
uint32_t num_steps = 0;
auto put_step = [&](uint8_t opcode, uint64_t a0, uint64_t a1) {
uint8_t step[CP_DRAW_STEP_BYTES] = {0};
step[0] = opcode;
std::memcpy(step + 4, &a0, sizeof(a0));
std::memcpy(step + 12, &a1, sizeof(a1));
desc.insert(desc.end(), step, step + CP_DRAW_STEP_BYTES);
++num_steps;
};
for (size_t i = 0; i < recs.size(); ++i) {
const CmdRec& rec = recs[i];
if (rec.is_launch) {
put_step(CP_OP_LAUNCH_QMD, staged[i].qmd_addr, 0);
put_step(CP_OP_CACHE_FLUSH, num_cores, 0);
} else {
put_step(CP_OP_DCR_WRITE, rec.dcr_addr, rec.dcr_value);
}
}
std::memcpy(desc.data(), &num_steps, sizeof(num_steps));
r = device_->args_slot_acquire(desc.size(), &desc_addr, &desc_pooled);
if (r == VX_SUCCESS) {
desc_active = true;
r = device_->dev_write(desc_addr, desc.data(), desc.size());
}
}
// Step 3 — submit. OP_DRAW: one ring command the CP expands on-device,
// draining each launch (the inter-stage barrier) with no host between.
// Fallback: the same sequence as a single ring batch (one doorbell).
if (r == VX_SUCCESS) {
std::lock_guard<std::mutex> g(enqueue_mu_);
*s = now_ns();
if (use_op_draw) {
r = device_->cp_submit_draw(desc_addr);
} else {
device_->cp_batch_begin();
for (size_t i = 0; i < recs.size(); ++i) {
const CmdRec& rec = recs[i];
r = rec.is_launch
? cmd_submit_launch(device_, staged[i])
: device_->cp_submit_dcr_write(rec.dcr_addr, rec.dcr_value);
if (r != VX_SUCCESS) break;
}
auto re = device_->cp_batch_end();
if (r == VX_SUCCESS) r = re;
}
*e = now_ns();
} else {
*s = *e = now_ns();
}
if (desc_active) device_->args_slot_release(desc_addr, desc_pooled);
for (auto& st : staged) {
if (st.args_active) device_->args_slot_release(st.args_addr, st.args_pooled);
if (st.qmd_active) device_->args_slot_release(st.qmd_addr, st.qmd_pooled);
}
for (const CmdRec& rec : recs) if (rec.kernel) rec.kernel->release();
return r;
};
auto r = this->enqueue(std::move(cmd), nw, w, out);
if (r != VX_SUCCESS) {
for (Kernel* k : retained) k->release();
return r;
}
return r;
}
vx_result_t Queue::enqueue_barrier(uint32_t nw, const vx_event_h* w,
vx_event_h* out) {
// A barrier is a no-op work item; its purpose is to introduce a
// synchronization point that completes only after all waits resolve.
Command cmd;
cmd.queued_ns = now_ns();
cmd.work = [](uint64_t* s, uint64_t* e) {
uint64_t t = now_ns();
*s = t; *e = t;
return VX_SUCCESS;
};
return this->enqueue(std::move(cmd), nw, w, out);
}
// ============================================================================
// Rect-DMA helpers — software fallback for the *_rect enqueues. Each rect is
// decomposed into linear transfers; a future CP DMA descriptor with native
// 3D stride can replace the per-row loop without an API change.
// ============================================================================
namespace {
struct ResolvedRect {
size_t region[3];
size_t buffer_origin[3];
size_t host_origin[3];
size_t buffer_row, buffer_slice;
size_t host_row, host_slice;
};
// Apply OpenCL pitch defaults (0 row pitch -> region[0]; 0 slice pitch ->
// region[1] * row_pitch) and reject a degenerate region or a pitch too
// small to hold its row / slice.
vx_result_t resolve_rect(const vx_rect_info_t& in, ResolvedRect* out) {
for (int i = 0; i < 3; ++i) {
out->region[i] = in.region[i];
out->buffer_origin[i] = in.buffer_origin[i];
out->host_origin[i] = in.host_origin[i];
}
if (in.region[0] == 0 || in.region[1] == 0 || in.region[2] == 0)
return VX_ERR_INVALID_VALUE;
out->buffer_row = in.buffer_row_pitch ? in.buffer_row_pitch
: in.region[0];
out->buffer_slice = in.buffer_slice_pitch ? in.buffer_slice_pitch
: out->buffer_row * in.region[1];
out->host_row = in.host_row_pitch ? in.host_row_pitch
: in.region[0];
out->host_slice = in.host_slice_pitch ? in.host_slice_pitch
: out->host_row * in.region[1];
if (out->buffer_row < in.region[0] || out->host_row < in.region[0])
return VX_ERR_INVALID_VALUE;
// The slice pitch only strides between slices, so it is used solely when
// region[2] > 1. A single-slice rect (region[2] == 1) never dereferences it
// -- e.g. a 1D image array is presented as {width, layers, 1} with a slice
// pitch of image_row_pitch, legitimately smaller than row * layers. Only
// enforce the lower bound where the slice pitch is actually addressed.
if (in.region[2] > 1 &&
(out->buffer_slice < out->buffer_row * in.region[1] ||
out->host_slice < out->host_row * in.region[1]))
return VX_ERR_INVALID_VALUE;
return VX_SUCCESS;
}
// Byte offset of row `row` of slice `sl` within one side of the rect.
uint64_t rect_off(const size_t origin[3], size_t row_pitch, size_t slice_pitch,
size_t sl, size_t row) {
return (uint64_t)(origin[2] + sl) * slice_pitch
+ (uint64_t)(origin[1] + row) * row_pitch
+ origin[0];
}
// Highest byte (exclusive) one side of the rect touches — for bounds checks.
uint64_t rect_span(const size_t region[3], const size_t origin[3],
size_t row_pitch, size_t slice_pitch) {
return rect_off(origin, row_pitch, slice_pitch,
region[2] - 1, region[1] - 1) + region[0];
}
// Walk the rect, invoking row_fn(buffer_off, host_off, len) for each
// contiguous run — once for the whole rect when it is fully contiguous,
// otherwise once per row. Stops at the first failing transfer.
template <class Fn>
vx_result_t rect_for_each(const ResolvedRect& r, Fn&& row_fn) {
const bool contig =
r.buffer_row == r.region[0] &&
r.host_row == r.region[0] &&
r.buffer_slice == r.region[1] * r.region[0] &&
r.host_slice == r.region[1] * r.region[0];
if (contig) {
return row_fn(
rect_off(r.buffer_origin, r.buffer_row, r.buffer_slice, 0, 0),
rect_off(r.host_origin, r.host_row, r.host_slice, 0, 0),
(uint64_t)r.region[0] * r.region[1] * r.region[2]);
}
for (size_t sl = 0; sl < r.region[2]; ++sl) {
for (size_t row = 0; row < r.region[1]; ++row) {
auto rc = row_fn(
rect_off(r.buffer_origin, r.buffer_row, r.buffer_slice, sl, row),
rect_off(r.host_origin, r.host_row, r.host_slice, sl, row),
(uint64_t)r.region[0]);
if (rc != VX_SUCCESS) return rc;
}
}
return VX_SUCCESS;
}
} // namespace
vx_result_t Queue::enqueue_read_rect(void* host_dst, Buffer* src,
const vx_rect_info_t* info,
uint32_t nw, const vx_event_h* w,