Repository navigation
Expand file tree
/
Copy pathcache.cpp
More file actions
1987 lines (1804 loc) · 73.8 KB
/
Copy pathcache.cpp
File metadata and controls
1987 lines (1804 loc) · 73.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
// Copyright © 2019-2023
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
#include "cache.h"
#include "mem_block_pool.h"
#include "debug.h"
#include "types.h"
#if VX_CFG_EXT_A_ENABLED
#include "amo_unit.h"
#endif
#include <cstring>
#include <list>
#include <queue>
#include <unordered_map>
#include <util.h>
#include <vector>
using namespace vortex;
struct params_t {
uint32_t sets_per_bank;
uint32_t lines_per_set;
uint32_t words_per_line;
uint32_t sectors_per_line; // = 1 when S == L (no sectoring); the mem/fill/evict granule
uint32_t log2_num_inputs;
int32_t sector_select_addr_start; // sector index within the line (high part of the in-line offset)
int32_t sector_select_addr_end;
int32_t word_select_addr_start;
int32_t word_select_addr_end;
int32_t bank_select_addr_start;
int32_t bank_select_addr_end;
int32_t set_select_addr_start;
int32_t set_select_addr_end;
int32_t tag_select_addr_start;
int32_t tag_select_addr_end;
params_t(const Cache::Config &config) {
int32_t offset_bits = config.L - config.W;
int32_t index_bits = config.C - (config.L + config.A + config.B);
assert(offset_bits >= 0);
assert(index_bits >= 0);
this->log2_num_inputs = log2ceil(config.num_inputs);
int32_t sector_bits = config.L - config.S; // sectors per line = 2^sector_bits
assert(sector_bits >= 0);
this->sets_per_bank = 1 << index_bits;
this->lines_per_set = 1 << config.A;
this->words_per_line = 1 << offset_bits;
this->sectors_per_line = 1 << sector_bits;
// Sector select: high part of the in-line offset, bits [S .. L-1].
this->sector_select_addr_start = config.S;
this->sector_select_addr_end = (config.S + sector_bits - 1);
// Word select
this->word_select_addr_start = config.W;
this->word_select_addr_end = (this->word_select_addr_start + offset_bits - 1);
// Bank select
this->bank_select_addr_start = (1 + this->word_select_addr_end);
this->bank_select_addr_end = (this->bank_select_addr_start + config.B - 1);
// Set select
this->set_select_addr_start = (1 + this->bank_select_addr_end);
this->set_select_addr_end = (this->set_select_addr_start + index_bits - 1);
// Tag select
this->tag_select_addr_start = (1 + this->set_select_addr_end);
this->tag_select_addr_end = (config.addr_width - 1);
}
uint32_t addr_bank_id(uint64_t addr) const {
if (bank_select_addr_end >= bank_select_addr_start)
return (uint32_t)bit_getw(addr, bank_select_addr_start, bank_select_addr_end);
else
return 0;
}
uint32_t addr_set_id(uint64_t addr) const {
if (set_select_addr_end >= set_select_addr_start)
return (uint32_t)bit_getw(addr, set_select_addr_start, set_select_addr_end);
else
return 0;
}
uint64_t addr_tag(uint64_t addr) const {
if (tag_select_addr_end >= tag_select_addr_start)
return bit_getw(addr, tag_select_addr_start, tag_select_addr_end);
else
return 0;
}
uint32_t addr_sector_id(uint64_t addr) const {
if (sector_select_addr_end >= sector_select_addr_start)
return (uint32_t)bit_getw(addr, sector_select_addr_start, sector_select_addr_end);
else
return 0;
}
uint64_t mem_addr(uint32_t bank_id, uint32_t set_id, uint64_t tag) const {
uint64_t addr(0);
if (bank_select_addr_end >= bank_select_addr_start)
addr = bit_setw(addr, bank_select_addr_start, bank_select_addr_end, bank_id);
if (set_select_addr_end >= set_select_addr_start)
addr = bit_setw(addr, set_select_addr_start, set_select_addr_end, set_id);
if (tag_select_addr_end >= tag_select_addr_start)
addr = bit_setw(addr, tag_select_addr_start, tag_select_addr_end, tag);
return addr;
}
// Byte address of a specific sector within the line (line base + sector offset).
// The memory below transacts at this (sector) granule.
uint64_t mem_addr_sector(uint32_t bank_id, uint32_t set_id, uint64_t tag, uint32_t sector_id) const {
uint64_t addr = mem_addr(bank_id, set_id, tag);
if (sector_select_addr_end >= sector_select_addr_start)
addr = bit_setw(addr, sector_select_addr_start, sector_select_addr_end, sector_id);
return addr;
}
};
// One sector of a line: the fill / eviction / mem-request granule (= mem_block).
struct sector_t {
bool valid;
bool dirty;
uint64_t dirty_mask; // per-byte dirty bits; the writeback byte-enable
std::shared_ptr<mem_block_t> data; // sector bytes (= one mem_block)
void reset() {
valid = false;
dirty = false;
dirty_mask = 0;
data.reset();
}
};
struct line_t {
uint64_t tag;
uint32_t lru_ctr;
std::vector<sector_t> sectors; // sectors_per_line (1 when no sectoring)
bool any_valid() const {
for (const auto &s : sectors) if (s.valid) return true;
return false;
}
bool any_dirty() const {
for (const auto &s : sectors) if (s.valid && s.dirty) return true;
return false;
}
void reset() {
lru_ctr = 0;
for (auto &s : sectors) s.reset();
}
};
static inline void sector_merge(sector_t& sec, const std::shared_ptr<mem_block_t>& src, uint64_t byteen) {
// Copy-on-write only when shared. sec.data may be aliased with in-flight
// responses, fill payloads, or writeback messages; mutating in place would
// corrupt them. When this sector is the sole owner, mutate in place to
// avoid a heap allocation on the hot path.
if (sec.data) {
if (sec.data.use_count() > 1) {
sec.data = make_mem_block_copy(*sec.data);
}
} else {
sec.data = make_mem_block();
std::memset(sec.data->data(), 0, sec.data->size());
}
if (src) {
for (uint32_t b = 0; b < VX_CFG_MEM_BLOCK_SIZE; ++b) {
if (byteen & (1ull << b)) {
(*sec.data)[b] = (*src)[b];
}
}
}
}
struct set_t {
std::vector<line_t> lines;
uint32_t fifo_ptr; // next victim for FIFO policy
set_t(uint32_t num_ways, uint32_t sectors_per_line)
: lines(num_ways), fifo_ptr(0) {
for (auto &line : lines)
line.sectors.resize(sectors_per_line);
}
void reset() {
for (auto &line : lines) {
line.reset();
}
fifo_ptr = 0;
}
// Pure tag lookup: returns the line-resident way (tag match with any sector
// valid) or -1; fills free/repl line ids. No mutation. The caller derives a
// sector hit by checking the requested sector's valid bit on the returned way.
// Callers must invoke update_lru() *after* all stall checks pass, otherwise
// PLRU counters drift on retry.
int tag_match(uint64_t tag, uint8_t policy, uint32_t rand_idx,
int *free_line_id, int *repl_line_id) const {
int hit_line_id = -1;
*free_line_id = -1;
*repl_line_id = 0;
uint32_t max_cnt = 0;
bool has_valid = false;
bool plru_chosen = false;
for (uint32_t i = 0, n = lines.size(); i < n; ++i) {
const auto &line = lines.at(i);
if (!line.any_valid()) {
if (*free_line_id == -1)
*free_line_id = i;
continue;
}
has_valid = true;
if (line.tag == tag)
hit_line_id = i;
if (policy == Cache::PLRU) {
if (!plru_chosen || line.lru_ctr >= max_cnt) {
max_cnt = line.lru_ctr;
*repl_line_id = i;
plru_chosen = true;
}
}
}
// Select victim per policy (for miss path).
switch (policy) {
case Cache::FIFO:
*repl_line_id = fifo_ptr % lines.size();
break;
case Cache::RANDOM:
*repl_line_id = rand_idx % lines.size();
break;
case Cache::PLRU:
default:
if (!has_valid)
*repl_line_id = (*free_line_id != -1) ? *free_line_id : 0;
break;
}
return hit_line_id;
}
// Apply PLRU age update for a tag access. Pass hit_line_id == -1 for a miss
// (all valid lines age, no reset). Call once per *committed* access.
void update_lru(int hit_line_id) {
for (uint32_t i = 0, n = lines.size(); i < n; ++i) {
auto &line = lines.at(i);
if (!line.any_valid())
continue;
if ((int)i == hit_line_id) {
line.lru_ctr = 0;
} else {
++line.lru_ctr;
}
}
}
// Choose a victim line for installing a fill. Does NOT mutate state.
int select_victim(uint8_t policy, uint32_t rand_idx,
int *free_line_id, int *repl_line_id) const {
*free_line_id = -1;
*repl_line_id = 0;
uint32_t max_cnt = 0;
bool has_valid = false;
for (uint32_t i = 0, n = lines.size(); i < n; ++i) {
const auto &line = lines.at(i);
if (!line.any_valid()) {
if (*free_line_id == -1)
*free_line_id = i;
continue;
}
has_valid = true;
if (policy == Cache::PLRU) {
if (line.lru_ctr >= max_cnt) {
max_cnt = line.lru_ctr;
*repl_line_id = i;
}
}
}
switch (policy) {
case Cache::FIFO:
*repl_line_id = fifo_ptr % lines.size();
break;
case Cache::RANDOM:
*repl_line_id = rand_idx % lines.size();
break;
case Cache::PLRU:
default:
if (!has_valid)
*repl_line_id = (*free_line_id != -1) ? *free_line_id : 0;
break;
}
return (*free_line_id != -1) ? *free_line_id : *repl_line_id;
}
// Find the way currently resident for `tag` (any sector valid), or -1.
int find_resident(uint64_t tag) const {
for (uint32_t i = 0, n = lines.size(); i < n; ++i) {
if (lines.at(i).any_valid() && lines.at(i).tag == tag)
return (int)i;
}
return -1;
}
};
struct bank_req_t {
using Ptr = std::shared_ptr<bank_req_t>;
enum ReqType {
None = 0,
Fill = 1,
Replay = 2,
Core = 3,
AmoProbe = 4 // non-LLC AMO passthrough: probe-and-invalidate then forward
};
uint64_t addr;
uint32_t hart_id;
uint64_t req_tag;
uint64_t uuid;
uint32_t mshr_id;
ReqType type;
bool write;
MemOp op; // AMO state (op/width/rhs/hart_id) is derived from
// this + byteen + data + hart_id without a sideband.
// For write-through write-misses that piggy-back on a pending fill MSHR:
// the core response was already sent at miss time, so Replay must not
// emit another response — only run sector_merge.
bool skip_core_rsp;
// TLM data:
// For Core writes: incoming write data + byteen.
// For Fill: captured fill data from below (mem_rsp.data).
std::shared_ptr<mem_block_t> data;
uint64_t byteen;
// Comparand, read only by compare-and-swap: data already carries the swap
// value, so the comparand cannot share it.
uint64_t amo_cmp;
// `flags.amo_unsigned` distinguishes signed vs unsigned MIN/MAX.
// Other bits ride along for future use.
MemFlags flags;
// Bank admission order, stamped once in processInputs. MSHR replay drains
// by this stamp, so chain order is admission order regardless of whether a
// request joins the chain at admission or later at pipe exit (a request
// still riding the pipe must not be ordered behind a younger request that
// coalesced at admission).
uint64_t adm_seq;
// AMO state. memop_is_atomic(op) classifies the request; the LSU
// packs rs2 into data at byte_off, so the cache extracts rhs via
// amo_load_word and width via byteen popcount.
bank_req_t() {
this->reset();
}
void reset() {
addr = 0;
hart_id = 0;
req_tag = 0;
uuid = 0;
mshr_id = 0;
type = ReqType::None;
write = false;
op = MemOp::LD;
skip_core_rsp = false;
data.reset();
byteen = 0;
amo_cmp = 0;
flags = MemFlags{};
adm_seq = 0;
}
friend std::ostream &operator<<(std::ostream &os, const bank_req_t &req) {
os << "addr=0x" << std::hex << req.addr;
os << ", rw=" << std::dec << req.write;
os << ", type=" << req.type;
os << ", req_tag=" << req.req_tag;
os << ", hart_id=" << req.hart_id;
os << " (#" << req.uuid << ")";
return os;
}
};
struct mshr_entry_t {
bank_req_t bank_req;
uint32_t set_id;
uint64_t addr_tag;
uint32_t sector_id; // coalescing/fill granule: a miss is per-(set,tag,sector)
uint32_t line_id;
uint64_t seq; // enqueue order; replays drain oldest-first to preserve
// program order among coalesced same-line accesses.
mshr_entry_t() {
this->reset();
}
void reset() {
bank_req.reset();
set_id = 0;
addr_tag = 0;
sector_id = 0;
line_id = 0;
seq = 0;
}
};
class MSHR {
public:
MSHR(uint32_t size)
: entries_(size), ready_reqs_(0), size_(0) {}
uint32_t capacity() const {
return entries_.size();
}
uint32_t size() const {
return size_;
}
bool empty() const {
return (0 == size_);
}
bool full() const {
assert(size_ <= entries_.size());
return (size_ == entries_.size());
}
bool has_ready_reqs() const {
return (ready_reqs_ != 0);
}
const mshr_entry_t &peek(uint32_t id) const {
return entries_.at(id);
}
// Returns true if there is an active pending request for the given
// set/tag/sector. Coalescing is per-sector: a fill installs one sector, so
// only same-sector misses share it (different-sector misses to the same line
// each get their own fill). If true, optionally returns the root entry id.
bool lookup(uint32_t set_id, uint64_t addr_tag, uint32_t sector_id, uint32_t *root_id = nullptr) const {
for (uint32_t i = 0, n = entries_.size(); i < n; ++i) {
const auto &entry = entries_.at(i);
if (entry.bank_req.type != bank_req_t::None && entry.set_id == set_id && entry.addr_tag == addr_tag && entry.sector_id == sector_id) {
if (root_id)
*root_id = i;
return true;
}
}
return false;
}
// True if any matching entry OTHER than exclude_id is still Core-typed,
// i.e. the chain's fill hasn't run replay() over it yet. Distinguishes a
// fill-pending chain (a plain enqueue will be marked by the fill) from a
// draining chain (a new entry must be marked ready via defer_to_replay).
bool has_pending_fill(uint32_t set_id, uint64_t addr_tag, uint32_t sector_id, int exclude_id) const {
for (int i = 0, n = (int)entries_.size(); i < n; ++i) {
if (i == exclude_id) {
continue;
}
const auto &entry = entries_.at(i);
if (entry.bank_req.type == bank_req_t::Core && entry.set_id == set_id && entry.addr_tag == addr_tag && entry.sector_id == sector_id) {
return true;
}
}
return false;
}
// True if any matching entry carries ordering weight for a later plain
// access: a write must merge before a younger same-line read responds, and
// an atomic needs its whole round-trip ordered. Pure-load chains carry no
// such constraint — a younger access may complete ahead of their drain.
bool has_ordered_reqs(uint32_t set_id, uint64_t addr_tag, uint32_t sector_id) const {
for (const auto &entry : entries_) {
if (entry.bank_req.type == bank_req_t::None) {
continue;
}
if (entry.set_id != set_id || entry.addr_tag != addr_tag || entry.sector_id != sector_id) {
continue;
}
if (entry.bank_req.write || memop_is_atomic(entry.bank_req.op)) {
return true;
}
}
return false;
}
// Enqueue a new core request and return the allocated entry id.
int enqueue(const bank_req_t &bank_req, uint32_t set_id, uint64_t addr_tag, uint32_t sector_id) {
assert(bank_req.type == bank_req_t::Core);
for (uint32_t i = 0, n = entries_.size(); i < n; ++i) {
auto &entry = entries_.at(i);
if (entry.bank_req.type == bank_req_t::None) {
entry.bank_req = bank_req;
entry.set_id = set_id;
entry.addr_tag = addr_tag;
entry.sector_id = sector_id;
entry.line_id = 0; // victim is selected at Fill time
entry.seq = bank_req.adm_seq;
++size_;
return i;
}
}
std::abort(); // no free slot found!
return -1;
}
// Mark all pending requests matching the entry's (set,tag,sector) for replay.
mshr_entry_t &replay(uint32_t id) {
auto &root_entry = entries_.at(id);
assert(root_entry.bank_req.type == bank_req_t::Core);
// A prior fill's replay batch may still be draining: a stalled fill
// (writeback egress backpressure) can sit in the pipe while a second
// fill is admitted behind it, so two fills' replays can be live at once.
// This is safe — the MSHR coalesces misses, so the two fills are for
// distinct lines and this replay marks only its own line's waiters; the
// accumulated ready_reqs_ stays correct and dequeue drains oldest-first, so
// each line's waiters still replay in program order. Double-replaying the
// same fill is caught by the Core-type assert above.
for (auto &entry : entries_) {
if (entry.bank_req.type == bank_req_t::Core && entry.set_id == root_entry.set_id && entry.addr_tag == root_entry.addr_tag && entry.sector_id == root_entry.sector_id) {
entry.bank_req.type = bank_req_t::Replay;
++ready_reqs_;
}
}
return root_entry;
}
// Convert a just-enqueued entry into a ready Replay. Used when a request
// targets a line whose same-(set,tag,sector) chain is already draining:
// nothing will mark the new entry, so it joins the replay stream itself
// and seq order keeps it behind the older chained accesses.
void defer_to_replay(uint32_t id) {
auto &entry = entries_.at(id);
assert(entry.bank_req.type == bank_req_t::Core);
entry.bank_req.type = bank_req_t::Replay;
++ready_reqs_;
}
// Peek the entry dequeue() would return next (oldest ready replay), or
// nullptr. Lets the fill-forward drain inspect the chain head without
// consuming it.
const mshr_entry_t* peek_next_replay() const {
const mshr_entry_t *picked = nullptr;
for (auto &entry : entries_) {
if (entry.bank_req.type != bank_req_t::Replay)
continue;
if (picked == nullptr || entry.seq < picked->seq)
picked = &entry;
}
return picked;
}
// Dequeue the next ready replay request in program order (oldest enqueue
// sequence first). Coalesced accesses to the same line must replay in the
// order they were issued: a store that precedes a load to the same line has
// to merge its data before the load captures its response, and a load that
// precedes a store must capture the pre-store line. A blanket reads-before-
// writes policy gets the store-then-load case wrong — under write-allocate
// the fill only carries memory's stale data, so the store's own replay is
// what installs the new value.
void dequeue(bank_req_t *out) {
assert(ready_reqs_ > 0);
mshr_entry_t *picked = nullptr;
for (auto &entry : entries_) {
if (entry.bank_req.type != bank_req_t::Replay)
continue;
if (picked == nullptr || entry.seq < picked->seq)
picked = &entry;
}
*out = picked->bank_req;
picked->bank_req.type = bank_req_t::None;
--ready_reqs_;
--size_;
}
void reset() {
for (auto &entry : entries_) {
entry.reset();
}
ready_reqs_ = 0;
size_ = 0;
}
private:
std::vector<mshr_entry_t> entries_;
uint32_t ready_reqs_;
uint32_t size_;
};
class CacheBank : public SimObject<CacheBank> {
public:
// A memory-side request leaves the bank through a queue with a registered
// output stage, so it becomes visible downstream two cycles after the bank
// commits it, not one.
static constexpr uint32_t MEM_REQ_DELAY = 2;
// Non-LLC AMO passthrough table capacity. Also partitions the memory-side
// tag namespace and counts toward the response-sink credit on mem_rsp_in.
static constexpr uint32_t AMO_PASSTHRU_CAP = 8;
SimChannel<MemReq> core_req_in;
SimChannel<MemRsp> core_rsp_out;
SimChannel<MemReq> mem_req_out;
// Sized to sink every response this bank can solicit (one per MSHR fill
// plus one per AMO passthrough probe). A memory response must never
// back-pressure its sender: if it can, the request and response channels
// deadlock through each other across cache levels — this bank's pipe head
// stalls on a full mem_req_out while the downstream cache's pipe head
// stalls on the response it cannot deliver here.
SimChannel<MemRsp> mem_rsp_in;
CacheBank(const SimContext &ctx,
const char *name,
const Cache::Config &config,
const params_t ¶ms,
uint32_t bank_id)
: SimObject<CacheBank>(ctx, name), core_req_in(this), core_rsp_out(this), mem_req_out(this), mem_rsp_in(this, config.mshr_size + AMO_PASSTHRU_CAP), config_(config), params_(params), bank_id_(bank_id), sets_(params.sets_per_bank, set_t(params.lines_per_set, params.sectors_per_line)), mshr_(config.mshr_size), pipe_req_(TFifo<bank_req_t>::Create("", config.latency)), rand_ctr_(0), adm_seq_ctr_(0)
#if VX_CFG_EXT_A_ENABLED
, amo_unit_(__MAX(2u, (uint32_t)VX_CFG_AMO_RS_SIZE))
#endif
{
this->on_reset();
}
const Cache::PerfStats &perf_stats() const {
return perf_stats_;
}
// Flush API.
// flush_begin() arms the bank; subsequent ticks scan all sets/ways and emit
// a writeback request for every dirty line via mem_req_out (write-back only;
// write-through caches have nothing to evict). flush_done() reports when the
// walk finishes AND the cache is otherwise idle.
void flush_begin() {
if (!config_.write_back) {
flushing_ = false;
flush_set_idx_ = 0;
flush_way_idx_ = 0;
flush_sector_idx_ = 0;
return;
}
flushing_ = true;
flush_set_idx_ = 0;
flush_way_idx_ = 0;
flush_sector_idx_ = 0;
this->tick_wake();
}
bool flush_done() const {
return !flushing_;
}
protected:
void on_reset() {
perf_stats_ = Cache::PerfStats();
pending_mshr_size_ = 0;
pending_amo_probes_ = 0;
pipe_fill_count_ = 0;
pipe_core_lines_.clear();
pending_read_reqs_ = 0;
pending_write_reqs_ = 0;
pending_fill_reqs_ = 0;
rand_ctr_ = 0;
adm_seq_ctr_ = 0;
for (auto &set : sets_) {
set.reset();
}
mshr_.reset();
fwd_active_ = false;
fwd_set_ = 0;
fwd_tag_ = 0;
fwd_sector_ = 0;
crsp_sent_ = false;
flushing_ = false;
flush_set_idx_ = 0;
flush_way_idx_ = 0;
flush_sector_idx_ = 0;
#if VX_CFG_EXT_A_ENABLED
amo_unit_.reset();
for (auto &e : amo_passthru_) {
e = amo_passthru_entry_t{};
}
#endif
}
void on_tick() {
// The core-response port takes one response per tick; the pipeline has
// priority and the forward drain yields (crsp_sent_).
crsp_sent_ = false;
// Process next request at the head of the pipeline.
if (!pipe_req_->empty()) {
this->processRequests();
}
// Fill-forward drain: complete the read-prefix of a fill's pending chain
// directly (one per tick), leaving the input slot free for new requests.
// On response backpressure the head falls back to the replay path.
bool fwd_consumed = false;
if (fwd_active_) {
fwd_consumed = this->processForward();
}
// Accept one new input if there's room.
if (!pipe_req_->full()) {
this->processInputs(fwd_consumed);
}
// flush walk: emit writebacks for dirty lines.
if (flushing_) {
this->processFlush();
}
// calculate memory latency
perf_stats_.mem_latency += pending_fill_reqs_;
// sleep when fully drained: no queued or in-flight input, an empty pipe,
// and no replay/forward/flush work. pending_fill_reqs_ == 0 also keeps
// the mem_latency accumulation exact. flush_begin() re-arms explicitly;
// any packet reserved toward core_req_in/mem_rsp_in re-arms on arrival.
if (this->core_req_in.size() == 0 && this->mem_rsp_in.size() == 0
&& pipe_req_->size() == 0 && !mshr_.has_ready_reqs()
&& !fwd_active_ && !flushing_ && pending_fill_reqs_ == 0) {
this->tick_sleep();
}
}
private:
// Pipeline front: per-tick input arbitration.
//
// Priority (highest → lowest):
// 1) replay — drains an already-marked Replay entry from the MSHR
// (guaranteed hit; no miss path)
// 2) fill — memory response; installs the line and marks pending
// MSHR entries for replay (gated on no pending replay so
// a fill never preempts an in-flight replay's line)
// 3) flush — handled by processFlush() (separate state machine)
// 4) core_req — new core request (may miss and allocate an MSHR slot)
//
// At most one input fires per tick. All inputs flow through pipe_req_.
void processInputs(bool fwd_consumed) {
// 1) replay: one chain entry leaves per tick — when the fill-forward
// drain completed the head this tick, this slot goes to new core
// requests instead
if (mshr_.has_ready_reqs() && !fwd_consumed) {
bank_req_t bank_req;
mshr_.dequeue(&bank_req);
pipe_req_->push(bank_req);
DT(3, this->name() << " replay-deq: " << bank_req);
return;
}
// 2) fill only when no replay is pending and no forward drain is in
// progress (a new fill would re-arm the window mid-chain). A fill still
// riding the pipe also blocks the next one: its chain is only marked for
// replay at install, and the replay priority above must get a chance to
// drain it before another fill is accepted.
if (!this->mem_rsp_in.empty() && !fwd_active_ && pipe_fill_count_ == 0) {
auto &mem_rsp = this->mem_rsp_in.peek();
#if VX_CFG_EXT_A_ENABLED
// Non-LLC AMO passthrough response: forward straight to core
// without filling. The original tag was rewritten to
// (mshr_capacity + pid) when the bank emitted the request from
// the AmoProbe handler — that namespace partition survives
// arbiter tag mangling.
if (mem_rsp.tag >= amo_passthru_tag_base()
&& mem_rsp.tag < amo_passthru_tag_base() + AMO_PASSTHRU_CAP) {
uint32_t pid = mem_rsp.tag - amo_passthru_tag_base();
auto &e = amo_passthru_.at(pid);
assert(e.valid && "AMO passthru response without entry");
if (this->core_rsp_out.full()) {
return; // stall
}
MemRsp core_rsp{e.req_tag, e.cid, e.uuid};
core_rsp.data = mem_rsp.data;
this->core_rsp_out.send(core_rsp);
DT(3, this->name() << " amo-passthru-rsp: " << core_rsp);
e.valid = false;
this->mem_rsp_in.pop();
--pending_fill_reqs_;
return;
}
#endif
uint32_t mshr_id = mem_rsp.tag;
const auto &root_peek = mshr_.peek(mshr_id);
bank_req_t bank_req;
bank_req.reset();
bank_req.type = bank_req_t::Fill;
bank_req.addr = params_.mem_addr_sector(bank_id_, root_peek.set_id, root_peek.addr_tag, root_peek.sector_id);
bank_req.hart_id = root_peek.bank_req.hart_id;
bank_req.uuid = root_peek.bank_req.uuid;
bank_req.mshr_id = mshr_id;
bank_req.data = mem_rsp.data;
pipe_req_->push(bank_req);
++pipe_fill_count_;
DT(3, this->name() << " fill-rsp: " << mem_rsp);
this->mem_rsp_in.pop();
--pending_fill_reqs_;
return;
}
// 3) core request
if (!this->core_req_in.empty()) {
auto &core_req = this->core_req_in.peek();
// Conservative MSHR occupancy check: any request that may miss must
// reserve a slot. Counts both currently-allocated entries and
// in-flight pipe requests that haven't reached MSHR allocation yet.
// AMO requests always need a return; at the LLC they reserve like
// a load; at non-LLC they don't fill so no MSHR slot is needed
// but a passthru side-table slot is.
const bool is_amo = memop_is_amo(core_req.op);
#if VX_CFG_EXT_A_ENABLED
const bool is_amo_passthru = is_amo && !config_.is_llc;
if (is_amo_passthru) {
// Reserve a passthru-table slot at admission, counting probes still
// in the pipe. A probe that reaches the pipe head with no free slot
// cannot retire, and a full pipe blocks the response processing that
// frees slots — a deadlock the input gate must make unreachable.
uint32_t free_slots = 0;
for (const auto &e : amo_passthru_) { if (!e.valid) ++free_slots; }
if (free_slots <= pending_amo_probes_) {
++perf_stats_.mshr_stalls;
return;
}
// Age-ordering: if a fill is in flight for this AMO's line, an
// older load/store missed on it and is awaiting its replay. Defer
// the AMO until that access completes — otherwise the AMO's
// self-invalidation removes the line after the fill installs it
// but before the replay (which re-queues behind this AMO) reads
// it, turning the replay into a miss.
uint32_t amo_set_id = params_.addr_set_id(core_req.addr);
uint64_t amo_addr_tag = params_.addr_tag(core_req.addr);
uint32_t amo_sector = params_.addr_sector_id(core_req.addr);
if (mshr_.lookup(amo_set_id, amo_addr_tag, amo_sector)) {
++perf_stats_.mshr_stalls;
return;
}
// An older same-line access still inside the bank pipe allocates its
// MSHR entry only at pipe exit; defer on those too, or the AMO
// overtakes it and a younger same-address load can chain onto the
// pre-AMO fill (and the fill installs a pre-AMO line).
for (const auto &kv : pipe_core_lines_) {
if (kv.first == amo_set_id && kv.second == amo_addr_tag) {
++perf_stats_.mshr_stalls;
return;
}
}
}
#else
const bool is_amo_passthru = false;
#endif
// Gate ALL non-AMO-passthru requests on MSHR occupancy because a
// write-through store can need a wt-merge MSHR slot if a fill is
// pending on the same line. Letting writes in past a full MSHR
// causes them to stall on the wt-merge enqueue, blocking fill
// responses from draining — a deadlock. Reserve at admission.
bool needs_mshr = !is_amo_passthru;
(void)is_amo;
if (needs_mshr && (mshr_.size() + pending_mshr_size_) >= mshr_.capacity()) {
++perf_stats_.mshr_stalls;
return;
}
// Admission-time coalescing: a request whose (set,tag,sector) chain is
// still awaiting its fill must join it here, not enter the pipe as an
// independent Core request. Deciding at pipe exit is too late — the
// chain's fill can ride the pipe ahead of this request and install the
// line, and the chain's replays dequeue (clearing their entries) while
// this request is still in the pipe. A chain that is merely DRAINING
// (its fill already installed the line) does not capture new arrivals:
// the line is resident, so they proceed through the pipe as ordinary
// hits, completing ahead of the drain. The exception is a draining
// chain that still carries a write or atomic — a younger access must
// not overtake those, so it joins the chain and drains in seq order.
// Write-through stores are exempt: their downstream write must be
// emitted at the commit stage so same-line stores reach memory in
// pipeline order (an at-admission emission could overtake an older
// store still riding the pipe). The pipe-exit paths order them
// correctly and defer only the line merge when a chain is pending.
const bool is_wt_store = (core_req.op == MemOp::ST) && !config_.write_back;
if (!is_amo_passthru && !is_wt_store && mshr_.size() != 0) {
uint32_t adm_set_id = params_.addr_set_id(core_req.addr);
uint64_t adm_addr_tag = params_.addr_tag(core_req.addr);
uint32_t adm_sector = params_.addr_sector_id(core_req.addr);
uint32_t chain_root = 0;
if (mshr_.lookup(adm_set_id, adm_addr_tag, adm_sector, &chain_root)) {
bool fill_pending = mshr_.has_pending_fill(adm_set_id, adm_addr_tag, adm_sector, -1);
bool must_order = !fill_pending
&& mshr_.has_ordered_reqs(adm_set_id, adm_addr_tag, adm_sector);
if (fill_pending || must_order) {
bank_req_t coalesced;
coalesced.reset();
coalesced.adm_seq = adm_seq_ctr_++;
coalesced.type = bank_req_t::Core;
coalesced.addr = core_req.addr;
coalesced.hart_id = core_req.hart_id;
coalesced.uuid = core_req.uuid;
coalesced.req_tag = core_req.tag;
coalesced.write = (core_req.op == MemOp::ST);
coalesced.op = core_req.op;
coalesced.data = core_req.data;
coalesced.byteen = core_req.byteen;
coalesced.amo_cmp = core_req.amo_cmp;
coalesced.flags = core_req.flags;
assert(!mshr_.full());
int id = mshr_.enqueue(coalesced, adm_set_id, adm_addr_tag, adm_sector);
// Drain-only join (ordering case): nothing will mark our entry —
// make it replay-ready ourselves, seq-ordered behind the
// in-flight replays.
if (!fill_pending) {
mshr_.defer_to_replay(id);
}
DT(3, this->name() << " mshr-coalesce@admission: " << coalesced);
// Chain joiners count as misses, matching the pipe-exit chaining
// they replace.
if (core_req.is_write()) {
++perf_stats_.writes;
++perf_stats_.write_misses;
} else {
++perf_stats_.reads;
++perf_stats_.read_misses;
}
this->core_req_in.pop();
return;
}
}
}
bank_req_t bank_req;
bank_req.reset();
bank_req.adm_seq = adm_seq_ctr_++;
bank_req.type = is_amo_passthru ? bank_req_t::AmoProbe : bank_req_t::Core;
bank_req.addr = core_req.addr;
bank_req.hart_id = core_req.hart_id;
bank_req.uuid = core_req.uuid;
bank_req.req_tag = core_req.tag;
// bank_req.write means "route through the cache write path" — true only
// for plain stores. AMOs (op == AMO_SC / AMO RMW) are write-bearing
// semantically (memop_is_write returns true) but use the dedicated AMO
// commit path, not the write-path; keep bank_req.write false for them.
bank_req.write = (core_req.op == MemOp::ST);
bank_req.op = core_req.op;
bank_req.data = core_req.data;
bank_req.byteen = core_req.byteen;
bank_req.amo_cmp = core_req.amo_cmp;
bank_req.flags = core_req.flags;
pipe_req_->push(bank_req);
DT(3, this->name() << " core-req: " << core_req);
// pending_mshr_size_ tracks Core-typed in-flight requests so the
// MSHR pre-reservation in processInputs is conservative. AmoProbe
// doesn't allocate an MSHR slot (no fill on the response), so it
// stays out of this counter.
if (!is_amo_passthru) {
++pending_mshr_size_;
pipe_core_lines_.emplace_back(params_.addr_set_id(core_req.addr),
params_.addr_tag(core_req.addr));
}
#if VX_CFG_EXT_A_ENABLED
if (is_amo_passthru) ++pending_amo_probes_;
#endif
if (core_req.is_write()) ++perf_stats_.writes;
else ++perf_stats_.reads;
this->core_req_in.pop();
return;
}