-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathcache.h
More file actions
508 lines (415 loc) · 15.2 KB
/
Copy pathcache.h
File metadata and controls
508 lines (415 loc) · 15.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
//
// Created by 徐向荣 on 2022/6/12.
//
#ifndef MICRO_GPUSIM_C_CACHE_H
#define MICRO_GPUSIM_C_CACHE_H
#define GPM_NUM 4
//#define g_mod 1
#include <string>
#include <map>
#include <bitset>
#include <list>
#include <utility>
#include <vector>
#include <queue>
#include <climits>
#include <iostream>
#include "mem_fetch.h"
#include "tlb.h"
#include "config_reader.h"
#define MAX_DEFAULT_CACHE_SIZE_MULTIBLIER 4
#define MAX_WARP_PER_SM 1 << 6
#define SECTOR_CHUNCK_SIZE 4
#define SECTOR_SIZE 32
class gpu;
class memory_partition;
class streaming_multiprocessor;
unsigned int LOGB2(unsigned int v);
enum set_index_function {
LINEAR_SET_FUNCTION = 0,
BITWISE_XORING_FUNCTION,
HASH_IPOLY_FUNCTION
};
enum cache_block_state {
INVALID = 0, RESERVED, VALID, MODIFIED
};
enum cache_request_status {
HIT = 0,
HIT_RESERVED,
MISS,
RESERVATION_FAIL,
SECTOR_MISS,
NUM_CACHE_REQUEST_STATUS
};
struct evicted_block_info {
new_addr_type m_block_addr;
unsigned m_modified_size;
evicted_block_info() {
m_block_addr = 0;
m_modified_size = 0;
}
void set_info(new_addr_type block_addr, unsigned modified_size) {
m_block_addr = block_addr;
m_modified_size = modified_size;
}
};
class mshr_table {
public:
mshr_table(unsigned num_entries, unsigned max_merged)
: m_num_entries(num_entries),
m_max_merged(max_merged) {
}
bool probe(new_addr_type block_addr) const;
bool full(new_addr_type block_addr) const;
void add(new_addr_type block_addr, mem_fetch *mf);
mem_fetch *next_access(new_addr_type address);
private:
// finite sized, fully associative table, with a finite maximum number of
// merged requests
const unsigned m_num_entries;
const unsigned m_max_merged;
struct mshr_entry {
std::list<mem_fetch *> m_list;
bool m_has_atomic;
mshr_entry() : m_has_atomic(false) {}
};
typedef std::map<new_addr_type, mshr_entry> table;
table m_data;
std::list<new_addr_type> m_current_response;
};
class cache_config {
public:
cache_config(bool secter_cache, int n_banks, int n_sets, int line_size, int assoc,
const std::string &replacement_policy,
const std::string &write_policy, const std::string &alloc_policy,
const std::string &write_alloc_policy,
const std::string &set_index_function, const std::string &mshr_type, int mshr_entries,
int mshr_max_merge,
bool is_streaming) {
m_n_banks = n_banks;
m_n_sets = n_sets;
m_line_size = line_size;
m_assoc = assoc;
m_replacement_policy = replacement_policy;
m_write_policy = write_policy;
m_alloc_policy = alloc_policy;
m_write_alloc_policy = write_alloc_policy;
if (set_index_function == "LINEAR_SET_FUNCTION") {
m_set_index_function = LINEAR_SET_FUNCTION;
} else if (set_index_function == "HASH_IPOLY_FUNCTION") {
m_set_index_function = HASH_IPOLY_FUNCTION;
} else {
abort();
}
m_mshr_type = mshr_type;
m_mshr_entries = mshr_entries;
m_mshr_max_merge = mshr_max_merge;
m_is_streaming = is_streaming;
m_sector_cache = secter_cache;
if (m_alloc_policy == "STREAMING") {
m_alloc_policy = "ON FILL";
m_mshr_entries = n_sets * m_assoc * MAX_DEFAULT_CACHE_SIZE_MULTIBLIER * SECTOR_CHUNCK_SIZE;
m_mshr_max_merge = MAX_WARP_PER_SM;
}
m_atom_size = m_sector_cache ? SECTOR_SIZE : m_line_size;
}
new_addr_type tag(new_addr_type addr) const {
return addr & ~(new_addr_type) (m_line_size - 1);
}
new_addr_type block_addr(new_addr_type addr) const {
return addr & ~(new_addr_type) (m_line_size - 1);
}
new_addr_type mshr_addr(new_addr_type addr) const {
return addr & ~(new_addr_type) (m_atom_size - 1);
}
unsigned set_index(new_addr_type addr) const {
if(m_set_index_function == LINEAR_SET_FUNCTION){
return hash_function(addr, m_n_sets, LOGB2(m_line_size), LOGB2(m_n_sets)); //linear
}
else if(m_set_index_function == HASH_IPOLY_FUNCTION){
new_addr_type higher_bits = addr >> (LOGB2(m_line_size) + LOGB2(m_n_sets));
unsigned index = (addr >> LOGB2(m_line_size)) & (m_n_sets - 1);
std::bitset<64> a(higher_bits);
std::bitset<6> b(index);
std::bitset<6> new_index(index);
new_index[0] = a[18] ^ a[17] ^ a[16] ^ a[15] ^ a[12] ^ a[10] ^ a[6] ^ a[5] ^
a[0] ^ b[0];
new_index[1] = a[15] ^ a[13] ^ a[12] ^ a[11] ^ a[10] ^ a[7] ^ a[5] ^ a[1] ^
a[0] ^ b[1];
new_index[2] = a[16] ^ a[14] ^ a[13] ^ a[12] ^ a[11] ^ a[8] ^ a[6] ^ a[2] ^
a[1] ^ b[2];
new_index[3] = a[17] ^ a[15] ^ a[14] ^ a[13] ^ a[12] ^ a[9] ^ a[7] ^ a[3] ^
a[2] ^ b[3];
new_index[4] = a[18] ^ a[16] ^ a[15] ^ a[14] ^ a[13] ^ a[10] ^ a[8] ^ a[4] ^
a[3] ^ b[4];
new_index[5] =
a[17] ^ a[16] ^ a[15] ^ a[14] ^ a[11] ^ a[9] ^ a[5] ^ a[4] ^ b[5];
return new_index.to_ulong();
}
else{
}
}
unsigned hash_function(new_addr_type addr, unsigned m_nset,
unsigned m_line_sz_log2,
unsigned m_n_set_log2) const;
unsigned get_max_num_lines() const {
return m_n_sets * m_assoc;
}
bool m_sector_cache;
int m_n_banks;
int m_n_sets;
int m_line_size;
int m_assoc;
std::string m_replacement_policy;
std::string m_write_policy;
std::string m_alloc_policy;
std::string m_write_alloc_policy;
int m_set_index_function;
std::string m_mshr_type;
int m_mshr_entries;
int m_mshr_max_merge;
bool m_is_streaming;
int m_atom_size;
};
class sector_cache_block { //cache line
public:
sector_cache_block() {
m_tag = 0;
m_block_addr = 0;
init();
}
void init() {
for (unsigned i = 0; i < SECTOR_CHUNCK_SIZE; ++i) {
m_sector_alloc_time[i] = 0; //sector alloc time
m_sector_fill_time[i] = 0; //sector fill time
m_last_sector_access_time[i] = 0;
m_status[i] = INVALID;
m_readable[i] = true;
}
m_line_alloc_time = 0;
m_line_last_access_time = 0;
m_line_fill_time = 0;
}
void allocate(new_addr_type tag, new_addr_type block_addr,
unsigned time, unsigned sector_mask) {
allocate_line(tag, block_addr, time, sector_mask);
}
void allocate_line(new_addr_type tag, new_addr_type block_addr, unsigned time,
unsigned sector_mask) {
init();
m_tag = tag;
m_block_addr = block_addr;
m_sector_alloc_time[sector_mask] = time;
m_last_sector_access_time[sector_mask] = time;
m_sector_fill_time[sector_mask] = 0;
m_status[sector_mask] = RESERVED;
m_line_alloc_time = time; // only set this for the first allocated sector
m_line_last_access_time = time;
m_line_fill_time = 0;
}
void allocate_sector(unsigned time, unsigned sector_mask) {
m_sector_alloc_time[sector_mask] = time;
m_last_sector_access_time[sector_mask] = time;
m_sector_fill_time[sector_mask] = 0;
m_status[sector_mask] = RESERVED;
m_readable[sector_mask] = true;
m_line_last_access_time = time;
m_line_fill_time = 0;
}
void fill(unsigned time, unsigned sector_mask) {
m_status[sector_mask] = VALID;
m_sector_fill_time[sector_mask] = time;
m_line_fill_time = time;
}
bool is_invalid_line() {
// all the sectors should be invalid
for (auto &m_statu: m_status) {
if (m_statu != INVALID) return false;
}
return true;
}
bool is_valid_line() { return !(is_invalid_line()); }
bool is_reserved_line() {
for (auto &m_statu: m_status) {
if (m_statu == RESERVED) return true;
}
return false;
}
bool is_modified_line() {
for (auto &m_statu: m_status) {
if (m_statu == MODIFIED) return true;
}
return false;
}
enum cache_block_state get_status(
unsigned sector_mask) {
return m_status[sector_mask];
}
void set_status(enum cache_block_state status,
unsigned sector_mask) {
m_status[sector_mask] = status;
}
unsigned long long get_last_access_time() const {
return m_line_last_access_time;
}
void set_last_access_time(unsigned long long time,
unsigned sector_mask) {
m_last_sector_access_time[sector_mask] = time;
m_line_last_access_time = time;
}
unsigned long long get_alloc_time() const { return m_line_alloc_time; }
void set_modified_on_fill(bool m_modified,
unsigned sector_mask) {
m_set_modified_on_fill[sector_mask] = m_modified;
}
void set_m_readable(bool readable,
unsigned sector_mask) {
m_readable[sector_mask] = readable;
}
bool is_readable(unsigned sector_mask) {
return m_readable[sector_mask];
}
new_addr_type m_tag;
new_addr_type m_block_addr;
unsigned m_sector_alloc_time[SECTOR_CHUNCK_SIZE];
unsigned m_last_sector_access_time[SECTOR_CHUNCK_SIZE];
unsigned m_sector_fill_time[SECTOR_CHUNCK_SIZE];
unsigned m_line_alloc_time;
unsigned m_line_last_access_time;
unsigned m_line_fill_time;
cache_block_state m_status[SECTOR_CHUNCK_SIZE];
bool m_set_modified_on_fill[SECTOR_CHUNCK_SIZE];
bool m_readable[SECTOR_CHUNCK_SIZE];
};
class tag_array {
public:
tag_array(const cache_config &config, unsigned core_id) : m_config(config) {
unsigned cache_lines_num = config.get_max_num_lines();
m_lines = new sector_cache_block *[cache_lines_num];
for (unsigned i = 0; i < cache_lines_num; ++i)
m_lines[i] = new sector_cache_block();
m_core_id = core_id; //sm_id
}
~tag_array() {
unsigned cache_lines_num = m_config.get_max_num_lines();
for (unsigned i = 0; i < cache_lines_num; ++i) delete m_lines[i];
delete[] m_lines;
};
unsigned
tag_array_probe(new_addr_type addr, unsigned sector_mask, mem_fetch *mem_fetch, unsigned &idx);
void tag_array_probe_idle(new_addr_type addr, unsigned int sector_mask, mem_fetch *mem_fetch,
int &idx);
unsigned tag_array_access(new_addr_type addr, unsigned time, mem_fetch *mem_fetch, unsigned &idx);
unsigned tag_array_access(new_addr_type addr, unsigned time, mem_fetch *mem_fetch, unsigned &idx, bool &wb);
void tag_array_fill(new_addr_type addr, unsigned time, unsigned mask);
sector_cache_block *get_block(unsigned idx) const { return m_lines[idx]; }
cache_config m_config;
sector_cache_block **m_lines; /* all_set x assoc lines in total */ //then map bank
unsigned m_core_id;
int m_type_id;
typedef std::map<new_addr_type, unsigned> line_table;
line_table pending_lines;
};
class data_cache {
public:
data_cache(const cache_config& cache_config, gpu_config &gpu_config, unsigned sm_id,gpu* gpu) :
m_cache_config(cache_config),
m_gpu_config(gpu_config),
sm_id(sm_id){
m_gpu = gpu;
m_tag_array = new tag_array(cache_config, sm_id);
m_mshrs = new mshr_table(cache_config.m_mshr_entries, cache_config.m_mshr_max_merge);
}
cache_config m_cache_config;
gpu_config m_gpu_config;
unsigned sm_id;
// std::vector<mem_fetch *> m_input_request_queue;
// std::vector<mem_fetch *> m_output_request_queue;
// std::vector<mem_fetch *> m_wait_fill;
tag_array *m_tag_array;
mshr_table *m_mshrs;
gpu* m_gpu;
int m_write_cache_access_num = 0;
int m_read_cache_access_num = 0;
int m_load_sector_hit = 0;
int m_store_sector_hit = 0;
int m_load_sector_miss = 0;
int m_store_sector_miss = 0;
int m_sector_load_to_next = 0;
int m_sector_store_to_next = 0;
static int sm_miss_to_l2_list[256];
static int l2_pop_list[256];
};
class l1_data_cache : public data_cache { //found l1
public:
l1_data_cache(cache_config &cache_config, gpu_config &gpu_config, unsigned sm_id,
std::map<int, std::queue<mem_fetch *>> &sm_to_l1, gpu* gpu) :
data_cache(cache_config, gpu_config, sm_id, gpu), m_sm_to_l1(sm_to_l1), m_gpu(gpu) {
m_l1_tlb = new tlb(32 * 1024 * 1024, LONG_LONG_MAX);
m_l1_to_l2_num = 0;
m_l1_to_l2_mshr = 0;
}
tlb *m_l1_tlb;
std::map<int, std::queue<mem_fetch *>> &m_sm_to_l1; // bank id: mf
std::queue<mem_fetch *> m_miss_queue;
std::queue<mem_fetch* > l1_to_sm; // mf: time
int m_l1_to_l2_num;
int m_l1_to_l2_mshr;
gpu* m_gpu;
void l1_cache_cycle(int cycles);
void m_miss_queue_cycle();
unsigned cache_access(new_addr_type addr, mem_fetch *mf, unsigned time);
void cache_fill(mem_fetch *mf, unsigned time);
// void new_l1_to_sm_Push(mem_fetch *mf);
};
class l2_data_cache : public data_cache {
public:
l2_data_cache(cache_config &cache_config, gpu_config &gpu_config, int m_partition_id,gpu * gpu) :
data_cache(cache_config, gpu_config, 999,gpu) {
m_device_id = std::stoi(gpu_config.m_gpu_config["sm_num"]) + m_partition_id;
}
void init(memory_partition* mem){
m_mem_partition = mem;
}
void l2_cache_cycle(int cycles);
void send_l2_to_sm(mem_fetch *mf, int latency) {
l2_to_sm.emplace_back(mf, latency);
}
void send_l2_to_dram(mem_fetch *mf, int latency) {
l2_to_dram.emplace_back(mf, latency);
}
unsigned cache_access(new_addr_type addr, mem_fetch *mf, unsigned time);
void cache_fill(mem_fetch *mf, unsigned time);
int m_device_id; // mem partition id
std::queue<mem_fetch *> m_l1_to_l2; // l1 to l2
std::vector<std::pair<mem_fetch *, int>> l2_to_sm;
std::vector<std::pair<mem_fetch *, int>> l2_to_dram;
memory_partition* m_mem_partition;
static std::vector<int> g_l2_list;
// static int a[44];
};
class memory_partition {
public:
memory_partition(cache_config &cache_config, gpu_config &gpu_config):
m_cache_config(cache_config),
m_gpu_config(gpu_config){
gpm_n_mem = std::stoi(m_gpu_config.m_gpu_config["mem_num"]) / std::stoi(m_gpu_config
.m_gpu_config["gpm_num"]);
}
void memory_partition_cycle(int cycles){
int access_num = 0;
for (int i = 0;i < gpm_n_mem; i++) {
m_l2_caches[i]->l2_cache_cycle(cycles);
access_num += m_l2_caches[i]->m_read_cache_access_num;
}
// std::cout<<"l2_access_num "<<access_num<<std::endl;
}
cache_config m_cache_config;
gpu_config m_gpu_config; //主gpm设置
unsigned gpm_n_mem;
std::vector<l2_data_cache *> m_l2_caches;
gpu* m_gpu;
void init(gpu* gpu, int n);
// static std::vector<int> g_l2_list;
};
#endif //MICRO_GPUSIM_C_CACHE_H