-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathgpu.h
More file actions
301 lines (234 loc) · 9.15 KB
/
Copy pathgpu.h
File metadata and controls
301 lines (234 loc) · 9.15 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
//
// Created by 刘浩 on 2022/6/14.
// class gpu sm
#ifndef FLEX_GPUSIM_GPU_H
#define FLEX_GPUSIM_GPU_H
#define GPM_NUM 4
#define g_mod 1
#include <utility>
#include <vector>
#include "config_reader.h"
#include "trace_reader.h"
#include "cache.h"
#include "mem_fetch.h"
#include "crossbar.h"
extern int request_id;
class streaming_multiprocessor;
class sm_unit;
class gpu {
public:
explicit gpu(std::string benchmark);
~gpu() {
delete traceReader;
delete m_active_kernel;
}
void init_config();
void build_gpu();
void creat_icnt(int i){
g_icnt_config.in_buffer_limit = m_gpu_configs.crossbar_config["crossbar_buffer_size"];
g_icnt_config.out_buffer_limit = m_gpu_configs.crossbar_config["crossbar_buffer_size"];
g_icnt_config.subnets = m_gpu_configs.crossbar_config["crossbar_subnets"];
g_icnt_config.arbiter_algo = NAIVE_RR;
m_icnt = LocalInterconnect::New(g_icnt_config);
m_icnt->CreateInterconnect(stoi(m_gpu_configs.m_gpu_config["sm_num"]),stoi(m_gpu_configs
.m_gpu_config["mem_num"]), i, stoi(m_gpu_configs.m_gpu_config["gpm_num"])); //new X-bar
for (auto x : m_gpu_configs.crossbar_config)
{
printf("%s %d \n", x.first.c_str(), x.second);
}
printf("%s %d \n", "gpm_num", stoi(m_gpu_configs.m_gpu_config["gpm_num"]));
}
void creat_m_icnts()
{
g_icnt_config.in_buffer_limit = 512;
g_icnt_config.out_buffer_limit = 512;
g_icnt_config.subnets = 2;
g_icnt_config.arbiter_algo = NAIVE_RR;
MCM_icnt = new class MCM_icnt::MCM_icnt(g_icnt_config, stoi(m_gpu_configs.m_gpu_config["sm_num"]), stoi
(m_gpu_configs.m_gpu_config["mem_num"]), stoi(m_gpu_configs.m_gpu_config["gpm_num"]), m_gpu_configs.crossbar_config); //初始化
}
void launch_kernel(int kernel_id);
void execute_kernel(int kernel_id);
void read_trace(int kernel_id, kernel_info &kernelInfo,
std::vector<trace_inst> &trace_insts);
void first_spawn_block();
void gpu_cycle();
void config_relevant_l1();
gpu_config m_gpu_configs;
struct inct_config g_icnt_config;
LocalInterconnect* m_icnt;
MCM_icnt* MCM_icnt;
std::string active_benchmark;
trace_reader* traceReader;
kernel* m_active_kernel;
std::vector<streaming_multiprocessor *> m_sm;
std::vector<std::vector<streaming_multiprocessor *> > g_sm;
// l2_data_cache* m_l2_data_cache;
memory_partition* m_memory_partition;
std::vector<memory_partition *> g_memory_partition;
// std::vector<int> g_l2_list;
std::map<int, std::queue<mem_fetch*>> m_queue_l1_to_l2; // bank id: mf
std::vector<std::pair<mem_fetch*, int>> l2_to_sm; // mf: time
unsigned sm_num;//n_shader
unsigned mem_num;//n_mem
unsigned gpm_num;
unsigned gpm_n_shader;
unsigned gpm_n_mem;
private:
int max_block_per_sm(kernel_info kernel_info);
static int max_block_limit_by_others();
int max_block_limit_by_warps(kernel_info kernelInfo);
int max_block_limit_by_regs(kernel_info kernelInfo);
int max_block_limit_by_smem(kernel_info kernelInfo);
int gpu_sim_cycles;
};
class streaming_multiprocessor {
public:
streaming_multiprocessor(unsigned sm_id, gpu_config &gpu_config, cache_config &cache_config)
: m_l1_cache_config(cache_config), m_gpu_config(gpu_config), cycles(0){
m_sm_id = sm_id;
is_active = false;
m_max_blocks = 0;
m_execute_inst = 0;
m_queue_sm_to_l1_busy.reset();
load_units();
gpm_num = std::stoi(m_gpu_config
.m_gpu_config["gpm_num"]);
sm_num = std::stoi(m_gpu_config.m_gpu_config["sm_num"]);
n_shader = sm_num / gpm_num;
// new_m_queue_sm_to_l1.resize(n_shader);
// new_m_queue_sm_to_l1_busy.resize(n_shader);
// for (auto & new_bitset : new_m_queue_sm_to_l1_busy)
// {
// new_bitset.reset();
// }
// m_queue_l1_to_sm_used = false;
//初始化
}
void init(gpu* gpu){
m_gpu = gpu;
// m_l1_data_cache = new l1_data_cache(m_l1_cache_config, m_gpu_config, m_sm_id, m_queue_sm_to_l1, gpu);
m_l1_data_cache = new l1_data_cache(m_l1_cache_config, m_gpu_config, m_sm_id, m_queue_sm_to_l1,
// new_m_queue_sm_to_l1[m_sm_id % n_shader],
gpu); //initialize // todo gqn: 感觉这里有点不对...?
}
// void config_relevant_l1(int sm_id)
int schedule_warp(std::vector<warp *> &warp_vec);
bool check_unit(warp &warp);
bool check_unit(int delay, const string &unit_name);
bool check_dependency(warp &);
bool check_warp_active(warp &warp);
bool check_at_barrier(warp &warp);
int issue_inst(warp &);
void sm_cycle();
void load_units();
bool sm_active() {
for (auto &block: block_vec) {
if (block->is_active(cycles)) {
is_active = true;
return is_active;
}
}
if (m_active_kernel->block_pointer < m_active_kernel->m_blocks.size()) {
is_active = true;
return is_active;
}
is_active = false;
return false;
}
void del_inactive_block();
void init_sm_to_l1() {
int n_banks = m_l1_cache_config.m_n_banks;
for (int i = 0; i < n_banks; i++) {
m_queue_sm_to_l1[i] = {};
}
}
block *get_block(int block_id) {
for (auto &block: block_vec) {
if (block->m_block_id == block_id) {
return block;
}
}
printf("ERROR %s:%d block:%d is not found!\n", __FILE__, __LINE__, block_id);
exit(1);
}
void write_back();
void write_back_request_l2(mem_fetch * mf);
cache_config m_l1_cache_config;
l1_data_cache *m_l1_data_cache; //暂时不改
// l2_data_cache *m_l2_data_cache;
// std::vector<l1_data_cache *> new_m_l1_data_cache;
// std::vector<std::queue<mem_fetch *>> new_m_queue_l1_to_sm;
// std::queue<mem_fetch *> new_m_queue_l1_to_sm;
// bool m_queue_l1_to_sm_used;
// research gpu structure again 先dev
gpu* m_gpu;
std::map<std::string, std::vector<sm_unit*> > m_units;
unsigned m_sm_id;
int m_max_blocks;
std::vector<block *> block_vec;
bool is_active;
gpu_config m_gpu_config;
int cycles;
kernel *m_active_kernel;
int m_execute_inst;
std::map<int, std::queue<mem_fetch *>> m_queue_sm_to_l1; // key: bank id value: mf queue
// std::vector<std::map<int, std::queue<mem_fetch *>>> new_m_queue_sm_to_l1;
// std::map<int, std::queue<mem_fetch *>> new_m_queue_sm_to_l1; // key: bank id value: mf queue
//sm_to_l1搞大点
std::bitset<4> m_queue_sm_to_l1_busy;
// std::vector<std::bitset<4>> new_m_queue_sm_to_l1_busy;
std::vector<mem_fetch *> m_response_fifo;
unsigned gpm_num;
unsigned sm_num;
unsigned n_shader;
// std::map<int, std::queue<mem_fetch *>> &m_queue_l1_to_l2; //to l2
// std::vector<std::pair<mem_fetch *, int>> &m_l2_to_sm;
// int block_warp_size = int(kernel.block_size / hardware_info.cc_configs['warp_size']);
// int max_blocks = max_blocks;
// vector<block> block_list; //1:block
// vector<block> wait_block; //正在launch的block key:block_id value:block
// kernel kernel = kernel;
};
class sm_unit{
public:
sm_unit(std::string name){
unit_name = std::move(name);
ready_time = 0;
mem_fetch_point = 0;
m_inst = nullptr;
m_warp = nullptr;
unit_latency = 0;
}
void set_sm(streaming_multiprocessor* sm){
m_sm = sm;
}
void set_ldst_inst(mem_inst* inst, warp* warp){
m_inst = inst;
m_warp = warp;
ready_time = INT_MAX;
mem_fetch_point = 0; //指向聚合地址的第n个请求
unit_latency = 4;
}
void unit_cycle(int cycles);
std::string unit_name;
int ready_time;
mem_inst* m_inst;
warp* m_warp;
int mem_fetch_point;
streaming_multiprocessor* m_sm;
int unit_latency;
};
class LdStInst{
public:
void process(gpu_config& gpu_config_t, sm_unit* unit, warp& warp, int cycles, long long int pc_t, int pc_index_t,
int active_thread_num_t, const string& opcode, unsigned sm_id_t,
std::map<int, std::queue<mem_fetch *>>& queue_sm_to_l1_t, cache_config& l1_cache_config_t, kernel* active_kernel_t);
void process_LDG_STG(gpu_config& gpu_config_t, sm_unit* unit, warp& warp, int cycles, long long int pc_t, int pc_index_t,
int active_thread_num_t, const string& opcode, unsigned sm_id_t, std::map<int, std::queue<mem_fetch *>>& queue_sm_to_l1_t, cache_config& l1_cache_config_t, kernel*active_kernel_t);
void process_LDS_STS_ATOMS(gpu_config& gpu_config_t, int active_thread_num_t, sm_unit* unit, warp& warp, int cycles);
void process_ATOM_ATOMG(sm_unit* unit, warp& warp, long long pc_t, int pc_index_t, unsigned sm_id_t, int cycles, std::map<int, std::queue<mem_fetch *>>& queue_sm_to_l1_t, cache_config& l1_cache_config_t);
};
int ceil(float x, float s);
int floor(float x, float s);
#endif //FLEX_GPUSIM_GPU_H