aboutsummaryrefslogtreecommitdiff
path: root/configs/4.x-cfgs/SM2_GTX480
diff options
context:
space:
mode:
authorMahmoud <[email protected]>2018-08-27 20:28:24 -0400
committerMahmoud <[email protected]>2018-08-27 20:28:24 -0400
commit944f6dbf23d792dde360d3a4f2334de3b541de52 (patch)
treed1a257a87769ef52f2d72a4cd62147e591d5cd97 /configs/4.x-cfgs/SM2_GTX480
parent5e7bd910c07c066d5d1cc4b12f8aa7abefcdb411 (diff)
fixing ead/write buffer and new configs files
Diffstat (limited to 'configs/4.x-cfgs/SM2_GTX480')
-rw-r--r--configs/4.x-cfgs/SM2_GTX480/config_fermi_islip.icnt4
-rw-r--r--configs/4.x-cfgs/SM2_GTX480/gpgpusim.config23
2 files changed, 20 insertions, 7 deletions
diff --git a/configs/4.x-cfgs/SM2_GTX480/config_fermi_islip.icnt b/configs/4.x-cfgs/SM2_GTX480/config_fermi_islip.icnt
index 7820e4e..d372b26 100644
--- a/configs/4.x-cfgs/SM2_GTX480/config_fermi_islip.icnt
+++ b/configs/4.x-cfgs/SM2_GTX480/config_fermi_islip.icnt
@@ -1,6 +1,6 @@
//21*1 fly with 32 flits per packet under gpgpusim injection mode
use_map = 0;
-flit_size = 32;
+flit_size = 40;
// currently we do not use this, see subnets below
network_count = 2;
@@ -17,7 +17,7 @@ routing_function = dest_tag;
// Flow control
num_vcs = 1;
-vc_buf_size = 8;
+vc_buf_size = 64;
wait_for_tail_credit = 0;
diff --git a/configs/4.x-cfgs/SM2_GTX480/gpgpusim.config b/configs/4.x-cfgs/SM2_GTX480/gpgpusim.config
index 03fcda1..7f8da49 100644
--- a/configs/4.x-cfgs/SM2_GTX480/gpgpusim.config
+++ b/configs/4.x-cfgs/SM2_GTX480/gpgpusim.config
@@ -50,20 +50,25 @@
# <nsets>:<bsize>:<assoc>,<rep>:<wr>:<alloc>:<wr_alloc>:<set_index_fn>,<mshr>:<N>:<merge>,<mq>:**<fifo_entry>
# ** Optional parameter - Required when mshr_type==Texture Fifo
# Note: Hashing set index function (H) only applies to a set size of 32 or 64.
--gpgpu_cache:dl1 N:32:128:4,L:L:m:N:H,A:32:8,8
+-gpgpu_cache:dl1 N:32:128:4,L:L:m:N:H,S:128:8,8
-gpgpu_shmem_size 49152
+-icnt_flit_size 40
+-gpgpu_n_cluster_ejection_buffer_size 32
# The alternative configuration for fermi in case cudaFuncCachePreferL1 is selected
-#-gpgpu_cache:dl1 N:64:128:6,L:L:m:N:H,A:32:8,8
+#-gpgpu_cache:dl1 N:64:128:6,L:L:m:N:H,S:32:8,8
#-gpgpu_shmem_size 16384
# 64 sets, each 128 bytes 8-way for each memory sub partition. This gives 786KB L2 cache
--gpgpu_cache:dl2 N:64:128:8,L:B:m:W:L,A:32:4,4:0,32
+-gpgpu_cache:dl2 S:64:128:8,L:B:m:L:L,A:256:4,4:0,32
-gpgpu_cache:dl2_texture_only 0
+-gpgpu_dram_partition_queues 64:64:64:64
+-perf_sim_memcpy 0
+-memory_partition_indexing 0
--gpgpu_cache:il1 N:4:128:4,L:R:f:N:L,A:2:32,4
+-gpgpu_cache:il1 N:4:128:4,L:R:f:N:L,S:2:32,4
-gpgpu_tex_cache:l1 N:4:128:24,L:R:m:N:L,F:128:4,128:2
--gpgpu_const_cache:l1 N:64:64:2,L:R:f:N:L,A:2:32,4
+-gpgpu_const_cache:l1 N:64:64:2,L:R:f:N:L,S:2:32,4
# enable operand collector
-gpgpu_operand_collector_num_units_sp 6
@@ -76,6 +81,7 @@
-gpgpu_shmem_num_banks 32
-gpgpu_shmem_limited_broadcast 0
-gpgpu_shmem_warp_parts 1
+-gpgpu_coalesce_arch 20
-gpgpu_max_insn_issue_per_warp 1
@@ -110,6 +116,13 @@
-gpgpu_dram_timing_opt "nbk=16:CCD=2:RRD=6:RCD=12:RAS=28:RP=12:RC=40:
CL=12:WL=4:CDLR=5:WR=12:nbkgrp=4:CCDL=3:RTPL=2"
+# select lower bits for bnkgrp to increase bnkgrp parallelism
+-dram_bnk_indexing_policy 0
+-dram_bnkgrp_indexing_policy 1
+
+#-Seperate_Write_Queue_Enable 1
+#-Write_Queue_Size 64:56:32
+
# Fermi has two schedulers per core
-gpgpu_num_sched_per_core 2
# Two Level Scheduler with active and pending pools