aboutsummaryrefslogtreecommitdiff
path: root/configs/4.x-cfgs/SM7_TITANV/gpgpusim.config
diff options
context:
space:
mode:
authorTimothy G Rogers <[email protected]>2018-10-11 14:16:33 -0400
committerGitHub Enterprise <[email protected]>2018-10-11 14:16:33 -0400
commit982d7e02ff64c8978d5635bbc2b3515e2145574b (patch)
tree04e71c63714464d17105f8ae2563a99365d821f8 /configs/4.x-cfgs/SM7_TITANV/gpgpusim.config
parenta43799f779a2cf23728659733649506a2d5420df (diff)
parentbeeafb66d2e2bb441ab1eacade75322a72961be0 (diff)
Merge pull request #28 from abdallm/dev-purdue-integration
Dev purdue integration
Diffstat (limited to 'configs/4.x-cfgs/SM7_TITANV/gpgpusim.config')
-rw-r--r--configs/4.x-cfgs/SM7_TITANV/gpgpusim.config23
1 files changed, 13 insertions, 10 deletions
diff --git a/configs/4.x-cfgs/SM7_TITANV/gpgpusim.config b/configs/4.x-cfgs/SM7_TITANV/gpgpusim.config
index 14faedb..6fe441b 100644
--- a/configs/4.x-cfgs/SM7_TITANV/gpgpusim.config
+++ b/configs/4.x-cfgs/SM7_TITANV/gpgpusim.config
@@ -27,9 +27,9 @@
#-gpgpu_clock_domains <Core Clock>:<Interconnect Clock>:<L2 Clock>:<DRAM Clock>
# Volta NVIDIA GV100 clock domains are adopted from
# https://en.wikipedia.org/wiki/Volta_(microarchitecture)
--gpgpu_clock_domains 1200.0:1200.0:2000.0:850.0
+-gpgpu_clock_domains 1200.0:2000.0:1200.0:850.0
# boost mode
-# -gpgpu_clock_domains 1455.0:1455.0:2000.0:850.0
+# -gpgpu_clock_domains 1455.0:2000.0:1455.0:850.0
# shader core pipeline config
-gpgpu_shader_registers 65536
@@ -65,18 +65,20 @@
# <nsets>:<bsize>:<assoc>,<rep>:<wr>:<alloc>:<wr_alloc>:<set_index_fn>,<mshr>:<N>:<merge>,<mq>:**<fifo_entry>
# ** Optional parameter - Required when mshr_type==Texture Fifo
-# Defualt config is 64KB DL1 and 64KB shared memory
--gpgpu_cache:dl1 S:4:128:128,L:L:s:N:L,A:256:8,16:0,32
--gpgpu_cache:dl1PrefL1 S:4:128:192,L:L:s:N:L,A:256:8,16:0,32
--gpgpu_cache:dl1PrefShared S:4:128:64,L:L:s:N:L,A:256:8,16:0,32
--gpgpu_shmem_size 65536
--gpgpu_shmem_size_PrefL1 32768
--gpgpu_shmem_size_PrefShared 98304
+# Defualt config is 32KB DL1 and 96KB shared memory
+# In Volta, we assign the remaining shared memory to L1 cache
+# if the assigned shd mem = 0, then L1 cache = 128KB
+# For more info, see https://docs.nvidia.com/cuda/cuda-c-programming-guide/index.html#shared-memory-7-x
+# disable this mode in case of multi kernels/apps execution
+-adpative_volta_cache_config 1
+-gpgpu_cache:dl1 S:4:128:64,L:L:s:N:L,A:256:8,16:0,32
+-gpgpu_shmem_size 98304
-gmem_skip_L1D 0
-icnt_flit_size 40
-gpgpu_n_cluster_ejection_buffer_size 32
-l1_latency 28
-smem_latency 19
+-gpgpu_flush_l1_cache 1
# 64 sets, each 128 bytes 24-way for each memory sub partition (192 KB per memory sub partition). This gives 4.5MB L2 cache
-gpgpu_cache:dl2 S:64:128:24,L:B:m:L:L,A:384:4,32:0,32
@@ -86,7 +88,8 @@
# 128 KB Inst.
-gpgpu_cache:il1 N:64:128:16,L:R:f:N:L,S:2:48,4
-# 48 KB Tex
+# 48 KB Tex
+# Note, TEX is deprected in Volta, It is used for legacy apps only. Use L1D cache instead with .nc modifier or __ldg mehtod
-gpgpu_tex_cache:l1 N:16:128:24,L:R:m:N:L,T:128:4,128:2
# 64 KB Const
-gpgpu_const_cache:l1 N:128:64:8,L:R:f:N:L,S:2:64,4