# Invocation command line: # /root/hpc2021v1.1.10/bin/harness/runhpc --reportable --configfile H100.cfg --tune base --pmodel ACC --define model=acc --threads 1 --ranks 4 --size ref --iterations 3 --nopower --runmode speed --tune base --size ref small # output_root was not used for this run ############################################################################ # Invocation command line: # /home/HPC2021v1.1.7/bin/harness/runhpc --reportable --configfile nv_example.cfg --tune base,peak --pmodel ACC --define model=acc --define THREADS=1 --ranks 8 --size ref --iterations 3 --nopower --runmode speed --tune base:peak --size ref tiny # output_root was not used for this run ############################################################################ ###################################################################### # Example configuration file for the NVIDIA HPC SDK Compilers # # Before using this config file, copy it to a new config (such as nvhpc.cfg) and edit as needed # # Defines: "model" => "mpi", "acc", "accmc", "omp", "tgt", "tgtgpu" default "mpi" # "label" => ext base label, default "nv" # # MPI-only Command: # runhpc -c nvhpc --reportable -T base --define model=mpi --ranks=40 tiny # # OpenACC offload to GPU Command: # runhpc -c nvhpc --reportable -T base --define model=acc --ranks=4 tiny # Add "--define ucx" if using OpenMPI 4 with UCX support. # # OpenACC offload to Multicore CPU Command: # runhpc -c nvhpc --reportable -T base --define model=accmc --ranks=4 tiny # # OpenMP Command: # runhpc -c nvhpc --reportable -T base --define model=omp --ranks=1 --threads=40 tiny # # OpenMP Target Offload to Host Command: # runhpc -c nvhpc --reportable -T base --define model=tgt --ranks=1 --threads=40 tiny # # OpenMP Target Offload to GPU Command: # runhpc -c nvhpc --reportable -T base --define model=tgtgpu --ranks=4 tiny # ####################################################################### # The following setting was inserted automatically as a result of # post-run basepeak application. basepeak = 1 %ifndef %{label} # IF label is not set use nv % define label nv %endif %ifndef %{model} # IF model is not set use mpi % define model mpi pmodel = MPI %endif teeout = yes # Display the Internal Timer info # Adjust the number of make jobs to use here makeflags=-j 40 flagsurl000=http://www.spec.org/hpc2021/flags/nv2021_flags_v1.0.3.2026-09-16.xml # Tester Information license_num = 28 showtimer = 0 test_sponsor = Lenovo Global Technology tester = Lenovo Global Technology ###################################################### # SUT Section ###################################################### #include: Example_SUT.inc # ----- Begin inclusion of 'Example_SUT.inc' ############################################################################ ###################################################### # Example configuration information for a # system under test (SUT) Section ###################################################### # General SUT info system_vendor = Lenovo Global Technology system_name000 = ThinkSystem SC777 V4(Arm Neoverse V2, Nvidia system_name001 = GB200) node_compute_sw_accel_driver = 610.43.02 hw_avail = Oct-2026 sw_avail = Oct-2026 prepared_by = Lenovo Global Technology # Computation node info # [Node_Description: Hardware] node_compute_syslbl = ThinkSystem SC777 V4 node_compute_order = 1 node_compute_count = 1 node_compute_purpose = compute node_compute_hw_vendor = Lenovo Global Technology node_compute_hw_model = ThinkSystem SC777 V4 node_compute_hw_cpu_name = NVIDIA Grace Neoverse V2 CPU node_compute_hw_ncpuorder = 2 chips node_compute_hw_nchips = 2 node_compute_hw_ncores = 144 node_compute_hw_ncoresperchip = 72 node_compute_hw_nthreadspercore = 1 node_compute_hw_cpu_char = Grace CPU @3.3GHz node_compute_hw_cpu_mhz = 3300 node_compute_hw_pcache = 64 KB I + 64 KB D on chip per core node_compute_hw_scache = 1 MB I+D on chip per core node_compute_hw_tcache = 117 MB I+D on chip per chip node_compute_hw_ocache = None node_compute_hw_memory = 960 GB (2x 480 GB LPDDR5X) node_compute_hw_disk = 1x ThinkSystem 2.5" 5300 480GB SSD node_compute_hw_other = None #[Node_Description: Accelerator] node_compute_hw_accel_model = NVIDIA GB200 node_compute_hw_accel_count = 4 node_compute_hw_accel_vendor= NVIDIA Corporation node_compute_hw_accel_type = GPU node_compute_hw_accel_connect = NVLink-C2C node_compute_hw_accel_ecc = Yes node_compute_hw_accel_desc = NVIDIA Grace Blackwell GB200 #[Node_Description: Software] node_compute_hw_adapter_fs_model = N/A node_compute_hw_adapter_fs_count = 1 node_compute_hw_adapter_fs_slot_type = N/A node_compute_hw_adapter_fs_data_rate = N/A node_compute_hw_adapter_fs_ports_used = 0 node_compute_hw_adapter_fs_interconnect = N/A node_compute_hw_adapter_fs_driver = N/A node_compute_hw_adapter_fs_firmware = N/A node_compute_sw_os000 = Ubuntu 24.04.4 LTS, node_compute_sw_os001 = Kernel 6.17.0-1018-nvidia-64k node_compute_sw_localfile = xfs node_compute_sw_sharedfile = XFS node_compute_sw_state = Multi-user, run level 3 node_compute_sw_other = None #[Fileserver] #[Interconnect] ####################################################################### # End of SUT section # If this config file were to be applied to several SUTs, edits would # be needed only ABOVE this point. ####################################################################### ---- End inclusion of '/home/HPC2021v1.1.7/config/Example_SUT.inc' #[Software] system_class = Homogeneous Cluster sw_compiler = NVIDIA HPC SDK 25.11 sw_mpi_library = Open MPI 4.1.9a1 sw_mpi_other = None sw_other = -- #[General notes] ####################################################################### # End of SUT section ###################################################################### ###################################################################### # The header section of the config file. Must appear # before any instances of "section markers" (see below) # # ext = how the binaries you generated will be identified # tune = specify "base" or "peak" or "all" label = %{label}_%{model} tune = base output_format = text use_submit_for_speed = 1 # Setting 'strict_rundir_verify=0' will allow direct source code modifications # but will disable the ability to create reportable results. # May be useful for academic and research purposes # strict_rundir_verify = 0 # Compiler Settings default: CC = mpicc CXX = mpicxx FC = mpif90 # Compiler Version Flags CC_VERSION_OPTION = -V CXX_VERSION_OPTION = -V FC_VERSION_OPTION = -V # if using OpenMPI with UCX support, these settings are needed with use of CUDA Aware MPI # without these flags, LBM is known to hang when using OpenACC and OpenMP Target to GPUs preENV_UCX_MEMTYPE_CACHE=n preENV_UCX_TLS=self,sm,cuda_copy,tcp preENV_UCX_NET_DEVICES = all ENV_OMPI_MCA_pml=ucx ENV_OMPI_MCA_topo=basic ENV_UCX_LOG_LEVEL=error ENV_OMPI_MCA_coll=^hcoll ENV_CUDA_CACHE_DISABLE=1 ENV_HCOLL_BUFFER_POOL_MEM_PER_NODE=1024Mb ENV_RETRY_COUNT=1000 ENV_UCX_RNDV_SCHEME=get_zcopy ENV_UCX_RNDV_THRESH=8192 ENV_UCX_MAX_RNDV_RAILS=1 # --- 針對單機 4 GPU 的綁定 (以 OpenMPI 為例) --- # 確保 4 個 rank 平分 4 顆 GPU # Adjust to match your system # Note that SPH_EXA is known to hang when using multiple nodes with some versions of UCX, # to work around, add the following setting: #MPIRUN_OPTS += --mca topo basic --bind-to core --mca coll_hcoll_enable 0 --mca pml ob1 --mca btl self,vader,tcp MPIRUN_OPTS += --mca topo basic %ifdef %{bindomp} # use the example bindomp.pl script #submit = mpirun --allow-run-as-root ${MPIRUN_OPTS} -np $ranks command submit = mpirun --allow-run-as-root ${MPIRUN_OPTS} -np $ranks $[top]/config/scripts/gpuwrap.sh $command %else #submit = mpirun --allow-run-as-root ${MPIRUN_OPTS} -np $ranks $command submit = mpirun --allow-run-as-root ${MPIRUN_OPTS} -np $ranks $[top]/config/scripts/gpuwrap.sh $command %endif ####################################################################### # Optimization # # Note that SPEC baseline rules require that all uses of a given compiler # use the same flags in the same order. See the SPEChpc Run Rules # for more details # http://www.spec.org/hpc2021/Docs/runrules.html # # OPTIMIZE = flags applicable to all compilers # FOPTIMIZE = flags appliable to the Fortran compiler # COPTIMIZE = flags appliable to the C compiler # CXXOPTIMIZE = flags appliable to the C++ compiler # # See your compiler manual for information on the flags available # for your compiler # # Compiler flags applied to all models default=base=default: OPTIMIZE = -w -Mfprelaxed -Mnouniform -Mstack_arrays -fast CXXPORTABILITY = --c++17 # OpenACC (GPU) flags %if %{model} eq 'acc' #pmodel=ACC #OPTIMIZE += -acc=gpu -Minfo=accel -DSPEC_ACCEL_AWARE_MPI pmodel=ACC OPTIMIZE = -w -O4 -DSPEC_ACCEL_AWARE_MPI -acc=gpu -gpu=ccnative -Mfprelaxed -Mnouniform -tp=host CXXPORTABILITY = --c++17 505.lbm_t,605.lbm_s,705.lbm_m,805.lbm_l: PORTABILITY += -DSPEC_OPENACC_NO_SELF %endif # OpenACC (Multicore CPU) flags %if %{model} eq 'accmc' pmodel=ACC OPTIMIZE += -acc=multicore -mp -Minfo=accel 505.lbm_t,605.lbm_s,705.lbm_m,805.lbm_l: PORTABILITY += -DSPEC_OPENACC_NO_SELF 521.miniswp_t: PORTABILITY+= -DSPEC_USE_HOST_THREADS=1 %endif # OpenMP Threaded (CPU) flags %if %{model} eq 'omp' pmodel=OMP OPTIMIZE += -mp -Minfo=mp %endif # OpenMP Targeting host flags %if %{model} eq 'tgt' pmodel=TGT OPTIMIZE += -mp -Minfo=mp # Note that while NVHPC added support for OpenMP # array reduction in v22.2, a compiler issue # prevents it's use. This may be used in future # versions of the compiler, in which case remove # -DSPEC_NO_VAR_ARRAY_REDUCE 513.soma_t,613.soma_s: PORTABILITY+=-DSPEC_NO_VAR_ARRAY_REDUCE -DSPEC_USE_HOST_THREADS=1 521.miniswp_t: PORTABILITY+=-DSPEC_USE_HOST_THREADS=1 %endif # OpenMP Targeting GPU flags %if %{model} eq 'tgtgpu' pmodel=TGT OPTIMIZE += -mp=gpu -Minfo=mp # Note that while NVHPC added support for OpenMP # array reduction in v22.2, a compiler issue # prevents it's use. This may be used in future # versions of the compiler, in which case comment # out the following two lines 513.soma_t,613.soma_s: PORTABILITY+=-DSPEC_NO_VAR_ARRAY_REDUCE %endif # No peak flags set, so make peak use the same flags as base default=peak=default: basepeak=0 505.lbm_t=peak=default: ranks = %{RANKS} OPTIMIZE = -w -fast -acc=gpu -O3 -Mfprelaxed -Mnouniform -DSPEC_ACCEL_AWARE_MPI 513.soma_t=peak=default: basepeak=1 518.tealeaf_t=peak=default: ranks = %{RANKS} OPTIMIZE = -w -fast -acc=gpu -Msafeptr -DSPEC_ACCEL_AWARE_MPI 519.clvleaf_t=peak=default: basepeak=1 521.miniswp_t=peak=default: basepeak=1 528.pot3d_t=peak=default: basepeak=1 532.sph_exa_t=peak=default: ranks = %{RANKS} OPTIMIZE = -w -fast -acc=gpu -O3 -Mfprelaxed -Mnouniform -Mstack_arrays -static-nvidia -DSPEC_ACCEL_AWARE_MPI 534.hpgmgfv_t=peak=default: ranks = %{RANKS} OPTIMIZE = -w -fast -acc=gpu -static-nvidia -DSPEC_ACCEL_AWARE_MPI 535.weather_t=peak=default: ranks = %{RANKS} OPTIMIZE = -w -fast -acc=gpu -O3 -Mfprelaxed -Mnouniform -Mstack_arrays -static-nvidia -DSPEC_ACCEL_AWARE_MPI # The following section was added automatically, and contains settings that # did not appear in the original configuration file, but were added to the # raw file after the run. default: notes_000 =Environment variables set by runhpc before the start of the run: notes_005 =UCX_MEMTYPE_CACHE = "n" notes_010 =UCX_NET_DEVICES = "all" notes_015 =UCX_TLS = "self,sm,cuda_copy,tcp"