perf vendor events power9: Cpi_breakdown & estimated_dcache_miss_cpi metrics
Descriptions of metrics for POWER9 processors can be found in the "POWER9 Performance Monitor Unit User’s Guide", which is currently available on the "IBM Portal for OpenPOWER" (https://www-355.ibm.com/systems/power/openpower/welcome.xhtml) at https://www-355.ibm.com/systems/power/openpower/posting.xhtml?postingId=4948CDE1963C9BCA852582F800718190 This patch is for metric groups: - cpi_breakdown - estimated_dcache_miss_cpi Signed-off-by: Paul Clarke <pc@us.ibm.com> Cc: Ananth N Mavinakayanahalli <ananth@linux.vnet.ibm.com> Cc: Carl Love <cel@us.ibm.com> Cc: Madhavan Srinivasan <maddy@linux.vnet.ibm.com> Cc: Michael Ellerman <mpe@ellerman.id.au> Cc: Naveen N. Rao <naveen.n.rao@linux.vnet.ibm.com> Cc: Sukadev Bhattiprolu <sukadev@linux.vnet.ibm.com> Cc: linuxppc-dev@ozlabs.org Link: http://lkml.kernel.org/r/20190209181429.23950-2-pc@us.ibm.com Signed-off-by: Arnaldo Carvalho de Melo <acme@redhat.com>
This commit is contained in:
parent
72ab50203f
commit
7f3cf5ac77
577
tools/perf/pmu-events/arch/powerpc/power9/metrics.json
Normal file
577
tools/perf/pmu-events/arch/powerpc/power9/metrics.json
Normal file
@ -0,0 +1,577 @@
|
||||
[
|
||||
{
|
||||
"BriefDescription": "Completion stall due to a Branch Unit",
|
||||
"MetricExpr": "PM_CMPLU_STALL_BRU/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "bru_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was routed to the crypto execution pipe and was waiting to finish",
|
||||
"MetricExpr": "PM_CMPLU_STALL_CRYPTO/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "crypto_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a load that missed the L1 and was waiting for the data to return from the nest",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DCACHE_MISS/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dcache_miss_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a multi-cycle instruction issued to the Decimal Floating Point execution pipe and waiting to finish.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DFLONG/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dflong_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Stalls due to short latency decimal floating ops.",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_DFU - PM_CMPLU_STALL_DFLONG)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dfu_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was issued to the Decimal Floating Point execution pipe and waiting to finish.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DFU/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dfu_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall by Dcache miss which resolved off node memory/cache",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_DMISS_L3MISS - PM_CMPLU_STALL_DMISS_L21_L31 - PM_CMPLU_STALL_DMISS_LMEM - PM_CMPLU_STALL_DMISS_REMOTE)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_distant_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall by Dcache miss which resolved on chip ( excluding local L2/L3)",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DMISS_L21_L31/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_l21_l31_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to cache miss that resolves in the L2 or L3 with a conflict",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DMISS_L2L3_CONFLICT/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_l2l3_conflict_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to cache miss that resolves in the L2 or L3 without conflict",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_DMISS_L2L3 - PM_CMPLU_STALL_DMISS_L2L3_CONFLICT)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_l2l3_noconflict_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall by Dcache miss which resolved in L2/L3",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DMISS_L2L3/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_l2l3_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to cache miss resolving missed the L3",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DMISS_L3MISS/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_l3miss_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to cache miss that resolves in local memory",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DMISS_LMEM/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_lmem_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall by Dcache miss which resolved outside of local memory",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_DMISS_L3MISS - PM_CMPLU_STALL_DMISS_L21_L31 - PM_CMPLU_STALL_DMISS_LMEM)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_non_local_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall by Dcache miss which resolved from remote chip (cache or memory)",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DMISS_REMOTE/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dmiss_remote_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Stalls due to short latency double precision ops.",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_DP - PM_CMPLU_STALL_DPLONG)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dp_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a scalar instruction issued to the Double Precision execution pipe and waiting to finish. Includes binary floating point instructions in 32 and 64 bit binary floating point format.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DP/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dp_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a scalar multi-cycle instruction issued to the Double Precision execution pipe and waiting to finish. Includes binary floating point instructions in 32 and 64 bit binary floating point format.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_DPLONG/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "dplong_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction is an EIEIO waiting for response from L2",
|
||||
"MetricExpr": "PM_CMPLU_STALL_EIEIO/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "eieio_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the next to finish instruction suffered an ERAT miss and the EMQ was full",
|
||||
"MetricExpr": "PM_CMPLU_STALL_EMQ_FULL/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "emq_full_stall_cpi"
|
||||
},
|
||||
{
|
||||
"MetricExpr": "(PM_CMPLU_STALL_ERAT_MISS + PM_CMPLU_STALL_EMQ_FULL)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "emq_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a load or store that suffered a translation miss",
|
||||
"MetricExpr": "PM_CMPLU_STALL_ERAT_MISS/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "erat_miss_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Cycles in which the NTC instruction is not allowed to complete because it was interrupted by ANY exception, which has to be serviced before the instruction can complete",
|
||||
"MetricExpr": "PM_CMPLU_STALL_EXCEPTION/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "exception_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to execution units for other reasons.",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_EXEC_UNIT - PM_CMPLU_STALL_FXU - PM_CMPLU_STALL_DP - PM_CMPLU_STALL_DFU - PM_CMPLU_STALL_PM - PM_CMPLU_STALL_CRYPTO - PM_CMPLU_STALL_VFXU - PM_CMPLU_STALL_VDP)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "exec_unit_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to execution units (FXU/VSU/CRU)",
|
||||
"MetricExpr": "PM_CMPLU_STALL_EXEC_UNIT/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "exec_unit_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Cycles in which the NTC instruction is not allowed to complete because any of the 4 threads in the same core suffered a flush, which blocks completion",
|
||||
"MetricExpr": "PM_CMPLU_STALL_FLUSH_ANY_THREAD/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "flush_any_thread_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to a long latency scalar fixed point instruction (division, square root)",
|
||||
"MetricExpr": "PM_CMPLU_STALL_FXLONG/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "fxlong_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Stalls due to short latency integer ops",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_FXU - PM_CMPLU_STALL_FXLONG)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "fxu_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall due to a scalar fixed point or CR instruction in the execution pipeline. These instructions get routed to the ALU, ALU2, and DIV pipes",
|
||||
"MetricExpr": "PM_CMPLU_STALL_FXU/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "fxu_stall_cpi"
|
||||
},
|
||||
{
|
||||
"MetricExpr": "(PM_NTC_ISSUE_HELD_DARQ_FULL + PM_NTC_ISSUE_HELD_ARB + PM_NTC_ISSUE_HELD_OTHER)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "issue_hold_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a larx waiting to be satisfied",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LARX/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "larx_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a load that hit on an older store and it was waiting for store data",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LHS/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lhs_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a load that missed in the L1 and the LMQ was unable to accept this load miss request because it was full",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LMQ_FULL/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lmq_full_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a load instruction with all its dependencies satisfied just going through the LSU pipe to finish",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LOAD_FINISH/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "load_finish_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a load that was held in LSAQ because the LRQ was full",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LRQ_FULL/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lrq_full_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall due to LRQ miscellaneous reasons, lost arbitration to LMQ slot, bank collisions, set prediction cleanup, set prediction multihit and others",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LRQ_OTHER/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lrq_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"MetricExpr": "(PM_CMPLU_STALL_LMQ_FULL + PM_CMPLU_STALL_ST_FWD + PM_CMPLU_STALL_LHS + PM_CMPLU_STALL_LSU_MFSPR + PM_CMPLU_STALL_LARX + PM_CMPLU_STALL_LRQ_OTHER)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lrq_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a load or store that was held in LSAQ because an older instruction from SRQ or LRQ won arbitration to the LSU pipe when this instruction tried to launch",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LSAQ_ARB/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lsaq_arb_stall_cpi"
|
||||
},
|
||||
{
|
||||
"MetricExpr": "(PM_CMPLU_STALL_LRQ_FULL + PM_CMPLU_STALL_SRQ_FULL + PM_CMPLU_STALL_LSAQ_ARB)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lsaq_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was an LSU op (other than a load or a store) with all its dependencies met and just going through the LSU pipe to finish",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LSU_FIN/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lsu_fin_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall of one cycle because the LSU requested to flush the next iop in the sequence. It takes 1 cycle for the ISU to process this request before the LSU instruction is allowed to complete",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LSU_FLUSH_NEXT/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lsu_flush_next_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a mfspr instruction targeting an LSU SPR and it was waiting for the register data to be returned",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LSU_MFSPR/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lsu_mfspr_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion LSU stall for other reasons",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_LSU - PM_CMPLU_STALL_LSU_FIN - PM_CMPLU_STALL_STORE_FINISH - PM_CMPLU_STALL_STORE_DATA - PM_CMPLU_STALL_EIEIO - PM_CMPLU_STALL_STCX - PM_CMPLU_STALL_SLB - PM_CMPLU_STALL_TEND - PM_CMPLU_STALL_PASTE - PM_CMPLU_STALL_TLBIE - PM_CMPLU_STALL_STORE_PIPE_ARB - PM_CMPLU_STALL_STORE_FIN_ARB - PM_CMPLU_STALL_LOAD_FINISH + PM_CMPLU_STALL_DCACHE_MISS - PM_CMPLU_STALL_LMQ_FULL - PM_CMPLU_STALL_ST_FWD - PM_CMPLU_STALL_LHS - PM_CMPLU_STALL_LSU_MFSPR - PM_CMPLU_STALL_LARX - PM_CMPLU_STALL_LRQ_OTHER + PM_CMPLU_STALL_ERAT_MISS + PM_CMPLU_STALL_EMQ_FULL - PM_CMPLU_STALL_LRQ_FULL - PM_CMPLU_STALL_SRQ_FULL - PM_CMPLU_STALL_LSAQ_ARB) / PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lsu_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall by LSU instruction",
|
||||
"MetricExpr": "PM_CMPLU_STALL_LSU/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "lsu_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall because the ISU is updating the register and notifying the Effective Address Table (EAT)",
|
||||
"MetricExpr": "PM_CMPLU_STALL_MTFPSCR/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "mtfpscr_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall because the ISU is updating the TEXASR to keep track of the nested tbegin. This is a short delay, and it includes ROT",
|
||||
"MetricExpr": "PM_CMPLU_STALL_NESTED_TBEGIN/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "nested_tbegin_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall because the ISU is updating the TEXASR to keep track of the nested tend and decrement the TEXASR nested level. This is a short delay",
|
||||
"MetricExpr": "PM_CMPLU_STALL_NESTED_TEND/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "nested_tend_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Number of cycles the ICT has no itags assigned to this thread",
|
||||
"MetricExpr": "PM_ICT_NOSLOT_CYC/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "nothing_dispatched_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was one that must finish at dispatch.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_NTC_DISP_FIN/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "ntc_disp_fin_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Cycles in which the oldest instruction in the pipeline (NTC) finishes. This event is used to account for cycles in which work is being completed in the CPI stack",
|
||||
"MetricExpr": "PM_NTC_FIN/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "ntc_fin_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to ntc flush",
|
||||
"MetricExpr": "PM_CMPLU_STALL_NTC_FLUSH/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "ntc_flush_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "The NTC instruction is being held at dispatch because it lost arbitration onto the issue pipe to another instruction (from the same thread or a different thread)",
|
||||
"MetricExpr": "PM_NTC_ISSUE_HELD_ARB/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "ntc_issue_held_arb_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "The NTC instruction is being held at dispatch because there are no slots in the DARQ for it",
|
||||
"MetricExpr": "PM_NTC_ISSUE_HELD_DARQ_FULL/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "ntc_issue_held_darq_full_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "The NTC instruction is being held at dispatch during regular pipeline cycles, or because the VSU is busy with multi-cycle instructions, or because of a write-back collision with VSU",
|
||||
"MetricExpr": "PM_NTC_ISSUE_HELD_OTHER/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "ntc_issue_held_other_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Cycles unaccounted for.",
|
||||
"MetricExpr": "(PM_RUN_CYC - PM_1PLUS_PPC_CMPL - PM_CMPLU_STALL_THRD - PM_CMPLU_STALL - PM_ICT_NOSLOT_CYC)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "other_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall for other reasons",
|
||||
"MetricExpr": "PM_CMPLU_STALL - PM_CMPLU_STALL_NTC_DISP_FIN - PM_CMPLU_STALL_NTC_FLUSH - PM_CMPLU_STALL_LSU - PM_CMPLU_STALL_EXEC_UNIT - PM_CMPLU_STALL_BRU)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a paste waiting for response from L2",
|
||||
"MetricExpr": "PM_CMPLU_STALL_PASTE/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "paste_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was issued to the Permute execution pipe and waiting to finish.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_PM/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "pm_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Run cycles per run instruction",
|
||||
"MetricExpr": "PM_RUN_CYC / PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "run_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Run_cycles",
|
||||
"MetricExpr": "PM_RUN_CYC/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "run_cyc_cpi"
|
||||
},
|
||||
{
|
||||
"MetricExpr": "(PM_CMPLU_STALL_FXU + PM_CMPLU_STALL_DP + PM_CMPLU_STALL_DFU + PM_CMPLU_STALL_PM + PM_CMPLU_STALL_CRYPTO)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "scalar_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was awaiting L2 response for an SLB",
|
||||
"MetricExpr": "PM_CMPLU_STALL_SLB/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "slb_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall while waiting for the non-speculative finish of either a stcx waiting for its result or a load waiting for non-critical sectors of data and ECC",
|
||||
"MetricExpr": "PM_CMPLU_STALL_SPEC_FINISH/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "spec_finish_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a store that was held in LSAQ because the SRQ was full",
|
||||
"MetricExpr": "PM_CMPLU_STALL_SRQ_FULL/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "srq_full_stall_cpi"
|
||||
},
|
||||
{
|
||||
"MetricExpr": "(PM_CMPLU_STALL_STORE_DATA + PM_CMPLU_STALL_EIEIO + PM_CMPLU_STALL_STCX + PM_CMPLU_STALL_SLB + PM_CMPLU_STALL_TEND + PM_CMPLU_STALL_PASTE + PM_CMPLU_STALL_TLBIE + PM_CMPLU_STALL_STORE_PIPE_ARB + PM_CMPLU_STALL_STORE_FIN_ARB)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "srq_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to store forward",
|
||||
"MetricExpr": "PM_CMPLU_STALL_ST_FWD/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "st_fwd_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Nothing completed and ICT not empty",
|
||||
"MetricExpr": "PM_CMPLU_STALL/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a stcx waiting for response from L2",
|
||||
"MetricExpr": "PM_CMPLU_STALL_STCX/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "stcx_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the next to finish instruction was a store waiting on data",
|
||||
"MetricExpr": "PM_CMPLU_STALL_STORE_DATA/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "store_data_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a store waiting for a slot in the store finish pipe. This means the instruction is ready to finish but there are instructions ahead of it, using the finish pipe",
|
||||
"MetricExpr": "PM_CMPLU_STALL_STORE_FIN_ARB/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "store_fin_arb_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a store with all its dependencies met, just waiting to go through the LSU pipe to finish",
|
||||
"MetricExpr": "PM_CMPLU_STALL_STORE_FINISH/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "store_finish_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a store waiting for the next relaunch opportunity after an internal reject. This means the instruction is ready to relaunch and tried once but lost arbitration",
|
||||
"MetricExpr": "PM_CMPLU_STALL_STORE_PIPE_ARB/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "store_pipe_arb_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a tend instruction awaiting response from L2",
|
||||
"MetricExpr": "PM_CMPLU_STALL_TEND/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "tend_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion Stalled because the thread was blocked",
|
||||
"MetricExpr": "PM_CMPLU_STALL_THRD/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "thread_block_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a tlbie waiting for response from L2",
|
||||
"MetricExpr": "PM_CMPLU_STALL_TLBIE/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "tlbie_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Vector stalls due to small latency double precision ops",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_VDP - PM_CMPLU_STALL_VDPLONG)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "vdp_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a vector instruction issued to the Double Precision execution pipe and waiting to finish.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_VDP/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "vdp_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall because the NTF instruction was a scalar multi-cycle instruction issued to the Double Precision execution pipe and waiting to finish. Includes binary floating point instructions in 32 and 64 bit binary floating point format.",
|
||||
"MetricExpr": "PM_CMPLU_STALL_VDPLONG/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "vdplong_stall_cpi"
|
||||
},
|
||||
{
|
||||
"MetricExpr": "(PM_CMPLU_STALL_VFXU + PM_CMPLU_STALL_VDP)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "vector_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Completion stall due to a long latency vector fixed point instruction (division, square root)",
|
||||
"MetricExpr": "PM_CMPLU_STALL_VFXLONG/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "vfxlong_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Vector stalls due to small latency integer ops",
|
||||
"MetricExpr": "(PM_CMPLU_STALL_VFXU - PM_CMPLU_STALL_VFXLONG)/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "vfxu_other_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "Finish stall due to a vector fixed point instruction in the execution pipeline. These instructions get routed to the ALU, ALU2, and DIV pipes",
|
||||
"MetricExpr": "PM_CMPLU_STALL_VFXU/PM_RUN_INST_CMPL",
|
||||
"MetricGroup": "cpi_breakdown",
|
||||
"MetricName": "vfxu_stall_cpi"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl2l3 distant MOD miss rates with measured DL2L3 MOD latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_DL2L3_MOD * PM_MRK_DATA_FROM_DL2L3_MOD_CYC / PM_MRK_DATA_FROM_DL2L3_MOD / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "dl2l3_mod_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl2l3 distant SHR miss rates with measured DL2L3 SHR latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_DL2L3_SHR * PM_MRK_DATA_FROM_DL2L3_SHR_CYC / PM_MRK_DATA_FROM_DL2L3_SHR / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "dl2l3_shr_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of distant L4 miss rates with measured DL4 latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_DL4 * PM_MRK_DATA_FROM_DL4_CYC / PM_MRK_DATA_FROM_DL4 / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "dl4_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of distant memory miss rates with measured DMEM latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_DMEM * PM_MRK_DATA_FROM_DMEM_CYC / PM_MRK_DATA_FROM_DMEM / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "dmem_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl21 MOD miss rates with measured L21 MOD latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_L21_MOD * PM_MRK_DATA_FROM_L21_MOD_CYC / PM_MRK_DATA_FROM_L21_MOD / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "l21_mod_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl21 SHR miss rates with measured L21 SHR latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_L21_SHR * PM_MRK_DATA_FROM_L21_SHR_CYC / PM_MRK_DATA_FROM_L21_SHR / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "l21_shr_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl2 miss rates with measured L2 latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_L2 * PM_MRK_DATA_FROM_L2_CYC / PM_MRK_DATA_FROM_L2 / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "l2_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl31 MOD miss rates with measured L31 MOD latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_L31_MOD * PM_MRK_DATA_FROM_L31_MOD_CYC / PM_MRK_DATA_FROM_L31_MOD / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "l31_mod_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl31 SHR miss rates with measured L31 SHR latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_L31_SHR * PM_MRK_DATA_FROM_L31_SHR_CYC / PM_MRK_DATA_FROM_L31_SHR / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "l31_shr_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl3 miss rates with measured L3 latency as a % of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_L3 * PM_MRK_DATA_FROM_L3_CYC / PM_MRK_DATA_FROM_L3 / PM_CMPLU_STALL_DCACHE_MISS * 100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "l3_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of Local memory miss rates with measured LMEM latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_LMEM * PM_MRK_DATA_FROM_LMEM_CYC / PM_MRK_DATA_FROM_LMEM / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "lmem_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl2l3 remote MOD miss rates with measured RL2L3 MOD latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_RL2L3_MOD * PM_MRK_DATA_FROM_RL2L3_MOD_CYC / PM_MRK_DATA_FROM_RL2L3_MOD / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "rl2l3_mod_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of dl2l3 shared miss rates with measured RL2L3 SHR latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_RL2L3_SHR * PM_MRK_DATA_FROM_RL2L3_SHR_CYC / PM_MRK_DATA_FROM_RL2L3_SHR / PM_CMPLU_STALL_DCACHE_MISS * 100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "rl2l3_shr_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of remote L4 miss rates with measured RL4 latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_RL4 * PM_MRK_DATA_FROM_RL4_CYC / PM_MRK_DATA_FROM_RL4 / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "rl4_cpi_percent"
|
||||
},
|
||||
{
|
||||
"BriefDescription": "estimate of remote memory miss rates with measured RMEM latency as a %of dcache miss cpi",
|
||||
"MetricExpr": "PM_DATA_FROM_RMEM * PM_MRK_DATA_FROM_RMEM_CYC / PM_MRK_DATA_FROM_RMEM / PM_CMPLU_STALL_DCACHE_MISS *100",
|
||||
"MetricGroup": "estimated_dcache_miss_cpi",
|
||||
"MetricName": "rmem_cpi_percent"
|
||||
}
|
||||
]
|
Loading…
Reference in New Issue
Block a user