changeset 0a65922d564d in /z/repo/gem5
details: http://repo.gem5.org/gem5?cmd=changeset;node=0a65922d564d
description:
gpu-compute: add instruction mix stats for the gpu
diffstat:
src/gpu-compute/compute_unit.cc | 141 ++++++++++++++++++++++++++++++++++++++++
src/gpu-compute/compute_unit.hh | 25 +++++++
src/gpu-compute/wavefront.cc | 4 +
3 files changed, 170 insertions(+), 0 deletions(-)
diffs (208 lines):
diff -r c3b4d57a15c5 -r 0a65922d564d src/gpu-compute/compute_unit.cc
--- a/src/gpu-compute/compute_unit.cc Wed Oct 26 22:47:27 2016 -0400
+++ b/src/gpu-compute/compute_unit.cc Wed Oct 26 22:47:30 2016 -0400
@@ -1408,6 +1408,114 @@
{
MemObject::regStats();
+ vALUInsts
+ .name(name() + ".valu_insts")
+ .desc("Number of vector ALU insts issued.")
+ ;
+ vALUInstsPerWF
+ .name(name() + ".valu_insts_per_wf")
+ .desc("The avg. number of vector ALU insts issued per-wavefront.")
+ ;
+ sALUInsts
+ .name(name() + ".salu_insts")
+ .desc("Number of scalar ALU insts issued.")
+ ;
+ sALUInstsPerWF
+ .name(name() + ".salu_insts_per_wf")
+ .desc("The avg. number of scalar ALU insts issued per-wavefront.")
+ ;
+ instCyclesVALU
+ .name(name() + ".inst_cycles_valu")
+ .desc("Number of cycles needed to execute VALU insts.")
+ ;
+ instCyclesSALU
+ .name(name() + ".inst_cycles_salu")
+ .desc("Number of cycles needed to execute SALU insts.")
+ ;
+ threadCyclesVALU
+ .name(name() + ".thread_cycles_valu")
+ .desc("Number of thread cycles used to execute vector ALU ops. "
+ "Similar to instCyclesVALU but multiplied by the number of "
+ "active threads.")
+ ;
+ vALUUtilization
+ .name(name() + ".valu_utilization")
+ .desc("Percentage of active vector ALU threads in a wave.")
+ ;
+ ldsNoFlatInsts
+ .name(name() + ".lds_no_flat_insts")
+ .desc("Number of LDS insts issued, not including FLAT "
+ "accesses that resolve to LDS.")
+ ;
+ ldsNoFlatInstsPerWF
+ .name(name() + ".lds_no_flat_insts_per_wf")
+ .desc("The avg. number of LDS insts (not including FLAT "
+ "accesses that resolve to LDS) per-wavefront.")
+ ;
+ flatVMemInsts
+ .name(name() + ".flat_vmem_insts")
+ .desc("The number of FLAT insts that resolve to vmem issued.")
+ ;
+ flatVMemInstsPerWF
+ .name(name() + ".flat_vmem_insts_per_wf")
+ .desc("The average number of FLAT insts that resolve to vmem "
+ "issued per-wavefront.")
+ ;
+ flatLDSInsts
+ .name(name() + ".flat_lds_insts")
+ .desc("The number of FLAT insts that resolve to LDS issued.")
+ ;
+ flatLDSInstsPerWF
+ .name(name() + ".flat_lds_insts_per_wf")
+ .desc("The average number of FLAT insts that resolve to LDS "
+ "issued per-wavefront.")
+ ;
+ vectorMemWrites
+ .name(name() + ".vector_mem_writes")
+ .desc("Number of vector mem write insts (excluding FLAT insts).")
+ ;
+ vectorMemWritesPerWF
+ .name(name() + ".vector_mem_writes_per_wf")
+ .desc("The average number of vector mem write insts "
+ "(excluding FLAT insts) per-wavefront.")
+ ;
+ vectorMemReads
+ .name(name() + ".vector_mem_reads")
+ .desc("Number of vector mem read insts (excluding FLAT insts).")
+ ;
+ vectorMemReadsPerWF
+ .name(name() + ".vector_mem_reads_per_wf")
+ .desc("The avg. number of vector mem read insts (excluding "
+ "FLAT insts) per-wavefront.")
+ ;
+ scalarMemWrites
+ .name(name() + ".scalar_mem_writes")
+ .desc("Number of scalar mem write insts.")
+ ;
+ scalarMemWritesPerWF
+ .name(name() + ".scalar_mem_writes_per_wf")
+ .desc("The average number of scalar mem write insts per-wavefront.")
+ ;
+ scalarMemReads
+ .name(name() + ".scalar_mem_reads")
+ .desc("Number of scalar mem read insts.")
+ ;
+ scalarMemReadsPerWF
+ .name(name() + ".scalar_mem_reads_per_wf")
+ .desc("The average number of scalar mem read insts per-wavefront.")
+ ;
+
+ vALUInstsPerWF = vALUInsts / completedWfs;
+ sALUInstsPerWF = sALUInsts / completedWfs;
+ vALUUtilization = (threadCyclesVALU / (64 * instCyclesVALU)) * 100;
+ ldsNoFlatInstsPerWF = ldsNoFlatInsts / completedWfs;
+ flatVMemInstsPerWF = flatVMemInsts / completedWfs;
+ flatLDSInstsPerWF = flatLDSInsts / completedWfs;
+ vectorMemWritesPerWF = vectorMemWrites / completedWfs;
+ vectorMemReadsPerWF = vectorMemReads / completedWfs;
+ scalarMemWritesPerWF = scalarMemWrites / completedWfs;
+ scalarMemReadsPerWF = scalarMemReads / completedWfs;
+
tlbCycles
.name(name() + ".tlb_cycles")
.desc("total number of cycles for all uncoalesced requests")
@@ -1567,6 +1675,39 @@
}
void
+ComputeUnit::updateInstStats(GPUDynInstPtr gpuDynInst)
+{
+ if (gpuDynInst->isScalar()) {
+ if (gpuDynInst->isALU() && !gpuDynInst->isWaitcnt()) {
+ sALUInsts++;
+ instCyclesSALU++;
+ } else if (gpuDynInst->isLoad()) {
+ scalarMemReads++;
+ } else if (gpuDynInst->isStore()) {
+ scalarMemWrites++;
+ }
+ } else {
+ if (gpuDynInst->isALU()) {
+ vALUInsts++;
+ instCyclesVALU++;
+ threadCyclesVALU += gpuDynInst->wavefront()->execMask().count();
+ } else if (gpuDynInst->isFlat()) {
+ if (gpuDynInst->isLocalMem()) {
+ flatLDSInsts++;
+ } else {
+ flatVMemInsts++;
+ }
+ } else if (gpuDynInst->isLocalMem()) {
+ ldsNoFlatInsts++;
+ } else if (gpuDynInst->isLoad()) {
+ vectorMemReads++;
+ } else if (gpuDynInst->isStore()) {
+ vectorMemWrites++;
+ }
+ }
+}
+
+void
ComputeUnit::updatePageDivergenceDist(Addr addr)
{
Addr virt_page_addr = roundDown(addr, TheISA::PageBytes);
diff -r c3b4d57a15c5 -r 0a65922d564d src/gpu-compute/compute_unit.hh
--- a/src/gpu-compute/compute_unit.hh Wed Oct 26 22:47:27 2016 -0400
+++ b/src/gpu-compute/compute_unit.hh Wed Oct 26 22:47:30 2016 -0400
@@ -301,6 +301,31 @@
LdsState &lds;
public:
+ Stats::Scalar vALUInsts;
+ Stats::Formula vALUInstsPerWF;
+ Stats::Scalar sALUInsts;
+ Stats::Formula sALUInstsPerWF;
+ Stats::Scalar instCyclesVALU;
+ Stats::Scalar instCyclesSALU;
+ Stats::Scalar threadCyclesVALU;
+ Stats::Formula vALUUtilization;
+ Stats::Scalar ldsNoFlatInsts;
+ Stats::Formula ldsNoFlatInstsPerWF;
+ Stats::Scalar flatVMemInsts;
+ Stats::Formula flatVMemInstsPerWF;
+ Stats::Scalar flatLDSInsts;
+ Stats::Formula flatLDSInstsPerWF;
+ Stats::Scalar vectorMemWrites;
+ Stats::Formula vectorMemWritesPerWF;
+ Stats::Scalar vectorMemReads;
+ Stats::Formula vectorMemReadsPerWF;
+ Stats::Scalar scalarMemWrites;
+ Stats::Formula scalarMemWritesPerWF;
+ Stats::Scalar scalarMemReads;
+ Stats::Formula scalarMemReadsPerWF;
+
+ void updateInstStats(GPUDynInstPtr gpuDynInst);
+
// the following stats compute the avg. TLB accesslatency per
// uncoalesced request (only for data)
Stats::Scalar tlbRequests;
diff -r c3b4d57a15c5 -r 0a65922d564d src/gpu-compute/wavefront.cc
--- a/src/gpu-compute/wavefront.cc Wed Oct 26 22:47:27 2016 -0400
+++ b/src/gpu-compute/wavefront.cc Wed Oct 26 22:47:30 2016 -0400
@@ -656,7 +656,11 @@
DPRINTF(GPUExec, "CU%d: WF[%d][%d]: wave[%d] Executing inst: %s "
"(pc: %i)\n", computeUnit->cu_id, simdId, wfSlotId, wfDynId,
ii->disassemble(), old_pc);
+
+ // update the instruction stats in the CU
+
ii->execute(ii);
+ computeUnit->updateInstStats(ii);
// access the VRF
computeUnit->vrf[simdId]->exec(ii, this);
srcRegOpDist.sample(ii->numSrcRegOperands());
_______________________________________________
gem5-dev mailing list
[email protected]
http://m5sim.org/mailman/listinfo/gem5-dev