diff --git a/include/gpufl/backends/nvidia/engine/pc_sampling_engine.cpp b/include/gpufl/backends/nvidia/engine/pc_sampling_engine.cpp index e55fc7b..bda426c 100644 --- a/include/gpufl/backends/nvidia/engine/pc_sampling_engine.cpp +++ b/include/gpufl/backends/nvidia/engine/pc_sampling_engine.cpp @@ -62,6 +62,16 @@ constexpr int kMaxGetDataCalls = 1024; // Rows per Monitor::PushProfileSamples call, bounding the staging vector. constexpr size_t kRowsPerPush = 8192; +// CUPTI counts each sample under its warp state and, when the scheduler +// issued nothing that cycle, again under the state's _not_issued twin, so +// only the other reasons add up to totalSamples. +bool IsNotIssuedReason(const std::string& name) { + static const std::string kSuffix = "_not_issued"; + return name.size() > kSuffix.size() && + name.compare(name.size() - kSuffix.size(), kSuffix.size(), + kSuffix) == 0; +} + PCSamplingBuffers* AllocatePcSamplingBuffers(const size_t numPcs, const size_t stallSlots) { auto* b = new PCSamplingBuffers(); @@ -867,6 +877,7 @@ void PcSamplingEngine::CollectPcSamplingData_() { std::vector rows; size_t rowsEmitted = 0; uint64_t samplesEmitted = 0; + uint64_t notIssuedEmitted = 0; auto pushRows = [&rows, &rowsEmitted] { if (rows.empty()) return; Monitor::PushProfileSamples(rows); @@ -931,6 +942,7 @@ void PcSamplingEngine::CollectPcSamplingData_() { " nonUsrKernelsTotalSamples=", nonUsrSamples); uint64_t samplesThisCall = 0; + uint64_t notIssuedThisCall = 0; for (size_t i = 0; i < numPcs; ++i) { const CUpti_PCSamplingPCData& pc = batch->pPcData[i]; if (pc.stallReasonCount == 0 || !pc.stallReason) continue; @@ -953,7 +965,10 @@ void PcSamplingEngine::CollectPcSamplingData_() { const uint32_t samples = pc.stallReason[j].samples; const uint32_t reason = pc.stallReason[j].pcSamplingStallReasonIndex; - samplesThisCall += samples; + const auto name = reasonNames.find(reason); + const bool notIssued = name != reasonNames.end() && + IsNotIssuedReason(name->second); + (notIssued ? notIssuedThisCall : samplesThisCall) += samples; if (samples == 0) continue; if (!deviceIdKnown) { deviceIdKnown = true; @@ -970,7 +985,6 @@ void PcSamplingEngine::CollectPcSamplingData_() { s.device_id = deviceId; s.function_key = source->second.functionKey; s.pc_offset = static_cast(pcOffset); - const auto name = reasonNames.find(reason); s.metric_name = name != reasonNames.end() ? name->second : "Stall_" + std::to_string(reason); @@ -980,14 +994,15 @@ void PcSamplingEngine::CollectPcSamplingData_() { s.source_file = source->second.sourceFile; s.source_line = source->second.sourceLine; rows.push_back(std::move(s)); - samplesEmitted += samples; + (notIssued ? notIssuedEmitted : samplesEmitted) += samples; } if (rows.size() >= kRowsPerPush) pushRows(); } pushRows(); GFL_LOG_DEBUG("[PC Sampling] GetData returned ", samplesThisCall, - " samples across ", numPcs, " PC records"); + " samples (+", notIssuedThisCall, " not issued) across ", + numPcs, " PC records"); // Drained once a call returns nothing and CUPTI reports nothing // pending. Two empty calls in a row with something still pending end // it too, so a stuck report cannot spin. @@ -1002,7 +1017,8 @@ void PcSamplingEngine::CollectPcSamplingData_() { } GFL_LOG_DEBUG("[PC Sampling] collect summary: ", rowsEmitted, " rows, ", - samplesEmitted, " samples across ", sourceByPc.size(), + samplesEmitted, " samples (+", notIssuedEmitted, + " not issued) across ", sourceByPc.size(), " PCs; totalSamples=", sumTotal, " dropped=", sumDropped, " nonUsrKernels=", sumNonUsr, hardwareBufferFull ? " (hardware buffer overflowed)" : ""); diff --git a/include/gpufl/report/hint_engine.hpp b/include/gpufl/report/hint_engine.hpp index dd7b95e..e2d27a6 100644 --- a/include/gpufl/report/hint_engine.hpp +++ b/include/gpufl/report/hint_engine.hpp @@ -13,6 +13,10 @@ namespace report { struct FuncProfile { std::map stalls; // display-name → count uint64_t totalStalls = 0; + // Samples also counted in `stalls`, taken on cycles where the scheduler + // issued no instruction (CUPTI's `_not_issued` reasons). + std::map notIssuedStalls; + uint64_t totalNotIssued = 0; uint64_t warpInsts = 0; uint64_t threadInsts = 0; uint64_t globalSectors = 0; diff --git a/include/gpufl/report/text_report.cpp b/include/gpufl/report/text_report.cpp index c7e768b..9681717 100644 --- a/include/gpufl/report/text_report.cpp +++ b/include/gpufl/report/text_report.cpp @@ -1550,22 +1550,23 @@ void TextReport::writeProfileAnalysis(std::ostringstream& out) const { return; } + // CUPTI counts each sample under its warp state and, when the scheduler + // issued nothing that cycle, again under the state's _not_issued twin. + const std::string notIssued = "_not_issued"; + auto isNotIssued = [¬Issued](const std::string& raw) { + return raw.size() > notIssued.size() && + raw.compare(raw.size() - notIssued.size(), notIssued.size(), + notIssued) == 0; + }; + // Convert a raw CUPTI stall metric name to a human-readable short name. - // e.g. "smsp__pcsamp_warps_issue_stalled_wait_not_issued" → "Wait (not issued)" - auto shortenStallName = [](const std::string& raw) -> std::string { + // e.g. "smsp__pcsamp_warps_issue_stalled_wait_not_issued" → "Wait" + auto shortenStallName = [&](const std::string& raw) -> std::string { const std::string prefix = "smsp__pcsamp_warps_issue_stalled_"; std::string s = raw; if (s.size() > prefix.size() && s.substr(0, prefix.size()) == prefix) s = s.substr(prefix.size()); - - // Handle "_not_issued" suffix - const std::string notIssued = "_not_issued"; - bool isNotIssued = false; - if (s.size() > notIssued.size() && - s.substr(s.size() - notIssued.size()) == notIssued) { - s = s.substr(0, s.size() - notIssued.size()); - isNotIssued = true; - } + if (isNotIssued(s)) s.resize(s.size() - notIssued.size()); // Replace underscores with spaces and capitalize first letter of each word for (size_t i = 0; i < s.size(); ++i) { @@ -1573,8 +1574,6 @@ void TextReport::writeProfileAnalysis(std::ostringstream& out) const { if (i == 0 || (i > 0 && s[i-1] == ' ')) s[i] = static_cast(std::toupper(static_cast(s[i]))); } - - if (isNotIssued) s += " (idle)"; return s; }; @@ -1601,8 +1600,14 @@ void TextReport::writeProfileAnalysis(std::ostringstream& out) const { if (ps.stall_reason > 1) { std::string reason = resolveStallDisplay(ps); - fp.stalls[reason] += ps.metric_value; - fp.totalStalls += ps.metric_value; + if (isNotIssued(ps.reason_name.empty() ? ps.metric_name + : ps.reason_name)) { + fp.notIssuedStalls[reason] += ps.metric_value; + fp.totalNotIssued += ps.metric_value; + } else { + fp.stalls[reason] += ps.metric_value; + fp.totalStalls += ps.metric_value; + } } if (ps.metric_name == "smsp__sass_inst_executed") fp.warpInsts += ps.metric_value; @@ -1627,6 +1632,18 @@ void TextReport::writeProfileAnalysis(std::ostringstream& out) const { if (static_cast(ranked.size()) > top_n_) ranked.resize(top_n_); + auto writeStallShares = [&out](const std::map& counts, + uint64_t total) { + auto stallRanked = sortedTopN(counts, 0, [](uint64_t v) { return static_cast(v); }); + for (const auto& [reason, count] : stallRanked) { + double pct = total > 0 ? count * 100.0 / total : 0; + out << " " << std::left << std::setw(28) << truncate(reason, 26) + << std::right << std::setw(8) << fmtCount(count) + << std::setw(7) << std::fixed << std::setprecision(1) << pct << "% " + << makeBar(pct) << "\n"; + } + }; + // ── Write per-function analysis ───────────────────────────────────────── for (const auto& [fn, fp] : ranked) { std::string shortName = shortenKernelName(fn); @@ -1637,15 +1654,13 @@ void TextReport::writeProfileAnalysis(std::ostringstream& out) const { // Stall distribution if (!fp->stalls.empty()) { - auto stallRanked = sortedTopN(fp->stalls, 0, [](uint64_t v) { return static_cast(v); }); out << " Stalls:\n"; - for (const auto& [reason, count] : stallRanked) { - double pct = fp->totalStalls > 0 ? count * 100.0 / fp->totalStalls : 0; - out << " " << std::left << std::setw(28) << truncate(reason, 26) - << std::right << std::setw(8) << fmtCount(count) - << std::setw(7) << std::fixed << std::setprecision(1) << pct << "% " - << makeBar(pct) << "\n"; - } + writeStallShares(fp->stalls, fp->totalStalls); + } + if (!fp->notIssuedStalls.empty()) { + out << " Not issued (" << fmtCount(fp->totalNotIssued) + << " samples on cycles with no instruction issued):\n"; + writeStallShares(fp->notIssuedStalls, fp->totalNotIssued); } // Instruction analysis diff --git a/python/gpufl/analyzer/analyzer.py b/python/gpufl/analyzer/analyzer.py index 4b2990b..9fc8e75 100644 --- a/python/gpufl/analyzer/analyzer.py +++ b/python/gpufl/analyzer/analyzer.py @@ -45,7 +45,8 @@ def _shorten_kernel_name(name: str) -> tuple[str, str]: return short_func, name -# CUPTI CUpti_ActivityPCSamplingStallReason - skip 0 (invalid) and 1 (none) +# CUPTI CUpti_ActivityPCSamplingStallReason - skip 0 (invalid) and 1 (none). +# Only Activity API rows use these indices; PC Sampling API rows carry names. _STALL_NAMES: dict[int, str] = { 2: "Instruction Fetch", 3: "Execution Dependency", @@ -55,12 +56,37 @@ def _shorten_kernel_name(name: str) -> tuple[str, str]: 7: "Constant Memory", 8: "Pipe Busy", 9: "Memory Throttle", - 10: "Branch Resolving", - 11: "Wait", - 12: "Barrier", - 13: "Sleeping", + 10: "Not Selected", + 11: "Other", + 12: "Sleeping", } +_PC_STALL_PREFIX = "smsp__pcsamp_warps_issue_stalled_" +_NOT_ISSUED_SUFFIX = "_not_issued" + + +def _pc_stall_reason(metric_name, stall_index) -> tuple: + """Return (reason, not_issued) for a pc_sampling row. + + The PC Sampling API counts each sample under its warp state and, when + the scheduler issued nothing that cycle, again under the state's + ``_not_issued`` twin, so the two need separate denominators. Other + ``smsp__pcsamp_`` counters (sample_count, samples_data_dropped) are not + warp states. Activity API rows have no name, only an index. + """ + if metric_name: + if metric_name.startswith(_PC_STALL_PREFIX): + reason = metric_name[len(_PC_STALL_PREFIX):] + if reason.endswith(_NOT_ISSUED_SUFFIX): + return reason[:-len(_NOT_ISSUED_SUFFIX)], True + return reason, False + if metric_name.startswith("smsp__pcsamp_"): + return None, False + return metric_name, False + if stall_index > 1: + return _STALL_NAMES.get(stall_index, f"Stall_{stall_index}"), False + return None, False + class GpuFlightSession: def __init__( @@ -535,6 +561,9 @@ def _expand_batches(self, device_df, scope_df, system_df, dict_maps) -> tuple: scope_name = dict_maps['scope_name'].get(sn_id) if sn_id else None stall = row[ci['stall_reason']] mv = row[ci['metric_value']] + reason, not_issued = ( + _pc_stall_reason(metric_name, stall) + if sample_kind == 'pc_sampling' else (None, False)) sample_rows.append({ 'type': 'profile_sample', 'session_id': batch.get('session_id'), @@ -549,7 +578,8 @@ def _expand_batches(self, device_df, scope_df, system_df, dict_maps) -> tuple: 'sample_kind': sample_kind, 'scope_name': scope_name, # Compatibility aliases used by inspect_stalls / inspect_profile_samples - 'reason_name': _STALL_NAMES.get(stall, f"Stall_{stall}") if stall > 1 else None, + 'reason_name': reason, + 'not_issued': not_issued, 'sample_count': mv if sample_kind == 'pc_sampling' else 0, }) scopes_df = pd.DataFrame(sample_rows) @@ -1080,13 +1110,17 @@ def safe_mode(x): self.console.print(table) - def inspect_stalls(self, top_n: int = 10): + def inspect_stalls(self, top_n: int = 10, not_issued: bool = False): """Show per-kernel stall distribution from PC-sampling data. Requires ``enablePCSampling=true`` at session init. Joins ``profile_sample`` events to kernels via ``corr_id``, then pivots by ``reason_name`` to show what fraction of samples each stall category accounts for in the hottest kernels. + + Shares are of each kernel's samples, which add up to CUPTI's sample + count. ``not_issued=True`` shows the ``_not_issued`` samples instead: + the subset taken on cycles where the scheduler issued nothing. """ if self.scopes.empty or 'type' not in self.scopes.columns: self.console.print("[yellow]No PC sampling data found - enable PC sampling at session init.[/yellow]") @@ -1101,12 +1135,16 @@ def inspect_stalls(self, top_n: int = 10): self.console.print("[yellow]No profile_sample events found - enable PC sampling at init.[/yellow]") return - required = {'corr_id', 'reason_name', 'sample_count'} + required = {'corr_id', 'reason_name', 'sample_count', 'not_issued'} if not required.issubset(samples.columns): self.console.print(f"[yellow]profile_sample records missing columns: {required - set(samples.columns)}[/yellow]") return samples['sample_count'] = pd.to_numeric(samples['sample_count'], errors='coerce').fillna(0) + samples = samples[samples['not_issued'] == not_issued] + if samples.empty: + self.console.print("[yellow]No not-issued samples found.[/yellow]") + return # Aggregate sample counts: (corr_id, reason_name) → total samples stall_agg = ( @@ -1139,7 +1177,8 @@ def inspect_stalls(self, top_n: int = 10): stall_cols = [c for c in pivot.columns if c not in ('name', 'total_samples')] - table = Table(title=f"Stall Distribution - Top {top_n} Kernels (PC Sampling)") + family = "PC Sampling, not issued" if not_issued else "PC Sampling" + table = Table(title=f"Stall Distribution - Top {top_n} Kernels ({family})") table.add_column("Kernel", style="cyan", no_wrap=False) table.add_column("Samples", justify="right") for col in stall_cols: @@ -1205,22 +1244,41 @@ def inspect_profile_samples(self, top_n: int = 10): pc_samples = pd.DataFrame() if not pc_samples.empty: + if 'not_issued' not in pc_samples.columns: + pc_samples['not_issued'] = False + # A _not_issued sample is also counted under its warp state, so + # each family gets its own denominator. + has_reason = pc_samples['reason_name'].notna() + states = pc_samples[has_reason & ~pc_samples['not_issued']] + twins = pc_samples[has_reason & pc_samples['not_issued']] by_reason = ( - pc_samples.groupby('reason_name', dropna=False)['sample_count'] + states.groupby('reason_name')['sample_count'] .sum() .sort_values(ascending=False) .head(top_n) ) + twin_by_reason = twins.groupby('reason_name')['sample_count'].sum() - reason_table = Table(title=f"PC Sampling Reasons - Top {top_n}") + reason_table = Table( + title=f"PC Sampling Reasons - Top {top_n}", + caption="Not issued: samples taken on cycles where the " + "scheduler issued nothing (a subset of Samples).", + ) reason_table.add_column("Reason", style="cyan") reason_table.add_column("Samples", justify="right") - total_samples = float(pc_samples['sample_count'].sum()) or 1.0 reason_table.add_column("Share", justify="right") + reason_table.add_column("Not issued", justify="right") + reason_table.add_column("Share", justify="right") + total_samples = float(states['sample_count'].sum()) or 1.0 + total_twins = float(twins['sample_count'].sum()) or 1.0 for reason, count in by_reason.items(): - label = str(reason) if pd.notna(reason) and str(reason) else "unknown" - reason_table.add_row(label, str(int(count)), f"{(count/total_samples)*100:.1f}%") + twin = float(twin_by_reason.get(reason, 0)) + reason_table.add_row( + str(reason), str(int(count)), + f"{(count/total_samples)*100:.1f}%", + str(int(twin)), f"{(twin/total_twins)*100:.1f}%", + ) self.console.print(reason_table) if not self.kernels.empty and 'corr_id' in self.kernels.columns and 'corr_id' in pc_samples.columns: @@ -1234,7 +1292,12 @@ def inspect_profile_samples(self, top_n: int = 10): ) corr_to_name, fallback_used = self._resolve_sample_kernel_names(pc_samples) - kernel_samples = pc_samples.groupby('corr_id', as_index=False)['sample_count'].sum() + kernel_samples = states.groupby('corr_id', as_index=False)['sample_count'].sum() + kernel_samples = kernel_samples.join( + twins.groupby('corr_id')['sample_count'].sum().rename('not_issued_count'), + on='corr_id', + ) + kernel_samples['not_issued_count'] = kernel_samples['not_issued_count'].fillna(0) if corr_to_name: map_df = pd.DataFrame( [(k, v) for k, v in corr_to_name.items()], @@ -1249,8 +1312,10 @@ def inspect_profile_samples(self, top_n: int = 10): kernel_table = Table(title=f"PC Sampling Kernels - Top {top_n}") kernel_table.add_column("Kernel", style="cyan") kernel_table.add_column("Samples", justify="right") + kernel_table.add_column("Not issued", justify="right") for _, row in kernel_samples.iterrows(): - kernel_table.add_row(str(row['name']), str(int(row['sample_count']))) + kernel_table.add_row(str(row['name']), str(int(row['sample_count'])), + str(int(row['not_issued_count']))) self.console.print(kernel_table) if fallback_used > 0: self.console.print( diff --git a/python/gpufl/report/text_report.py b/python/gpufl/report/text_report.py index 59f6952..ab44102 100644 --- a/python/gpufl/report/text_report.py +++ b/python/gpufl/report/text_report.py @@ -657,17 +657,26 @@ def _section_profile_analysis(self) -> list[str]: # Stall reason distribution if has_stalls: stall_data = samples[samples["reason_name"].notna()] + # A _not_issued sample is also counted under its warp state. + twins = stall_data[stall_data["not_issued"]] + stall_data = stall_data[~stall_data["not_issued"]] stall_counts = stall_data.groupby("reason_name")["sample_count"].sum() + twin_counts = twins.groupby("reason_name")["sample_count"].sum() total_stalls = stall_counts.sum() + total_twins = twin_counts.sum() if total_stalls > 0: lines.append(" Stall Reason Distribution:") - hdr = f" {'Reason':<30}{'Samples':>10}{'Pct':>8}" + hdr = (f" {'Reason':<30}{'Samples':>10}{'Pct':>8}" + f"{'Not issued':>12}{'Pct':>8}") lines.append(hdr) - lines.append(" " + "-" * 48) + lines.append(" " + "-" * 68) for reason, count in stall_counts.sort_values(ascending=False).items(): pct = count / total_stalls * 100 - lines.append(f" {str(reason):<30}{int(count):>10}{pct:>7.1f}%") + twin = twin_counts.get(reason, 0) + twin_pct = twin / total_twins * 100 if total_twins > 0 else 0.0 + lines.append(f" {str(reason):<30}{int(count):>10}{pct:>7.1f}%" + f"{int(twin):>12}{twin_pct:>7.1f}%") # Per-kernel stall breakdown if "function_name" in stall_data.columns: diff --git a/tests/core/test_text_report.cpp b/tests/core/test_text_report.cpp index 50edc6d..cb327a2 100644 --- a/tests/core/test_text_report.cpp +++ b/tests/core/test_text_report.cpp @@ -170,4 +170,31 @@ TEST_F(TextReportTest, LegacySynchronizationEventProducesSummary) { EXPECT_NE(report.find("Event Synchronize"), std::string::npos); } +TEST_F(TextReportTest, PcStallSharesExcludeNotIssuedSamples) { + // Reason names and indices as the PC Sampling API enumerates them on an + // RTX 5060; counts are synthetic. A _not_issued sample is also counted + // under its warp state, so the 100 samples here carry 90 twins. + WriteLog("scope", { + R"({"type":"job_start","session_id":"s1","app":"pcs","ts_ns":1000})", + R"({"type":"dictionary_update","session_id":"s1","function_dict":{"1":"memBound@"},"metric_dict":{"1":"smsp__pcsamp_warps_issue_stalled_long_scoreboard","2":"smsp__pcsamp_warps_issue_stalled_long_scoreboard_not_issued","3":"smsp__pcsamp_warps_issue_stalled_wait","4":"smsp__pcsamp_warps_issue_stalled_wait_not_issued","5":"smsp__pcsamp_warps_issue_stalled_selected"}})", + R"({"type":"profile_sample_batch","session_id":"s1","columns":["function_id","metric_id","metric_value","stall_reason","sample_kind"],"rows":[[1,1,90,12,0],[1,2,85,13,0],[1,3,6,34,0],[1,4,5,35,0],[1,5,4,26,0]]})", + R"({"type":"shutdown","session_id":"s1","ts_ns":3000})", + }); + + const std::string report = Generate(); + EXPECT_NE(report.find("(100 stall samples)"), std::string::npos) << report; + const auto stalls = report.find(" Stalls:\n"); + const auto notIssued = report.find(" Not issued"); + ASSERT_NE(stalls, std::string::npos) << report; + ASSERT_NE(notIssued, std::string::npos) << report; + const auto base = report.find("Long Scoreboard", stalls); + ASSERT_LT(base, notIssued) << report; + EXPECT_NE(report.substr(base, 60).find(" 90.0%"), std::string::npos) + << report; + const auto twin = report.find("Long Scoreboard", notIssued); + ASSERT_NE(twin, std::string::npos) << report; + EXPECT_NE(report.substr(twin, 60).find(" 94.4%"), std::string::npos) + << report; +} + } // namespace diff --git a/tests/python/test_pc_stall_shares.py b/tests/python/test_pc_stall_shares.py new file mode 100644 index 0000000..2433311 --- /dev/null +++ b/tests/python/test_pc_stall_shares.py @@ -0,0 +1,157 @@ +import io +import json +import sys +from pathlib import Path + +from rich.console import Console + +sys.path.insert(0, str(Path(__file__).parent.parent.parent / "python")) + +from gpufl.analyzer import GpuFlightSession +from gpufl.report import generate_report + +# Reason names and indices as the PC Sampling API enumerates them on an +# RTX 5060 (CUDA 13.3). Sample counts are synthetic. CUPTI counts each sample +# once under its warp state and again under the _not_issued twin when the +# scheduler issued nothing that cycle, so twin <= state for every PC. +_P = "smsp__pcsamp_warps_issue_stalled_" +_METRICS = { + "1": _P + "long_scoreboard", + "2": _P + "long_scoreboard_not_issued", + "3": _P + "wait", + "4": _P + "wait_not_issued", + "5": _P + "selected", + "6": _P + "not_selected", +} +_COLUMNS = ["dt_ns", "corr_id", "device_id", "function_id", "pc_offset", + "metric_id", "metric_value", "stall_reason", "sample_kind", + "scope_name_id", "source_file_id", "source_line"] + + +def _row(corr_id, function_id, pc, metric_id, samples, stall_index): + return [0, corr_id, 0, function_id, pc, metric_id, samples, stall_index, + 0, 0, 0, 0] + + +# memBound (corr 4): 100 samples, 90 not-issued. +# issueBound (corr 5): 120 samples, 10 not-issued. +_ROWS = [ + _row(4, 1, 192, 1, 90, 12), + _row(4, 1, 192, 2, 85, 13), + _row(4, 1, 128, 3, 6, 34), + _row(4, 1, 128, 4, 5, 35), + _row(4, 1, 64, 5, 4, 26), + _row(5, 2, 64, 5, 70, 26), + _row(5, 2, 128, 3, 30, 34), + _row(5, 2, 128, 4, 10, 35), + _row(5, 2, 32, 6, 20, 24), +] + + +def _write_session(tmp_path, prefix="pcs"): + log_dir = tmp_path / "logs" + log_dir.mkdir() + device = [ + {"version": 1, "type": "job_start", "session_id": "pcs-session", + "app": "pcs", "pid": 1, "ts_ns": 100, "host": {}, "devices": []}, + {"version": 1, "type": "dictionary_update", + "session_id": "pcs-session", + "kernel_dict": {"1": "memBound", "2": "issueBound"}, + "scope_name_dict": {}, "function_dict": {}, "metric_dict": {}}, + {"version": 1, "type": "kernel_event_batch", + "session_id": "pcs-session", "batch_id": 1, "base_time_ns": 1000, + "columns": ["dt_ns", "kernel_id", "stream_id", "duration_ns", + "corr_id", "dyn_shared", "num_regs", "has_details"], + "rows": [[0, 1, 0, 100, 4, 0, 16, 0], + [200, 2, 0, 100, 5, 0, 16, 0]]}, + {"type": "shutdown", "session_id": "pcs-session", "app": "pcs", + "pid": 1, "ts_ns": 5000}, + ] + scope = [ + {"version": 1, "type": "dictionary_update", + "session_id": "pcs-session", + "function_dict": {"1": "memBound@", "2": "issueBound@"}, + "metric_dict": _METRICS}, + {"version": 1, "type": "profile_sample_batch", + "session_id": "pcs-session", "batch_id": 1, "base_time_ns": 2000, + "columns": _COLUMNS, "rows": _ROWS}, + ] + for channel, events in (("device", device), ("scope", scope)): + with open(log_dir / f"{prefix}.{channel}.log", "w") as f: + for ev in events: + f.write(json.dumps(ev) + "\n") + return log_dir, prefix + + +def _session(tmp_path): + log_dir, prefix = _write_session(tmp_path) + session = GpuFlightSession(log_dir, log_prefix=prefix) + buf = io.StringIO() + session.console = Console(file=buf, width=200, color_system=None) + return session, buf + + +def _lines(text, label): + return [line for line in text.splitlines() if label in line] + + +def test_pc_reason_names_come_from_the_sampling_api_name(tmp_path): + session, _ = _session(tmp_path) + pc = session.scopes[session.scopes["type"] == "profile_sample"] + + assert set(pc["reason_name"]) == { + "long_scoreboard", "wait", "selected", "not_selected"} + assert int(pc.loc[pc["not_issued"], "sample_count"].sum()) == 100 + assert int(pc.loc[~pc["not_issued"], "sample_count"].sum()) == 220 + + +def test_stall_shares_exclude_not_issued_samples(tmp_path): + session, buf = _session(tmp_path) + + session.inspect_stalls() + out = buf.getvalue() + + mem = _lines(out, "memBound") + issue = _lines(out, "issueBound") + assert mem and issue, out + assert " 100 " in mem[0] and "90.0%" in mem[0], out + assert " 120 " in issue[0] and "58.3%" in issue[0], out + # Ranked by samples, not by samples plus their not-issued twins. + assert out.index("issueBound") < out.index("memBound"), out + + +def test_not_issued_breakdown_uses_its_own_denominator(tmp_path): + session, buf = _session(tmp_path) + + session.inspect_stalls(not_issued=True) + out = buf.getvalue() + + mem = _lines(out, "memBound") + assert mem and " 90 " in mem[0] and "94.4%" in mem[0], out + + +def test_reason_table_reports_both_families_separately(tmp_path): + session, buf = _session(tmp_path) + + session.inspect_profile_samples() + out = buf.getvalue() + + ls = _lines(out, "long_scoreboard") + assert ls, out + # 90 of 220 samples; 85 of 100 not-issued samples. + assert "40.9%" in ls[0] and "85.0%" in ls[0], out + kernels = _lines(out, "issueBound") + assert kernels and " 120 " in kernels[-1], out + + +def test_text_report_stall_distribution_excludes_not_issued(tmp_path): + log_dir, prefix = _write_session(tmp_path) + + report = generate_report(str(log_dir), log_prefix=prefix) + + ls = _lines(report, "long_scoreboard") + assert ls and "40.9%" in ls[0] and "85.0%" in ls[0], report + ranked = _lines(report, " samples") + assert len(ranked) == 2, report + assert "issueBound" in ranked[0] and "120 samples" in ranked[0], report + assert "memBound" in ranked[1] and "100 samples" in ranked[1], report