Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 1 addition & 2 deletions daemon/launcher/cli_parse.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -127,8 +127,7 @@ TraceParseResult parseTraceArgs(const std::vector<std::string>& argv) {
"a deep window needs a bound: pass --deep-launches <n> or "
"--deep-for <duration> (prefer --deep-launches: it is what "
"the engines actually scale with. The replay engines cover "
"far less work per second of wall time, and PC sampling "
"returns nothing at all below a few thousand launches)"};
"far less work per second of wall time)"};
}
return {out, ""};
}
Expand Down
4 changes: 2 additions & 2 deletions daemon/launcher/cli_trace_options.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -568,8 +568,8 @@ const CliOptionManager<TraceArgs>& traceOptions() {
// Sampling
.add({"--pc-sample-period"}, "<N>",
"PC sampling period: log2 of GPU cycles per sample (5..31; "
"default 10). Lower = more frequent - for short kernels that "
"yield no PC samples by default.",
"default 10). Lower = more samples per kernel, at higher "
"overhead.",
kSection(TraceHelpSection::Sampling), &parsePcSamplePeriod)

// Removed flags: dispatched so the migration hint prints, but
Expand Down
6 changes: 3 additions & 3 deletions daemon/launcher/trace_command_common.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -786,9 +786,9 @@ int runTraceCommon(const TraceArgs& args, const TracePlatform& platform) {
return 2;
}

// --pc-sample-period: sample more frequently than the default so short
// kernels that yield no samples produce data. 0 = leave the engine default;
// the injected target reads GPUFL_PC_SAMPLING_PERIOD in gpufl::init().
// --pc-sample-period: override the engine's sampling period. 0 = leave the
// engine default; the injected target reads GPUFL_PC_SAMPLING_PERIOD in
// gpufl::init().
if (args.pc_sample_period != 0 &&
!setEnvOrPrint(platform, env::kPcSamplingPeriod,
std::to_string(args.pc_sample_period))) {
Expand Down
121 changes: 61 additions & 60 deletions example/cuda/cupti_pc_sampling.cu
Original file line number Diff line number Diff line change
Expand Up @@ -6,8 +6,11 @@
#include <cstdio>
#include <cstdlib>
#include <chrono>
#include <map>
#include <set>
#include <thread>
#include <string>
#include <utility>
#include <vector>
#include <cupti_pcsampling.h>

Expand Down Expand Up @@ -73,12 +76,14 @@ struct PCSamplingBuffers {
CUpti_PCSamplingPCData* pcRecords;
};

struct PcSampleRecord {
std::string functionName;
uint64_t pcOffset;
uint64_t samples;
uint32_t correlationId;
};
void freePCSamplingBuffers(PCSamplingBuffers* buffers) {
for (size_t i = 0; i < buffers->data->collectNumPcs; ++i) {
std::free(buffers->pcRecords[i].stallReason);
}
std::free(buffers->pcRecords);
std::free(buffers->data);
std::free(buffers);
}

PCSamplingBuffers* configurePCSampling(CUcontext ctx) {
const size_t kMaxPcs = 65536;
Expand Down Expand Up @@ -121,7 +126,11 @@ int main() {

CUcontext ctx = ensureContext();

// `buffers` becomes the SAMPLING_DATA_BUFFER. In KERNEL_SERIALIZED mode
// CUPTI moves every finished kernel's records into it by itself, so
// GetData needs a buffer of its own or it overwrites them.
PCSamplingBuffers* buffers = configurePCSampling(ctx);
PCSamplingBuffers* readBuffers = configurePCSampling(ctx);

printf("Enabling PC sampling...\n"); fflush(stdout);
CUpti_PCSamplingEnableParams enableParams = {};
Expand Down Expand Up @@ -195,68 +204,60 @@ int main() {
CUpti_PCSamplingGetDataParams getDataParams = {};
getDataParams.size = sizeof(CUpti_PCSamplingGetDataParams);
getDataParams.ctx = ctx;
getDataParams.pcSamplingData = buffers->data;
checkCupti(cuptiPCSamplingGetData(&getDataParams), "cuptiPCSamplingGetData");
getDataParams.pcSamplingData = readBuffers->data;

// Each call moves records out of the configured buffer (one set per
// kernel launch), then the per-PC overflow of launches that did not fit.
// Done when a call returns nothing.
std::map<std::pair<std::string, uint64_t>, uint64_t> samplesByPc;
std::set<uint32_t> correlationIds;
size_t recordCount = 0;
unsigned long long totalSamples = 0;
for (;;) {
for (size_t i = 0; i < readBuffers->data->collectNumPcs; ++i) {
readBuffers->pcRecords[i].stallReasonCount = 128;
}
readBuffers->data->totalNumPcs = 0;
checkCupti(cuptiPCSamplingGetData(&getDataParams), "cuptiPCSamplingGetData");
const size_t pcCount = readBuffers->data->totalNumPcs;
if (pcCount == 0) {
break;
}
recordCount += pcCount;
totalSamples += readBuffers->data->totalSamples;
for (size_t i = 0; i < pcCount; ++i) {
// Copies: the CUPTI structs are packed, and GCC refuses to bind
// packed fields to references.
const CUpti_PCSamplingPCData& pc = readBuffers->data->pPcData[i];
const std::string functionName = pc.functionName ? pc.functionName : "<unknown>";
const uint64_t pcOffset = pc.pcOffset;
const uint32_t correlationId = pc.correlationId;
uint64_t samples = 0;
for (size_t j = 0; j < pc.stallReasonCount; ++j) {
samples += pc.stallReason[j].samples;
}
samplesByPc[{functionName, pcOffset}] += samples;
correlationIds.insert(correlationId);
}
}

std::fprintf(stdout, "Disabling PC sampling...\n"); std::fflush(stdout);
CUpti_PCSamplingDisableParams disableParams = {};
disableParams.size = sizeof(CUpti_PCSamplingDisableParams);
disableParams.ctx = ctx;
checkCupti(cuptiPCSamplingDisable(&disableParams), "cuptiPCSamplingDisable");
std::fprintf(stdout, "PC sampling data: totalSamples=%llu dropped=%llu totalPCs=%zu remaining=%zu rangeId=%llu\n",
static_cast<unsigned long long>(buffers->data->totalSamples),
static_cast<unsigned long long>(buffers->data->droppedSamples),
buffers->data->totalNumPcs,
buffers->data->remainingNumPcs,
static_cast<unsigned long long>(buffers->data->rangeId));
std::fflush(stdout);

size_t pcCount = buffers->data->totalNumPcs;
if (pcCount > buffers->data->collectNumPcs) {
pcCount = buffers->data->collectNumPcs;
}
//
std::vector<PcSampleRecord> records;
records.reserve(pcCount);
//
for (size_t i = 0; i < pcCount; ++i) {
const CUpti_PCSamplingPCData& pc = buffers->data->pPcData[i];
uint64_t samples = 0;
if (pc.stallReason) {
for (size_t j = 0; j < pc.stallReasonCount; ++j) {
samples += pc.stallReason[j].samples;
}
}
PcSampleRecord rec;
if (pc.functionName) {
rec.functionName = pc.functionName;
}
rec.pcOffset = pc.pcOffset;
rec.samples = samples;
rec.correlationId = pc.correlationId;
records.push_back(std::move(rec));
}

std::fprintf(stdout, "Collected %zu PC records\n", records.size());
for (size_t i = 0; i < records.size(); ++i) {
const PcSampleRecord& rec = records[i];
std::fprintf(stdout, " [%zu] %s pcOffset=0x%llx samples=%llu corr=%u\n",
i,
rec.functionName.empty() ? "<unknown>" : rec.functionName.c_str(),
static_cast<unsigned long long>(rec.pcOffset),
static_cast<unsigned long long>(rec.samples),
rec.correlationId);
}

size_t maxPcs = buffers->data->collectNumPcs;
for (size_t i = 0; i < maxPcs; ++i) {
if (buffers->pcRecords[i].stallReason) {
std::free(buffers->pcRecords[i].stallReason);
}
// Launches past the configured buffer share one correlation id.
std::fprintf(stdout, "Collected %zu PC records, %zu correlation ids (totalSamples=%llu)\n",
recordCount, correlationIds.size(), totalSamples);
for (const auto& [pc, samples] : samplesByPc) {
std::fprintf(stdout, " %s pcOffset=0x%llx samples=%llu\n", pc.first.c_str(),
static_cast<unsigned long long>(pc.second),
static_cast<unsigned long long>(samples));
}
std::free(buffers->pcRecords);
std::free(buffers->data);
std::free(buffers);

freePCSamplingBuffers(readBuffers);
freePCSamplingBuffers(buffers);

std::fprintf(stdout, "PC sampling stopped.\n");
return 0;
Expand Down
27 changes: 7 additions & 20 deletions include/gpufl/backends/nvidia/capture_capability_resolver.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -22,27 +22,14 @@ std::vector<std::string> BuildCaptureCapabilityWarnings(
std::vector<std::string> warnings;
if (input.requests.pc && input.engine_state.pc.active &&
!input.engine_state.pc.has_data) {
// What decides this is the number of kernel launches sampled, not
// wall time: in KERNEL_SERIALIZED mode nothing is readable until
// enough kernel ranges have accumulated. Measured on an RTX 5060 /
// CUDA 13.3, same 8 s window: 876 launches returned 0 samples, 2002
// returned 87.8M. So the old advice here was actively wrong - a
// "heavier" workload means fewer launches, and a finer
// --pc-sample-period costs enough overhead to cut the launches
// covered (992 -> 58 in a fixed window), moving further from the
// threshold in both cases.
// No count in the text: launch_count is process-wide, while what
// matters is how many launches happened while sampling was armed -
// for a deep window those differ by a lot, and quoting the wrong one
// contradicts the advice.
// Every sampled kernel is returned regardless of how many ran, so
// zero means no user kernel was sampled while the sampler was armed
// (or the hardware buffer overflowed, which the engine logs).
warnings.push_back(
"PC sampling collected 0 stall samples - too few kernel "
"launches were sampled. PC sampling accumulates per kernel and "
"needs a few thousand launches before any samples are readable. "
"Sample more launches - note that a longer wall-clock window is "
"not the same thing: a slow-kernel workload can run for seconds "
"and still cover too few. With a deep window, bound it with "
"`--deep-launches <n>` rather than `--deep-for <duration>`.");
"PC sampling collected 0 stall samples - no user kernel was "
"sampled while the sampler was armed. Check that kernels run "
"inside the sampled scope or deep window; very short kernels "
"collect more samples with a lower --pc-sample-period.");
}
if (input.requests.sass && input.engine_state.sass.active &&
!input.engine_state.sass.has_data) {
Expand Down
Loading
Loading