diff --git a/profile/database/static_info/aie_util.cpp b/profile/database/static_info/aie_util.cpp index 2c278ec9..35156229 100755 --- a/profile/database/static_info/aie_util.cpp +++ b/profile/database/static_info/aie_util.cpp @@ -419,6 +419,11 @@ namespace xdp::aie { boost::property_tree::ptree pt; pt.put("start_col", e.start_col); pt.put("num_cols", e.num_cols); + // The query is device-wide, so it also reports partitions belonging to + // other contexts and other processes. These two identify the owner: + // "id" is the driver's context id, which matches hwctx_handle::get_slotidx(). + pt.put("id", e.metadata.id); + pt.put("pid", e.pid); infoPt.push_back(std::make_pair("", pt)); } } diff --git a/profile/plugin/aie_dtrace/aie_dtrace_impl.h b/profile/plugin/aie_dtrace/aie_dtrace_impl.h index 7c505be7..0d4d4157 100644 --- a/profile/plugin/aie_dtrace/aie_dtrace_impl.h +++ b/profile/plugin/aie_dtrace/aie_dtrace_impl.h @@ -43,9 +43,23 @@ namespace xdp { virtual void endPoll() {} virtual void freeResources() {} - virtual void generateCTForRun(void* /*run_impl_ptr*/, void* /*hwctx*/, uint32_t /*run_uid*/, - const std::string& /*kernel_name*/, - void* /*elf_handle*/) {} + // Called once per xrt::run construction, where the ELF is available: + // generates every CT file the configured inference sequence needs so that + // run start only has to pick one. When a single metric set covers every + // inference there is nothing to pick, so the CT is also programmed here. + virtual void generateCTsForRun(void* /*run_impl_ptr*/, void* /*hwctx*/, uint32_t /*run_uid*/, + const std::string& /*kernel_name*/, + void* /*elf_handle*/) {} + + // Called on every xrt::run start: advances this kernel's inference counter + // and hands the matching pre-generated CT to XRT. Does nothing unless a + // multi-inference sequence is configured. + virtual void applyCTForRun(void* /*run_impl_ptr*/, void* /*hwctx*/, uint32_t /*run_uid*/, + const std::string& /*kernel_name*/) {} + + // Called at hardware context teardown, to report a configured inference + // sequence that the application did not run far enough to consume. + virtual void reportUnusedSelections() {} uint64_t getDeviceID() { return deviceID; } }; diff --git a/profile/plugin/aie_dtrace/aie_dtrace_metadata.cpp b/profile/plugin/aie_dtrace/aie_dtrace_metadata.cpp index f0cd2902..1554d698 100644 --- a/profile/plugin/aie_dtrace/aie_dtrace_metadata.cpp +++ b/profile/plugin/aie_dtrace/aie_dtrace_metadata.cpp @@ -81,15 +81,31 @@ namespace xdp { const bool usingBlob = profiling_runtime_config::has_control_instrumentation(); const auto& ci = profiling_runtime_config::control_instrumentation(); + const auto& runs = ci.profile_runs; + + const bool useProfileRuns = usingBlob && ci.has_explicit_profile_runs && !runs.empty(); + multiInference = useProfileRuns; + + // configMetrics describes the hardware context as a whole: it is what + // isConfigured() gates on and what createAIEProfileConfig() reports. A + // profile_runs sequence has no single answer for it, so resolve it from the + // first inference (falling back to any top-level key the sequence omits) + // and let the per-inference differences live in metricSelections instead. + const auto& effAieTile = (useProfileRuns && runs[0].aie_tile.has_value()) + ? runs[0].aie_tile : ci.aie_tile; + const auto& effInterfaceTile = (useProfileRuns && runs[0].interface_tile.has_value()) + ? runs[0].interface_tile : ci.interface_tile; + const auto& effMemTile = (useProfileRuns && runs[0].mem_tile.has_value()) + ? runs[0].mem_tile : ci.mem_tile; // Core (aie) tile metrics (e.g. compute_io_bound). Only used to enable the // metric; the tiles themselves are fixed to the first column. std::vector aieMetricsSettings; - if (usingBlob && ci.aie_tile.has_value() && !ci.aie_tile->empty()) { + if (usingBlob && effAieTile.has_value() && !effAieTile->empty()) { xrt_core::message::send(severity_level::info, "XRT", - "AIE dtrace: using aie_tile metric '" + *ci.aie_tile + "AIE dtrace: using aie_tile metric '" + *effAieTile + "' from Debug.profiling_runtime_config."); - aieMetricsSettings = getSettingsVector("all:" + *ci.aie_tile); + aieMetricsSettings = getSettingsVector("all:" + *effAieTile); } else { const std::string tileBasedAie = @@ -100,11 +116,11 @@ namespace xdp { getConfigMetricsForAIETiles(CORE_MODULE_IDX, aieMetricsSettings); std::vector metricsSettings; - if (usingBlob && ci.interface_tile.has_value() && !ci.interface_tile->empty()) { + if (usingBlob && effInterfaceTile.has_value() && !effInterfaceTile->empty()) { xrt_core::message::send(severity_level::info, "XRT", - "AIE dtrace: using interface_tile metric '" + *ci.interface_tile + "AIE dtrace: using interface_tile metric '" + *effInterfaceTile + "' from Debug.profiling_runtime_config."); - metricsSettings = getSettingsVector("all:" + *ci.interface_tile); + metricsSettings = getSettingsVector("all:" + *effInterfaceTile); } else { const std::string tileBased = @@ -128,11 +144,30 @@ namespace xdp { const std::string iniPorts = xrt_core::config::get_aie_dtrace_settings_memory_tile_input_ports(); const bool iniPortsSet = !iniPorts.empty(); + // The design points (which memory tile ports exist) are a property of the + // design, so they stay global even when the metric sets vary per inference; + // only whether to instrument them in a given inference is per-entry. + const bool anyRunWantsL2L2 = useProfileRuns + && std::any_of(runs.begin(), runs.end(), [](const auto& r) { + return r.mem_tile.has_value() && *r.mem_tile == INPUT_PORTS_METRIC_SET; + }); + + if (useProfileRuns + && std::any_of(runs.begin(), runs.end(), [](const auto& r) { + return r.memory_tile_input_ports.has_value() && !r.memory_tile_input_ports->empty(); + })) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: memory_tile_input_ports inside a profile_runs entry is not supported; " + "design points are taken from control_instrumentation.memory_tile_input_ports " + "(or AIE_dtrace_settings.memory_tile_input_ports) for every inference."); + } + const bool blobPortsSet = usingBlob && ci.memory_tile_input_ports.has_value() && !ci.memory_tile_input_ports->empty(); - const bool memTileFieldFromBlob = usingBlob && ci.mem_tile.has_value() - && !ci.mem_tile->empty(); - const bool memTileUsesBlob = usingBlob && (memTileFieldFromBlob || blobPortsSet); + const bool memTileFieldFromBlob = usingBlob && effMemTile.has_value() + && !effMemTile->empty(); + const bool memTileUsesBlob = usingBlob + && (memTileFieldFromBlob || blobPortsSet || anyRunWantsL2L2); // Mem tile settings that are not L2-L2 select a per-tile counter metric set such // as output_channels_details. Both families program the same mem tile performance @@ -142,17 +177,21 @@ namespace xdp { bool l2L2FromBlob = false; if (memTileUsesBlob) { - if (memTileFieldFromBlob && *ci.mem_tile == INPUT_PORTS_METRIC_SET) { + const bool blobEnablesL2L2 = anyRunWantsL2L2 + || (memTileFieldFromBlob && *effMemTile == INPUT_PORTS_METRIC_SET); + + if (blobEnablesL2L2) { l2L2TransferEnabled = true; l2L2FromBlob = true; xrt_core::message::send(severity_level::info, "XRT", - "AIE dtrace: enabling L2-L2 via mem_tile metric '" + *ci.mem_tile + "AIE dtrace: enabling L2-L2 via mem_tile metric '" + + std::string(INPUT_PORTS_METRIC_SET) + "' from Debug.profiling_runtime_config."); } else if (memTileFieldFromBlob) { xrt_core::message::send(severity_level::info, "XRT", - "AIE dtrace: using mem_tile metric '" + *ci.mem_tile + "AIE dtrace: using mem_tile metric '" + *effMemTile + "' from Debug.profiling_runtime_config."); - memTileMetricsSettings = getSettingsVector("all:" + *ci.mem_tile); + memTileMetricsSettings = getSettingsVector("all:" + *effMemTile); } } else { @@ -216,9 +255,223 @@ namespace xdp { } } + // Built last so it sees the final l2L2TransferEnabled, which the design + // point validation above can still turn back off. + if (useProfileRuns) { + metricSelections.reserve(runs.size()); + for (size_t i = 0; i < runs.size(); ++i) + metricSelections.push_back(buildSelectionFromProfileRun(runs[i], i)); + + std::stringstream msg; + msg << "AIE dtrace: profiling the first " << metricSelections.size() + << " inferences of each kernel:"; + for (size_t i = 0; i < metricSelections.size(); ++i) + msg << "\n inference " << i << ": " + << metricSelections[i].describe(); + xrt_core::message::send(severity_level::info, "XRT", msg.str()); + } + else { + metricSelections.push_back(buildSelectionFromConfigMetrics()); + } + xrt_core::message::send(severity_level::info, "XRT", "Finished parsing AIE dtrace metadata."); } + std::string MetricSelection::describe() const + { + std::stringstream msg; + const char* sep = ""; + + if (includeBandwidth) { + msg << "interface_tile=" << bandwidthMetricSet + << ":" << static_cast(bandwidthChannel); + sep = ", "; + } + if (!coreMetricSet.empty()) { + msg << sep << "aie_tile=" << coreMetricSet; + sep = ", "; + } + if (includeL2L2) { + msg << sep << "mem_tile=" << INPUT_PORTS_METRIC_SET; + sep = ", "; + } + if (!memTileMetricSet.empty()) + msg << sep << "mem_tile=" << memTileMetricSet + << ":" << static_cast(memTileChannel); + + const std::string out = msg.str(); + return out.empty() ? std::string("no metrics") : out; + } + + // Reduce the whole-context config maps to the one selection that every + // inference shares. This is the single-configuration form: the CT is the same + // no matter how many times the kernel runs. + MetricSelection AieDtraceMetadata::buildSelectionFromConfigMetrics() + { + MetricSelection selection; + + for (const auto& tc : getConfigMetricsVec(CORE_MODULE_IDX)) { + selection.coreMetricSet = tc.second; + break; + } + + // Interface-tile bandwidth metrics are configured by default unless the user + // turned interface tiles off, which leaves the shim config map empty. + const auto shimConfigMetrics = getConfigMetricsVec(SHIM_MODULE_IDX); + selection.includeBandwidth = !shimConfigMetrics.empty(); + + if (selection.includeBandwidth) { + selection.bandwidthMetricSet = shimConfigMetrics.front().second; + // The detailed_ddr_*_bandwidth metric sets carry a DMA channel in their + // ":" suffix, which getConfigMetricsForInterfaceTiles stored in + // configChannel0 keyed by tile. + const auto& shimTile = shimConfigMetrics.front().first; + for (const auto& tc : configChannel0) { + if ((tc.first.col == shimTile.col) && (tc.first.row == shimTile.row)) { + selection.bandwidthChannel = tc.second; + break; + } + } + } + + selection.includeL2L2 = l2L2TransferEnabled; + + // Per-tile mem tile counters (output_channels_details / mm2s_channels_details). + // L2-L2 and these share the mem tile performance counters, and the constructor + // already drops the per-tile set when L2-L2 is enabled for this configuration. + const auto memTileConfigMetrics = getConfigMetricsVec(MEM_TILE_MODULE_IDX); + if (!memTileConfigMetrics.empty()) { + selection.memTileMetricSet = memTileConfigMetrics.front().second; + const auto& memTile = memTileConfigMetrics.front().first; + for (const auto& tc : configChannel0) { + if ((tc.first.col == memTile.col) && (tc.first.row == memTile.row)) { + selection.memTileChannel = tc.second; + break; + } + } + // An empty list means every partition column. A named-column setting + // copies those columns; "all" leaves the list empty. + if (!memTileAllColumns) { + for (const auto& tc : memTileConfigMetrics) + selection.memTileColumns.push_back(tc.first.col); + } + } + + return selection; + } + + // Resolve one profile_runs entry. Unlike the single-configuration form these + // are not routed through getConfigMetricsFor*Tiles: the CT writer derives its + // own tiles, so all that is needed here is the metric set names and the DMA + // channel carried in the interface tile's ":" suffix. + MetricSelection + AieDtraceMetadata::buildSelectionFromProfileRun( + const profiling_runtime_config::profile_run_t& run, size_t index) const + { + const std::string scope = "profile_runs[" + std::to_string(index) + "]"; + MetricSelection selection; + + if (run.interface_tile.has_value() && !run.interface_tile->empty()) { + std::vector parts; + boost::split(parts, *run.interface_tile, boost::is_any_of(":")); + const std::string& metricSet = parts.front(); + + if (!isBandwidthMetricSet(metricSet)) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: " + scope + ".interface_tile='" + *run.interface_tile + + "' is not a known interface tile metric set; no bandwidth counters " + "will be programmed for that inference."); + } + else if (metricSet != "off") { + selection.includeBandwidth = true; + selection.bandwidthMetricSet = metricSet; + + if (parts.size() > 1) { + try { + selection.bandwidthChannel = aie::convertStringToUint8(parts[1]); + } + catch (const std::invalid_argument&) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: channel '" + parts[1] + "' in " + scope + + ".interface_tile is not an integer; using channel 0."); + } + } + } + } + + if (run.aie_tile.has_value() && !run.aie_tile->empty()) { + // Accept "all:" as well as a bare "", matching what + // getConfigMetricsForAIETiles allows for the single-configuration form. + std::vector parts; + boost::split(parts, *run.aie_tile, boost::is_any_of(":")); + const std::string& metricSet = parts.back(); + + if (!isCoreMetricSet(metricSet)) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: " + scope + ".aie_tile='" + *run.aie_tile + + "' is not a known core (aie) tile metric set. Supported: compute_io_bound, off."); + } + else if (metricSet != "off") { + selection.coreMetricSet = metricSet; + } + } + + if (run.mem_tile.has_value() && !run.mem_tile->empty()) { + if (*run.mem_tile == INPUT_PORTS_METRIC_SET) { + selection.includeL2L2 = l2L2TransferEnabled; + } + else { + std::vector parts; + boost::split(parts, *run.mem_tile, boost::is_any_of(":")); + auto metricPos = std::find_if(parts.begin(), parts.end(), + [this](const std::string& part) { return isMemTileMetricSet(part); }); + + if (metricPos == parts.end()) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: " + scope + ".mem_tile='" + *run.mem_tile + + "' is not a known mem tile metric set. Supported: input_ports, " + "output_channels_details, mm2s_channels_details, off."); + } + else if (*metricPos != "off") { + selection.memTileMetricSet = *metricPos; + // A column ahead of the metric names one column. "all" and a bare + // metric leave memTileColumns empty, which means every partition column. + if ((metricPos != parts.begin()) && (parts.front().compare("all") != 0)) { + try { + selection.memTileColumns.push_back(aie::convertStringToUint8(parts.front())); + } + catch (const std::invalid_argument&) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: column '" + parts.front() + "' in " + scope + + ".mem_tile is not an integer; using every partition column."); + } + } + auto channelPos = std::next(metricPos); + if (channelPos != parts.end()) { + try { + const uint8_t requested = aie::convertStringToUint8(*channelPos); + if (requested < NUM_MEM_TILE_DMA_CHANNELS) + selection.memTileChannel = requested; + else + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: mem tile MM2S channel " + std::to_string(requested) + + " in " + scope + ".mem_tile is out of range (0-" + + std::to_string(NUM_MEM_TILE_DMA_CHANNELS - 1) + + "). Using channel 0."); + } + catch (const std::invalid_argument&) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: channel '" + *channelPos + "' in " + scope + + ".mem_tile is not an integer; using channel 0."); + } + } + } + } + } + + return selection; + } + void AieDtraceMetadata::checkDtraceSettings() { using boost::property_tree::ptree; diff --git a/profile/plugin/aie_dtrace/aie_dtrace_metadata.h b/profile/plugin/aie_dtrace/aie_dtrace_metadata.h index b9a07577..64d278fb 100644 --- a/profile/plugin/aie_dtrace/aie_dtrace_metadata.h +++ b/profile/plugin/aie_dtrace/aie_dtrace_metadata.h @@ -12,9 +12,36 @@ #include "xdp/profile/database/static_info/aie_constructs.h" #include "xdp/profile/database/static_info/filetypes/base_filetype_impl.h" +#include "xdp/profile/plugin/vp_base/profiling_runtime_config.h" namespace xdp { +// Everything that distinguishes one inference's CT file from another's. The +// Nth profiled inference of a kernel is programmed with metricSelections[N], +// so this has to be self-contained rather than something the CT writer reads +// back off the shared, whole-context config maps. +struct MetricSelection { + bool includeBandwidth = false; + std::string bandwidthMetricSet = "peak_read_bandwidth"; + uint8_t bandwidthChannel = 0; + std::string coreMetricSet; // empty means no core (aie) tile metrics + bool includeL2L2 = false; + std::string memTileMetricSet; // empty means no per-tile mem tile counters + uint8_t memTileChannel = 0; // MM2S channel for output/mm2s_channels_details + // Empty means every mem tile column in the partition. A ":" + // prefix lists those columns here. + std::vector memTileColumns; + + bool empty() const { + return !includeBandwidth && coreMetricSet.empty() && !includeL2L2 + && memTileMetricSet.empty(); + } + + // Human-readable "interface_tile=..., aie_tile=..." form used in log + // messages and to distinguish CT files in diagnostics. + std::string describe() const; +}; + class AieDtraceMetadata { private: static constexpr int SHIM_MODULE_IDX = static_cast(module_type::shim); @@ -51,9 +78,17 @@ class AieDtraceMetadata { std::map configChannel0; std::map configChannel1; + // One entry per inference to profile, in execution order. Always holds at + // least one entry once the metadata is configured. + std::vector metricSelections; + bool multiInference = false; + const aie::BaseFiletypeImpl* metadataReader = nullptr; void checkDtraceSettings(); + MetricSelection buildSelectionFromProfileRun( + const profiling_runtime_config::profile_run_t& run, size_t index) const; + MetricSelection buildSelectionFromConfigMetrics(); void getConfigMetricsForInterfaceTiles(int moduleIdx, const std::vector& metricsSettings); void getConfigMetricsForAIETiles(int moduleIdx, @@ -83,11 +118,12 @@ class AieDtraceMetadata { bool isConfigOnePartition() const { return configOnePartition; } - bool isL2L2Enabled() const { return l2L2TransferEnabled; } + // Per-inference metric selections, in execution order. + const std::vector& getMetricSelections() const { return metricSelections; } - // When true the mem tile metric applies to every column in the partition and - // the config map holds only a placeholder column. - bool isMemTileAllColumns() const { return memTileAllColumns; } + // True when the user asked for a "profile_runs" sequence rather than a + // single configuration applied to every inference. + bool isMultiInference() const { return multiInference; } bool aieMetadataEmpty() { return metadataReader == nullptr; } diff --git a/profile/plugin/aie_dtrace/aie_dtrace_plugin.cpp b/profile/plugin/aie_dtrace/aie_dtrace_plugin.cpp index 3c5f1abc..85dc5054 100644 --- a/profile/plugin/aie_dtrace/aie_dtrace_plugin.cpp +++ b/profile/plugin/aie_dtrace/aie_dtrace_plugin.cpp @@ -25,7 +25,7 @@ namespace xdp { using severity_level = xrt_core::message::severity_level; bool AieDtracePlugin::live = false; - bool AieDtracePlugin::configuredOnePartition = false; + std::atomic AieDtracePlugin::configuredOnePartition{false}; AieDtracePlugin::AieDtracePlugin() : XDPPlugin() @@ -65,6 +65,14 @@ namespace xdp { return (db->getStaticInfo()).getDeviceContextUniqueId(handle); } + std::shared_ptr AieDtracePlugin::findImpl(void* handle) const + { + std::lock_guard lock(implMutex); + + auto itr = handleToAIEDtraceImpl.find(handle); + return (itr == handleToAIEDtraceImpl.end()) ? nullptr : itr->second; + } + void AieDtracePlugin::updateAIEDtraceDevice(void* handle, bool hw_context_flow) { xrt_core::message::send(severity_level::info, "XRT", "AIE dtrace: update device."); @@ -116,6 +124,11 @@ namespace xdp { } #endif + // Held for the rest of the device update so a run constructor on another + // thread either sees no implementation for this context or sees a fully + // installed one, never one that is midway through being replaced. + std::lock_guard lock(implMutex); + auto deviceID = getDeviceIDFromHandle(handle); { @@ -148,9 +161,9 @@ namespace xdp { if (metadata->isConfigOnePartition() && metadata->isConfigured()) configuredOnePartition = true; - std::unique_ptr implementation; + std::shared_ptr implementation; #if defined(XDP_VE2_BUILD) - implementation = std::make_unique(db, metadata, deviceID); + implementation = std::make_shared(db, metadata, deviceID); #else xrt_core::message::send(severity_level::warning, "XRT", "AIE dtrace: no implementation for this build; skipping."); @@ -162,13 +175,26 @@ namespace xdp { handleToAIEDtraceImpl[handle] = std::move(implementation); } + void AieDtracePlugin::retireImpl(void* handle, AieDtraceImpl& impl) + { + (db->getStaticInfo()).unregisterPluginFromHwContext(handle); + + // Last chance to tell the user that the application did not run enough + // inferences to collect everything they configured. + impl.reportUnusedSelections(); + } + void AieDtracePlugin::writeAll(bool /*openNewFiles*/) { - for (const auto& kv : handleToAIEDtraceImpl) - endPollforDevice(kv.first); + { + std::lock_guard lock(implMutex); + for (const auto& kv : handleToAIEDtraceImpl) + retireImpl(kv.first, *kv.second); + + handleToAIEDtraceImpl.clear(); + } XDPPlugin::endWrite(); - handleToAIEDtraceImpl.clear(); } void AieDtracePlugin::endPollforDevice(void* handle) @@ -176,11 +202,15 @@ namespace xdp { if (!handle) return; - (db->getStaticInfo()).unregisterPluginFromHwContext(handle); + std::lock_guard lock(implMutex); auto itr = handleToAIEDtraceImpl.find(handle); - if (itr == handleToAIEDtraceImpl.end()) + if (itr == handleToAIEDtraceImpl.end()) { + (db->getStaticInfo()).unregisterPluginFromHwContext(handle); return; + } + + retireImpl(handle, *itr->second); // Drop implementation without endPoll(): dtrace must not read/offload on hwctx teardown; // ~AieDtrace_VE2Impl releases FAL resources only. @@ -193,13 +223,14 @@ namespace xdp { if (!xrt_core::config::get_aie_dtrace()) return; - auto itr = handleToAIEDtraceImpl.find(hwctx); - if (itr == handleToAIEDtraceImpl.end()) { + auto impl = findImpl(hwctx); + if (!impl) { xrt_core::message::send(severity_level::debug, "XRT", "AIE dtrace: no implementation for hwctx in runConstructorHook"); return; } - itr->second->generateCTForRun(run_impl_ptr, hwctx, run_uid, kernel_name, elf_handle); + + impl->generateCTsForRun(run_impl_ptr, hwctx, run_uid, kernel_name, elf_handle); } void AieDtracePlugin::runStartImpl(void* run_impl_ptr, void* hwctx, uint32_t run_uid, @@ -208,10 +239,15 @@ namespace xdp { if (!xrt_core::config::get_aie_dtrace()) return; - (void)run_impl_ptr; - (void)hwctx; - (void)run_uid; - (void)kernel_name; + auto impl = findImpl(hwctx); + if (!impl) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: no implementation for hwctx in runStartHook; " + "this inference will not be profiled."); + return; + } + + impl->applyCTForRun(run_impl_ptr, hwctx, run_uid, kernel_name); } void AieDtracePlugin::runWaitImpl(void* run_impl_ptr, void* hwctx, uint32_t run_uid, @@ -229,6 +265,11 @@ namespace xdp { void AieDtracePlugin::endPoll() { + std::lock_guard lock(implMutex); + + for (const auto& kv : handleToAIEDtraceImpl) + kv.second->reportUnusedSelections(); + // Destroy implementations directly; no counter read or sample offload in teardown. handleToAIEDtraceImpl.clear(); } diff --git a/profile/plugin/aie_dtrace/aie_dtrace_plugin.h b/profile/plugin/aie_dtrace/aie_dtrace_plugin.h index b57e07af..2f7232e1 100644 --- a/profile/plugin/aie_dtrace/aie_dtrace_plugin.h +++ b/profile/plugin/aie_dtrace/aie_dtrace_plugin.h @@ -4,6 +4,10 @@ #ifndef XDP_AIE_DTRACE_PLUGIN_DOT_H #define XDP_AIE_DTRACE_PLUGIN_DOT_H +#include +#include +#include + #include "xdp/profile/plugin/aie_dtrace/aie_dtrace_impl.h" #include "xdp/profile/plugin/aie_dtrace/aie_dtrace_metadata.h" #include "xdp/profile/plugin/vp_base/vp_base_plugin.h" @@ -25,6 +29,10 @@ namespace xdp { // aie_dtrace_cb.cpp must NOT call these directly; it calls the // public XDPPlugin::run*Hook wrappers, which filter out runs // submitted by XDP plugins themselves before delegating here. + // + // The CT files are built here rather than at run start because this is + // where the ELF, and so the SAVE_TIMESTAMPS locations they program, is + // available; run start only picks which of them this inference uses. void runConstructorImpl(void* run_impl_ptr, void* hwctx, uint32_t run_uid, const std::string& kernel_name, void* elf_handle) override; @@ -36,12 +44,23 @@ namespace xdp { private: void writeAll(bool openNewFiles) override; - uint64_t getDeviceIDFromHandle(void* handle); void endPoll(); + // Callers must hold implMutex. + uint64_t getDeviceIDFromHandle(void* handle); + void retireImpl(void* handle, AieDtraceImpl& impl); + + // Returns a shared owner so the caller can use the implementation after + // releasing implMutex: the run hooks do real work (ELF parsing, CT + // generation) that must not serialize every hardware context in the + // process, and teardown may erase the map entry meanwhile. + std::shared_ptr findImpl(void* handle) const; + static bool live; - static bool configuredOnePartition; - std::map> handleToAIEDtraceImpl; + static std::atomic configuredOnePartition; + + mutable std::mutex implMutex; + std::map> handleToAIEDtraceImpl; }; } // namespace xdp diff --git a/profile/plugin/aie_dtrace/util/aie_dtrace_util.cpp b/profile/plugin/aie_dtrace/util/aie_dtrace_util.cpp index 4aee0fe7..2f7072f0 100644 --- a/profile/plugin/aie_dtrace/util/aie_dtrace_util.cpp +++ b/profile/plugin/aie_dtrace/util/aie_dtrace_util.cpp @@ -7,10 +7,15 @@ #include #include +#include "core/common/api/hw_context_int.h" #include "core/common/config_reader.h" #include "core/common/message.h" +#include "core/common/shim/hwctx_handle.h" +#include "xdp/profile/database/static_info/aie_util.h" +#include #include +#include namespace xdp::aie::dtrace { @@ -114,6 +119,59 @@ namespace xdp::aie::dtrace { }; } + PartitionGeometry getPartitionGeometry(void* hwctx) + { + PartitionGeometry geometry; + if (!hwctx) + return geometry; + + boost::property_tree::ptree partitions; + try { + partitions = xdp::aie::getAIEPartitionInfo(hwctx); + } + catch (const std::exception& e) { + xrt_core::message::send(severity_level::warning, "XRT", + std::string("AIE dtrace: Error getting partition info: ") + e.what()); + return geometry; + } + + if (partitions.empty()) + return geometry; + + auto ctx = xrt_core::hw_context_int::create_hw_context_from_implementation(hwctx); + const auto slotIdx = static_cast(ctx)->get_slotidx(); + const auto pid = static_cast(getpid()); + + for (const auto& entry : partitions) { + const auto& partition = entry.second; + if (partition.get("pid", -1) != pid) + continue; + if (partition.get("id", "") != std::to_string(slotIdx)) + continue; + + geometry.valid = true; + geometry.startCol = static_cast(partition.get("start_col", 0)); + geometry.numCols = static_cast(partition.get("num_cols", 0)); + return geometry; + } + + // Drivers that do not report the owning context fall back to the last + // partition, which is what this code did before the owner was reported at + // all. It is only correct when this process owns a single partition. + const auto& last = partitions.back().second; + geometry.valid = true; + geometry.startCol = static_cast(last.get("start_col", 0)); + geometry.numCols = static_cast(last.get("num_cols", 0)); + + xrt_core::message::send(severity_level::debug, "XRT", + "AIE dtrace: Could not match a partition to hardware context slot " + + std::to_string(slotIdx) + "; using the last reported partition (start_col=" + + std::to_string(geometry.startCol) + ", num_cols=" + + std::to_string(geometry.numCols) + ")."); + + return geometry; + } + std::vector parseL2L2DesignPoints(const std::string& spec) { std::vector points; diff --git a/profile/plugin/aie_dtrace/util/aie_dtrace_util.h b/profile/plugin/aie_dtrace/util/aie_dtrace_util.h index 1a9bcb1c..1b202412 100644 --- a/profile/plugin/aie_dtrace/util/aie_dtrace_util.h +++ b/profile/plugin/aie_dtrace/util/aie_dtrace_util.h @@ -18,6 +18,21 @@ namespace xdp::aie::dtrace { // Shim bandwidth metric sets used for Debug.aie_dtrace (not part of standard aie_profile ini). std::map> getBandwidthInterfaceTileEventSets(int hwGen); + // ============================ Partition geometry =========================== + + // The columns one hardware context owns. + struct PartitionGeometry { + bool valid = false; + uint8_t startCol = 0; + uint32_t numCols = 0; + }; + + // Resolves the partition belonging to this hardware context. The underlying + // aie_partition_info query is device-wide, so it also reports partitions of + // other contexts and other processes; picking the wrong one puts every + // counter address in the CT file in somebody else's columns. + PartitionGeometry getPartitionGeometry(void* hwctx); + // ===========================L2L2 transfer metrics ========================================== // Inter-stamp memtile halo dst paths; design points come from xrt.ini. diff --git a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.cpp b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.cpp index 188676c3..481c8ce7 100644 --- a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.cpp +++ b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.cpp @@ -678,14 +678,14 @@ std::vector AieDtraceCTWriter::getShimTileColumns(void* hwctx) } try { - boost::property_tree::ptree aiePartitionPt = xdp::aie::getAIEPartitionInfo(hwctx); - if (aiePartitionPt.empty()) { + const auto partition = aie::dtrace::getPartitionGeometry(hwctx); + if (!partition.valid) { xrt_core::message::send(severity_level::debug, "XRT", "AIE dtrace: No partition info available"); return columns; } - uint8_t numCols = static_cast(aiePartitionPt.back().second.get("num_cols")); + const uint8_t numCols = static_cast(partition.numCols); // Return relative columns (0, 1, 2, ...) for hardware configuration for (uint8_t i = 0; i < numCols; ++i) { @@ -1393,10 +1393,11 @@ void AieDtraceCTWriter::appendComputeIoBoundConfig( void AieDtraceCTWriter::appendL2L2Config( void* hwctx, + bool includeL2L2, std::vector& counters, std::vector& beginWrites) { - if (!metadata || !metadata->isL2L2Enabled()) + if (!includeL2L2) return; if (!hwctx) { @@ -1405,22 +1406,11 @@ void AieDtraceCTWriter::appendL2L2Config( return; } - boost::property_tree::ptree aiePartitionPt; - try { - aiePartitionPt = xdp::aie::getAIEPartitionInfo(hwctx); - } - catch (const std::exception& e) { - xrt_core::message::send(severity_level::warning, "XRT", - std::string("AIE dtrace: Error getting partition info for L2-L2: ") + e.what()); - return; - } - if (aiePartitionPt.empty()) + const auto partition = aie::dtrace::getPartitionGeometry(hwctx); + if (!partition.valid || partition.numCols == 0) return; - const uint32_t numCols = - static_cast(aiePartitionPt.back().second.get("num_cols", 0)); - if (numCols == 0) - return; + const uint32_t numCols = partition.numCols; const auto instrumentPoints = aie::dtrace::parseL2L2DesignPoints( profiling_runtime_config::resolveMemoryTileInputPorts()); if (instrumentPoints.empty()) @@ -1473,20 +1463,15 @@ void AieDtraceCTWriter::appendL2L2Config( bool AieDtraceCTWriter::appendMemTileConfig( void* hwctx, const std::string& metricSet, uint8_t channel, + const std::vector& requestedColumns, std::vector& counters, std::vector& beginWrites) { - // Mem tiles occupy the same columns as the shim tiles, so the partition column - // discovery is shared. A setting that named one column instead of "all" leaves those - // columns in the config map. - std::vector columns; - if (metadata->isMemTileAllColumns()) { - columns = getShimTileColumns(hwctx); - } - else { - for (const auto& tileMetric : - metadata->getConfigMetricsVec(static_cast(module_type::mem_tile))) - columns.push_back(tileMetric.first.col); - } + // An empty list means every mem tile column. Those occupy the same columns + // as the shim tiles, so the partition column discovery is shared. The list + // comes from this inference's selection: the shared config map only describes + // the first inference. + const std::vector columns = requestedColumns.empty() + ? getShimTileColumns(hwctx) : requestedColumns; if (columns.empty()) { xrt_core::message::send(severity_level::warning, "XRT", @@ -1529,12 +1514,7 @@ bool AieDtraceCTWriter::generateCT( const std::string& outputPath, void* hwctx, const std::vector& opLocations, - bool includeBandwidth, - const std::string& bandwidthMetricSet, - uint8_t bandwidthChannel, - const std::string& coreMetricSet, - const std::string& memTileMetricSet, - uint8_t memTileChannel) + const MetricSelection& selection) { if (opLocations.empty()) { xrt_core::message::send(severity_level::debug, "XRT", @@ -1557,30 +1537,31 @@ bool AieDtraceCTWriter::generateCT( // of tiles in column 0. Memtile L2-L2 counters are appended when enabled. // filterCountersByColumn keys by column, so all land in the matching UC group and read // distinct addresses. - if (includeBandwidth) - appendBandwidthConfig(hwctx, bandwidthMetricSet, bandwidthChannel, allCounters, beginBlockWrites); + if (selection.includeBandwidth) + appendBandwidthConfig(hwctx, selection.bandwidthMetricSet, selection.bandwidthChannel, + allCounters, beginBlockWrites); - if (coreMetricSet == "compute_io_bound") + if (selection.coreMetricSet == "compute_io_bound") appendComputeIoBoundConfig(allCounters, beginBlockWrites); - else if (!coreMetricSet.empty()) + else if (!selection.coreMetricSet.empty()) xrt_core::message::send(severity_level::warning, "XRT", - "AIE dtrace: Unsupported core (aie) tile metric set '" + coreMetricSet + "AIE dtrace: Unsupported core (aie) tile metric set '" + selection.coreMetricSet + "'; no core tile counters configured."); - // Both mem tile families program the same performance counters, so at most one of - // them is ever configured: the metadata resolves the contention and clears the - // per-tile metric set when L2-L2 wins. - appendL2L2Config(hwctx, allCounters, beginBlockWrites); + // Both mem tile families program the same performance counters, so a single + // selection carries at most one of them. L2-L2 is per inference; the per-tile + // metric set is selection.memTileMetricSet. + appendL2L2Config(hwctx, selection.includeL2L2, allCounters, beginBlockWrites); // Mem tile counters live on rows between the shim and the core tiles, so they land in // the same column-keyed UC groups as the other two families. - if ((memTileMetricSet == "output_channels_details") - || (memTileMetricSet == "mm2s_channels_details")) - appendMemTileConfig(hwctx, memTileMetricSet, memTileChannel, allCounters, - beginBlockWrites); - else if (!memTileMetricSet.empty()) + if ((selection.memTileMetricSet == "output_channels_details") + || (selection.memTileMetricSet == "mm2s_channels_details")) + appendMemTileConfig(hwctx, selection.memTileMetricSet, selection.memTileChannel, + selection.memTileColumns, allCounters, beginBlockWrites); + else if (!selection.memTileMetricSet.empty()) xrt_core::message::send(severity_level::warning, "XRT", - "AIE dtrace: Unsupported mem tile metric set '" + memTileMetricSet + "AIE dtrace: Unsupported mem tile metric set '" + selection.memTileMetricSet + "'; no mem tile counters configured."); if (allCounters.empty()) { @@ -1605,9 +1586,12 @@ bool AieDtraceCTWriter::generateBandwidthCT( const std::string& metricSet, uint8_t channel) { - return generateCT(outputPath, hwctx, opLocations, - /*includeBandwidth=*/true, metricSet, channel, - /*coreMetricSet=*/""); + MetricSelection selection; + selection.includeBandwidth = true; + selection.bandwidthMetricSet = metricSet; + selection.bandwidthChannel = channel; + + return generateCT(outputPath, hwctx, opLocations, selection); } namespace { diff --git a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.h b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.h index 5274c3bb..e8904148 100644 --- a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.h +++ b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ct_writer.h @@ -21,6 +21,7 @@ namespace xdp { class VPDatabase; class AieDtraceMetadata; struct AIECounter; +struct MetricSelection; /** * @brief Information about a SAVE_TIMESTAMPS instruction found in ASM files @@ -191,28 +192,23 @@ class AieDtraceCTWriter { * bandwidth counters and the core-tile counters are emitted into the same CT * file. * + * The selection is passed in rather than read back off the metadata because + * a single hardware context can generate one CT per inference, each with a + * different set of counters. + * * @param outputPath Full path for the generated CT file * @param hwctx Hardware context handle for partition info access * @param opLocations Vector of op_loc from aiebu_assembler::get_op_locations - * @param includeBandwidth Emit interface-tile bandwidth counters - * @param bandwidthMetricSet Bandwidth metric set (used when includeBandwidth) - * @param bandwidthChannel DMA channel for detailed_ddr_*_bandwidth sets - * @param coreMetricSet Core (aie) tile metric set to emit, or empty for none. - * Supported: compute_io_bound - * @param memTileMetricSet Mem tile (L2) metric set to emit, or empty for none. - * Supported: output_channels_details, mm2s_channels_details - * @param memTileChannel MM2S channel (0-5) monitored by the mem tile metric set + * @param selection Metric sets and DMA channels for this one CT file. + * Mem tile counters (output_channels_details, + * mm2s_channels_details) are selection.memTileMetricSet + * and selection.memTileChannel. * @return true if CT file was generated successfully, false otherwise */ bool generateCT(const std::string& outputPath, void* hwctx, const std::vector& opLocations, - bool includeBandwidth, - const std::string& bandwidthMetricSet, - uint8_t bandwidthChannel, - const std::string& coreMetricSet, - const std::string& memTileMetricSet = "", - uint8_t memTileChannel = 0); + const MetricSelection& selection); private: /** @@ -372,6 +368,7 @@ class AieDtraceCTWriter { * @param beginWrites [in,out] Accumulated begin-block register writes */ void appendL2L2Config(void* hwctx, + bool includeL2L2, std::vector& counters, std::vector& beginWrites); @@ -385,11 +382,13 @@ class AieDtraceCTWriter { * @param hwctx Hardware context handle for partition column discovery * @param metricSet Mem tile metric set * @param channel MM2S channel (0-5) to monitor + * @param columns Columns to program. Empty means every mem tile column in the partition. * @param counters [in,out] Accumulated counter list * @param beginWrites [in,out] Accumulated begin-block register writes * @return true if mem tile config was appended */ bool appendMemTileConfig(void* hwctx, const std::string& metricSet, uint8_t channel, + const std::vector& columns, std::vector& counters, std::vector& beginWrites); /** diff --git a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.cpp b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.cpp index 4d7fc5bf..4d025c1e 100644 --- a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.cpp +++ b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.cpp @@ -16,16 +16,41 @@ #include "xdp/profile/database/static_info/aie_util.h" -#include +#include #include +#include #include +#include namespace xdp { using severity_level = xrt_core::message::severity_level; - static constexpr int SHIM_MODULE_IDX = static_cast(module_type::shim); - static constexpr int CORE_MODULE_IDX = static_cast(module_type::core); - static constexpr int MEM_TILE_MODULE_IDX = static_cast(module_type::mem_tile); + namespace { + + // Hands a CT file to XRT for this run object. "what" names the caller's + // unit of work for the log line, since this runs both at run construction + // and at run start. + void programCT(void* run_impl_ptr, uint32_t run_uid, const std::string& ctFile, + const std::string& what) + { + try { + xrt_core::kernel_int::set_dtrace_control_file( + static_cast(run_impl_ptr), ctFile); + + std::stringstream msg; + msg << "AIE dtrace: " << what << " (run uid=" << run_uid << ") will use CT '" + << ctFile << "'"; + xrt_core::message::send(severity_level::info, "XRT", msg.str()); + } + catch (const std::exception& e) { + std::stringstream msg; + msg << "AIE dtrace: Could not set CT file '" << ctFile << "' for run uid=" << run_uid + << ": " << e.what(); + xrt_core::message::send(severity_level::warning, "XRT", msg.str()); + } + } + + } // namespace AieDtrace_VE2Impl::AieDtrace_VE2Impl(VPDatabase* database, std::shared_ptr metadata, @@ -82,133 +107,240 @@ namespace xdp { } } - void AieDtrace_VE2Impl::generateCTForRun(void* run_impl_ptr, void* hwctx, uint32_t run_uid, - const std::string& kernel_name, - void* elf_handle) + void AieDtrace_VE2Impl::generateCTsForRun(void* run_impl_ptr, void* hwctx, uint32_t run_uid, + const std::string& kernel_name, + void* elf_handle) { if (!xrt_core::config::get_aie_dtrace()) return; - auto ctx = xrt_core::hw_context_int::create_hw_context_from_implementation(hwctx); - auto slotIdx = static_cast(ctx)->get_slotidx(); + const auto& selections = metadata->getMetricSelections(); + if (selections.empty()) + return; - std::string filename = "aie_dtrace_ctx_" + std::to_string(slotIdx) - + "_run_" + std::to_string(run_uid) + ".ct"; - std::string outputPath = (std::filesystem::current_path() / filename).string(); + if (std::all_of(selections.begin(), selections.end(), + [](const MetricSelection& s) { return s.empty(); })) { + xrt_core::message::send(severity_level::info, "XRT", + "AIE dtrace: No metrics configured; skipping CT generation."); + return; + } + + std::string ctFile; + + { + std::lock_guard lock(m_mutex); + + // Every run object of a kernel shares the same ELF, so the CT files only + // have to be built for the first one. + if (!m_ct_files.count(kernel_name)) + generateCTFiles(hwctx, run_uid, kernel_name, elf_handle); + + // A single metric set describes every inference, so the CT is programmed + // once here and run start has nothing left to select. A multi-inference + // sequence cannot be resolved yet, since which CT applies depends on the + // inference number, which is only known at start. + if (!metadata->isMultiInference()) { + if (const auto* ctFiles = findCTFiles(kernel_name)) + ctFile = ctFiles->front(); + } + } + + if (!ctFile.empty()) + programCT(run_impl_ptr, run_uid, ctFile, "Kernel '" + kernel_name + "'"); + } + + // Builds one CT per configured inference and records them under kernel_name. + void AieDtrace_VE2Impl::generateCTFiles(void* hwctx, uint32_t run_uid, + const std::string& kernel_name, void* elf_handle) + { + const auto& selections = metadata->getMetricSelections(); computeOpLocations(elf_handle, kernel_name); - auto it = m_op_locations_cache.find(kernel_name); - if (it == m_op_locations_cache.end() || it->second.empty()) { + auto opLocations = m_op_locations_cache.find(kernel_name); + if (opLocations == m_op_locations_cache.end() || opLocations->second.empty()) { xrt_core::message::send(severity_level::debug, "XRT", "AIE dtrace: No op_locations for kernel '" + kernel_name + "'; skipping CT generation."); return; } - boost::property_tree::ptree aiePartitionPt = xdp::aie::getAIEPartitionInfo(hwctx); - uint8_t partitionStartCol = aiePartitionPt.empty() ? 0 - : static_cast(aiePartitionPt.back().second.get("start_col")); + auto ctx = xrt_core::hw_context_int::create_hw_context_from_implementation(hwctx); + const auto slotIdx = static_cast(ctx)->get_slotidx(); - AieDtraceCTWriter ctWriter(db, metadata, deviceID, partitionStartCol); + const auto partition = aie::dtrace::getPartitionGeometry(hwctx); + AieDtraceCTWriter ctWriter(db, metadata, deviceID, partition.startCol); - // Determine which metric families are configured for this run. Both the - // interface-tile bandwidth metrics and one core-tile metric set can be - // emitted into the same per-run CT file. - std::string coreMetricSet; - for (const auto& tc : metadata->getConfigMetricsVec(CORE_MODULE_IDX)) { - coreMetricSet = tc.second; - break; - } + std::vector ctFiles(selections.size()); + size_t generated = 0; - // Mem tile (L2) metrics carry the MM2S channel to monitor in configChannel0, the - // same way the detailed_ddr_*_bandwidth sets carry theirs. Every configured mem tile - // shares one metric set and one channel, so the first entry speaks for all of them. - std::string memTileMetricSet; - uint8_t memTileChannel = 0; - auto memTileConfigMetrics = metadata->getConfigMetricsVec(MEM_TILE_MODULE_IDX); - if (!memTileConfigMetrics.empty()) { - memTileMetricSet = memTileConfigMetrics.front().second; - auto memTileChannels = metadata->getConfigChannel0(); - const auto& memTile = memTileConfigMetrics.front().first; - for (const auto& tc : memTileChannels) { - if ((tc.first.col == memTile.col) && (tc.first.row == memTile.row)) { - memTileChannel = tc.second; - break; - } + for (size_t i = 0; i < selections.size(); ++i) { + const auto& selection = selections[i]; + + if (selection.empty()) { + xrt_core::message::send(severity_level::info, "XRT", + "AIE dtrace: No metrics configured for inference " + + std::to_string(i) + + " of kernel '" + kernel_name + "'; no CT will be generated for it."); + continue; } - xrt_core::message::send(severity_level::info, "XRT", - "AIE dtrace: Using mem tile metric set '" + memTileMetricSet + "' (MM2S channel " - + std::to_string(memTileChannel) + ") from configuration"); - } - // Interface-tile bandwidth metrics are configured by default unless the user - // turned interface tiles off (which leaves the shim config map empty). - auto shimConfigMetrics = metadata->getConfigMetricsVec(SHIM_MODULE_IDX); - bool includeBandwidth = !shimConfigMetrics.empty(); - std::string bandwidthMetricSet = "peak_read_bandwidth"; - uint8_t bandwidthChannel = 0; - if (includeBandwidth) { - bandwidthMetricSet = shimConfigMetrics.front().second; - // The detailed_ddr_*_bandwidth metric sets carry a DMA channel in their - // ":" suffix (stored in configChannel0). Match by column/row. - auto configChannel0 = metadata->getConfigChannel0(); - const auto& shimTile = shimConfigMetrics.front().first; - for (const auto& tc : configChannel0) { - if ((tc.first.col == shimTile.col) && (tc.first.row == shimTile.row)) { - bandwidthChannel = tc.second; - break; - } + // run_uid is the run object whose constructor builds these files. Two + // models have different ELFs and different run ids, so neither + // constructor replaces the other's control trace. Later run objects of + // this kernel reuse the set stored under kernel_name. + const std::string filename = "aie_dtrace_ctx_" + std::to_string(slotIdx) + + "_run_" + std::to_string(run_uid) + + "_inf_" + + std::to_string(i) + + ".ct"; + const auto finalPath = std::filesystem::current_path() / filename; + + // Written under a temporary name and renamed into place so a run start on + // another thread can never hand aiebu a half-written CT file. + const auto tempPath = std::filesystem::path(finalPath).concat( + ".tmp." + std::to_string(run_uid)); + + if (!ctWriter.generateCT(tempPath.string(), hwctx, opLocations->second, selection)) + continue; + + std::error_code ec; + std::filesystem::rename(tempPath, finalPath, ec); + if (ec) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: Could not move generated CT into place ('" + finalPath.string() + + "'): " + ec.message()); + std::filesystem::remove(tempPath, ec); + continue; } - xrt_core::message::send(severity_level::info, "XRT", - "AIE dtrace: Using interface tile metric set '" + bandwidthMetricSet + "' (channel " - + std::to_string(bandwidthChannel) + ") from configuration"); + + ctFiles[i] = finalPath.string(); + ++generated; + + xrt_core::message::send(severity_level::debug, "XRT", + "AIE dtrace: CT generated for kernel '" + kernel_name + "' inference " + + std::to_string(i) + + " (" + selection.describe() + "): " + ctFiles[i]); } - if (!includeBandwidth && coreMetricSet.empty() && memTileMetricSet.empty() - && !metadata->isL2L2Enabled()) { - xrt_core::message::send(severity_level::info, "XRT", - "AIE dtrace: No metrics configured; skipping CT generation."); + if (generated == 0) { + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: No CT files could be generated for kernel '" + kernel_name + + "'; dtrace data will not be collected for it."); return; } - if (!ctWriter.generateCT(outputPath, hwctx, it->second, - includeBandwidth, bandwidthMetricSet, bandwidthChannel, - coreMetricSet, memTileMetricSet, memTileChannel)) + aie::dtrace::initDtraceOutputConfig(); + m_ct_files.emplace(kernel_name, std::move(ctFiles)); + } + + const std::vector* + AieDtrace_VE2Impl::findCTFiles(const std::string& kernel_name) const + { + auto itr = m_ct_files.find(kernel_name); + return (itr == m_ct_files.end()) ? nullptr : &itr->second; + } + + void AieDtrace_VE2Impl::applyCTForRun(void* run_impl_ptr, void* /*hwctx*/, uint32_t run_uid, + const std::string& kernel_name) + { + if (!xrt_core::config::get_aie_dtrace()) return; - aie::dtrace::initDtraceOutputConfig(); + // A single metric set was programmed onto this run at construction and + // covers every one of its inferences, so there is nothing to select here. + if (!metadata->isMultiInference()) + return; - std::stringstream genMsg; - genMsg << "AIE dtrace: CT generated for kernel '" << kernel_name << "' ("; - if (includeBandwidth) - genMsg << "interface_tile=" << bandwidthMetricSet; - if (includeBandwidth && (!coreMetricSet.empty() || metadata->isL2L2Enabled())) - genMsg << ", "; - if (!coreMetricSet.empty()) - genMsg << "aie_tile=" << coreMetricSet; - if (!coreMetricSet.empty() && metadata->isL2L2Enabled()) - genMsg << ", "; - if (metadata->isL2L2Enabled()) - genMsg << "memtile=input_ports"; - if (!memTileMetricSet.empty()) - genMsg << ((includeBandwidth || !coreMetricSet.empty()) ? ", " : "") - << "memtile=" << memTileMetricSet << ":ch" - << static_cast(memTileChannel); - genMsg << ")"; - xrt_core::message::send(severity_level::debug, "XRT", genMsg.str()); - - auto* run_impl = static_cast(run_impl_ptr); - try { - xrt_core::kernel_int::set_dtrace_control_file(run_impl, outputPath); - std::stringstream msg; - msg << "AIE dtrace: Set per-run CT file '" << outputPath - << "' for run uid=" << run_uid << " ctx slot=" << slotIdx; - xrt_core::message::send(severity_level::info, "XRT", msg.str()); + std::string ctFile; + uint64_t inferenceNumber = 0; + + { + std::lock_guard lock(m_mutex); + + // 0-based. The map keeps the count of inferences started, so the value + // returned here is the index of this one and the stored value is the + // count reportUnusedSelections reads. Each kernel is profiled only for + // its first profile_runs entries, which is at most four. + inferenceNumber = m_inference_counts[kernel_name]++; + + const auto* ctFiles = findCTFiles(kernel_name); + if (!ctFiles) { + // Only on the kernel's first inference: an application that runs + // thousands of them should not get thousands of identical warnings. + if (inferenceNumber == 0) + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: No CT files were generated for kernel '" + kernel_name + + "'; its inferences will not be profiled."); + return; + } + + if (inferenceNumber >= ctFiles->size()) { + // Warn only on the first inference past the window; a long-running + // application would otherwise log this on every remaining one. + if (inferenceNumber == ctFiles->size()) + xrt_core::message::send(severity_level::warning, "XRT", + "AIE dtrace: Kernel '" + kernel_name + "' has run more than " + + std::to_string(ctFiles->size()) + + " inferences; inference " + std::to_string(inferenceNumber) + + " onwards will not be profiled."); + } + else { + ctFile = (*ctFiles)[inferenceNumber]; + } } - catch (const std::exception& e) { + + // An empty selection means this inference is outside the configured window, + // or its CT could not be generated. The run object may well be the same one + // a profiled inference used, so dtrace has to be turned off explicitly + // rather than just left alone. + if (ctFile.empty()) { + try { + xrt_core::kernel_int::clear_dtrace_control_file( + static_cast(run_impl_ptr)); + } + catch (const std::exception& e) { + std::stringstream msg; + msg << "AIE dtrace: Could not disable dtrace for run uid=" << run_uid + << ": " << e.what(); + xrt_core::message::send(severity_level::debug, "XRT", msg.str()); + } + return; + } + + programCT(run_impl_ptr, run_uid, ctFile, + "Inference " + std::to_string(inferenceNumber) + " of kernel '" + kernel_name + "'"); + } + + void AieDtrace_VE2Impl::reportUnusedSelections() + { + if (!metadata->isMultiInference()) + return; + + const uint64_t configured = metadata->getMetricSelections().size(); + + std::lock_guard lock(m_mutex); + + for (const auto& entry : m_ct_files) { + const auto& kernel_name = entry.first; + // The map stores how many inferences have started, not the last index. + // The window is inferences 0 through configured-1, at most four. + const uint64_t ran = m_inference_counts.count(kernel_name) + ? m_inference_counts.at(kernel_name) : 0; + const uint64_t profiled = std::min(ran, configured); + + if (profiled >= configured) + continue; + std::stringstream msg; - msg << "AIE dtrace: Could not set per-run CT file: " << e.what(); - xrt_core::message::send(severity_level::debug, "XRT", msg.str()); + msg << "AIE dtrace: Kernel '" << kernel_name << "' ran " << ran + << " inferences, so only " << profiled << " of the " << configured + << " configured profile_runs were collected. Run the kernel at least " + << configured + << " times to collect the whole sequence. Missing:"; + for (uint64_t i = profiled; i < configured; ++i) + msg << "\n inference " << i << ": " + << metadata->getMetricSelections()[i].describe(); + xrt_core::message::send(severity_level::warning, "XRT", msg.str()); } } diff --git a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.h b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.h index 3fa6926a..d7cc14c9 100644 --- a/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.h +++ b/profile/plugin/aie_dtrace/ve2/aie_dtrace_ve2.h @@ -6,6 +6,7 @@ #include #include +#include #include #include @@ -29,14 +30,42 @@ namespace xdp { void updateDevice() override; - void generateCTForRun(void* run_impl_ptr, void* hwctx, uint32_t run_uid, - const std::string& kernel_name, - void* elf_handle) override; + void generateCTsForRun(void* run_impl_ptr, void* hwctx, uint32_t run_uid, + const std::string& kernel_name, + void* elf_handle) override; + + void applyCTForRun(void* run_impl_ptr, void* hwctx, uint32_t run_uid, + const std::string& kernel_name) override; + + void reportUnusedSelections() override; private: + // Callers must hold m_mutex. void computeOpLocations(void* elf_handle, const std::string& kernel_name); + void generateCTFiles(void* hwctx, uint32_t run_uid, const std::string& kernel_name, + void* elf_handle); + const std::vector* findCTFiles(const std::string& kernel_name) const; + + // Serializes every mutable member below. One instance exists per hardware + // context, but XRT lets an application construct and start runs on the + // same context from several threads. + mutable std::mutex m_mutex; std::map> m_op_locations_cache; + + // Kernel name -> one CT path per configured inference, in execution + // order. An entry is empty when that inference's CT could not be + // generated. Generated once per kernel at run construction and then + // shared by every run object of that kernel, since the ELF, and so the + // SAVE_TIMESTAMPS locations, are identical across them. + std::map> m_ct_files; + + // Kernel name -> inferences started so far on this hardware context. + // Counted at start rather than at construction because an application is + // free to reuse one run object for every inference, or to build a pool of + // run objects up front. Only used for a multi-inference sequence; a + // single metric set is programmed at construction and never re-selected. + std::map m_inference_counts; }; } diff --git a/profile/plugin/vp_base/profiling_runtime_config.cpp b/profile/plugin/vp_base/profiling_runtime_config.cpp index b1b7a217..09997792 100644 --- a/profile/plugin/vp_base/profiling_runtime_config.cpp +++ b/profile/plugin/vp_base/profiling_runtime_config.cpp @@ -71,14 +71,102 @@ namespace xdp::profiling_runtime_config { } } + // The four per-inference tile keys, shared by an entry of "profile_runs" + // and by control_instrumentation's own single-configuration form. + const std::set& + tile_keys() + { + static const std::set keys{ + "aie_tile", "mem_tile", "interface_tile", "memory_tile_input_ports" + }; + return keys; + } + + void + warn_unknown_key(const std::string& scope, const std::string& key, + const std::set& known_keys) + { + std::stringstream msg; + msg << "Unknown key '" << scope << "." << key << "' ignored. Supported keys:"; + const char* sep = " "; + for (const auto& k : known_keys) { + msg << sep << k; + sep = ", "; + } + warn(msg.str()); + } + + // Parse one entry of the "profile_runs" array. Unlike the top-level form + // these are not logged individually; the resolved sequence is logged once + // by the caller. + profile_run_t + parse_profile_run(const pt::ptree& tree, size_t index) + { + profile_run_t run; + + for (const auto& kv : tree) { + const auto& key = kv.first; + const auto value = kv.second.get_value(""); + + if (key == "aie_tile") + run.aie_tile = value; + else if (key == "mem_tile") + run.mem_tile = value; + else if (key == "interface_tile") + run.interface_tile = value; + else if (key == "memory_tile_input_ports") + run.memory_tile_input_ports = value; + else + warn_unknown_key("profiling_runtime_config.control_instrumentation.profile_runs[" + + std::to_string(index) + "]", key, tile_keys()); + } + + return run; + } + + // Shorthand form: "profile_runs": "". A super metric set + // names a fixed sequence of per-inference selections, so a user who wants + // the standard report does not have to spell the sequence out. + std::vector + expand_super_metric_set(const std::string& name) + { + std::vector runs; + + if (name != "compute_io_bound") { + warn("Unknown super metric set 'profiling_runtime_config.control_instrumentation." + "profile_runs=" + name + "'. Supported: compute_io_bound. " + "No profile runs configured."); + return runs; + } + + // Compute boundness on the first inference, then the four DDR bandwidth + // channels that cannot share a run because they contend for the same + // interface tile counters. + runs.resize(4); + runs[0].aie_tile = "compute_io_bound"; + runs[0].interface_tile = "detailed_ddr_read_bandwidth:0"; + runs[1].interface_tile = "detailed_ddr_read_bandwidth:1"; + runs[2].interface_tile = "detailed_ddr_write_bandwidth:0"; + runs[3].interface_tile = "detailed_ddr_write_bandwidth:1"; + + info("profiling_runtime_config.control_instrumentation.profile_runs='" + name + + "' expanded to " + std::to_string(runs.size()) + " inferences."); + + return runs; + } + + // A kernel is profiled for at most this many inferences. A longer + // profile_runs list is truncated, and a kernel that keeps running past + // the list is not profiled further. + constexpr size_t max_profiled_inferences = 4; + // Parse the control_instrumentation subtree: copy known string keys into // the returned struct and warn about any unknown keys. control_instrumentation_t parse_control_instrumentation(const pt::ptree& ci_tree) { - static const std::set known_keys{ - "aie_tile", "mem_tile", "interface_tile", "memory_tile_input_ports" - }; + std::set known_keys = tile_keys(); + known_keys.insert("profile_runs"); control_instrumentation_t ci; @@ -107,19 +195,45 @@ namespace xdp::profiling_runtime_config { info("profiling_runtime_config.control_instrumentation.memory_tile_input_ports='" + value + "'"); } - else { - std::stringstream msg; - msg << "Unknown key 'profiling_runtime_config.control_instrumentation." - << key << "' ignored. Supported keys:"; - const char* sep = " "; - for (const auto& k : known_keys) { - msg << sep << k; - sep = ", "; + else if (key == "profile_runs") { + // A ptree node with no children is a scalar, i.e. the shorthand + // string; one with children is the JSON array, whose elements ptree + // exposes as children with empty keys. + if (kv.second.empty()) { + ci.profile_runs = expand_super_metric_set(value); } - warn(msg.str()); + else { + size_t index = 0; + for (const auto& entry : kv.second) + ci.profile_runs.push_back(parse_profile_run(entry.second, index++)); + } + if (ci.profile_runs.size() > max_profiled_inferences) { + warn("profiling_runtime_config.control_instrumentation.profile_runs has " + + std::to_string(ci.profile_runs.size()) + + " entries; only the first " + + std::to_string(max_profiled_inferences) + + " inferences of each kernel are profiled."); + ci.profile_runs.resize(max_profiled_inferences); + } + ci.has_explicit_profile_runs = !ci.profile_runs.empty(); + } + else { + warn_unknown_key("profiling_runtime_config.control_instrumentation", key, known_keys); } } + // Without an explicit sequence the single configuration above is itself + // the one and only profile run, so consumers never have to branch on + // which form the blob used. + if (ci.profile_runs.empty()) { + profile_run_t single; + single.aie_tile = ci.aie_tile; + single.mem_tile = ci.mem_tile; + single.interface_tile = ci.interface_tile; + single.memory_tile_input_ports = ci.memory_tile_input_ports; + ci.profile_runs.push_back(std::move(single)); + } + return ci; } @@ -205,7 +319,8 @@ namespace xdp::profiling_runtime_config { out.has_ci = out.ci.aie_tile.has_value() || out.ci.mem_tile.has_value() || out.ci.interface_tile.has_value() - || out.ci.memory_tile_input_ports.has_value(); + || out.ci.memory_tile_input_ports.has_value() + || out.ci.has_explicit_profile_runs; } if (const auto et_opt = root.get_child_optional("event_trace")) { @@ -259,6 +374,12 @@ namespace xdp::profiling_runtime_config { return get_parsed().ci; } + const std::vector& + profile_runs() + { + return get_parsed().ci.profile_runs; + } + std::string resolveMemoryTileInputPorts() { diff --git a/profile/plugin/vp_base/profiling_runtime_config.h b/profile/plugin/vp_base/profiling_runtime_config.h index 7271b000..e63405a1 100644 --- a/profile/plugin/vp_base/profiling_runtime_config.h +++ b/profile/plugin/vp_base/profiling_runtime_config.h @@ -7,6 +7,7 @@ #include #include #include +#include #include "xdp/config.h" @@ -28,14 +29,51 @@ // Example blob: // {"control_instrumentation":{"aie_tile":"func_stalls","mem_tile":"input_ports","interface_tile":"ddr_bandwidth", // "memory_tile_input_ports":"{1,1:2},{5,1:1}"},"event_trace":{"tile_based_aie_tile_metrics":"all:functions"}} +// +// control_instrumentation may instead carry "profile_runs", which collects the +// metric sets that used to require one design run each into a single run: entry +// i is applied to inference i of each kernel. The first inference is 0, and at +// most four inferences of a kernel are profiled. Either an explicit array: +// "control_instrumentation": { +// "profile_runs": [ +// {"aie_tile":"compute_io_bound","interface_tile":"detailed_ddr_read_bandwidth:0"}, +// {"interface_tile":"detailed_ddr_read_bandwidth:1"}, +// {"interface_tile":"detailed_ddr_write_bandwidth:0"}, +// {"interface_tile":"detailed_ddr_write_bandwidth:1"}]} +// or the super-metric-set shorthand that expands to that same sequence: +// "control_instrumentation": {"profile_runs": "compute_io_bound"} namespace xdp::profiling_runtime_config { + // One inference's metric selection. Carries the same four tile keys as the + // single-configuration form, so an entry of "profile_runs" is parsed exactly + // like control_instrumentation itself. + struct profile_run_t { + std::optional aie_tile; // maps to "core" module internally + std::optional mem_tile; // maps to "mem_tile" module internally + std::optional interface_tile; // maps to "shim" module internally + std::optional memory_tile_input_ports; // L2-L2 {column,row:port} list + }; + struct control_instrumentation_t { std::optional aie_tile; // maps to "core" module internally std::optional mem_tile; // maps to "mem_tile" module internally std::optional interface_tile; // maps to "shim" module internally std::optional memory_tile_input_ports; // L2-L2 {column,row:port} list + + // Per-inference metric selections in execution order: profile_runs[i] is + // applied to inference i of each kernel. The first inference is 0, and a + // kernel is profiled for at most four inferences. Populated from the + // "profile_runs" array, or from its super-metric-set shorthand. When the + // blob carries no "profile_runs" this holds a single entry synthesized + // from the four fields above, so a consumer can always iterate it rather + // than special-casing the single-configuration form. + std::vector profile_runs; + + // True when "profile_runs" came from the blob rather than being synthesized + // from the four fields above. Distinguishes "the user asked for multiple + // inferences" from "the user gave one configuration the old way". + bool has_explicit_profile_runs = false; }; // Mirrors the AIE_trace_settings.* xrt.ini keys 1:1. When event_trace is @@ -116,6 +154,10 @@ namespace xdp::profiling_runtime_config { // has_control_instrumentation() is false (all members will be empty). XDP_CORE_EXPORT const control_instrumentation_t& control_instrumentation(); + // Per-inference metric selections. Empty when the blob carried no + // control_instrumentation at all; otherwise it always has at least one entry. + XDP_CORE_EXPORT const std::vector& profile_runs(); + // When control_instrumentation carries mem_tile or memory_tile_input_ports, // ports come only from the blob (requires mem_tile "input_ports"). Otherwise // AIE_dtrace_settings.memory_tile_input_ports from xrt.ini is used.