diff --git a/CMakeLists.txt b/CMakeLists.txt index 09eacc34..10e9b18c 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -33,6 +33,7 @@ else() endif() find_package(Threads REQUIRED) +find_package(rocprofiler-sdk PATHS "/opt/rocm/lib/cmake") set(TRACY_PUBLIC_DIR ${CMAKE_CURRENT_SOURCE_DIR}/public) @@ -56,6 +57,10 @@ target_link_libraries( Threads::Threads ${CMAKE_DL_LIBS} ) +if(rocprofiler-sdk_FOUND) + target_compile_definitions(TracyClient PUBLIC TRACY_ROCPROF) + target_link_libraries(TracyClient PUBLIC rocprofiler-sdk::rocprofiler-sdk) +endif() if(TRACY_Fortran) add_library(TracyClientF90 ${TRACY_VISIBILITY} "${TRACY_PUBLIC_DIR}/TracyClient.F90") @@ -142,6 +147,10 @@ set_option(TRACY_VERBOSE "[advanced] Verbose output from the profiler" OFF) mark_as_advanced(TRACY_VERBOSE) set_option(TRACY_DEMANGLE "[advanced] Don't use default demangling function - You'll need to provide your own" OFF) mark_as_advanced(TRACY_DEMANGLE) +if(rocprofiler-sdk_FOUND) + set_option(TRACY_ROCPROF_CALIBRATION "[advanced] Use continuous calibration of the Rocprof GPU time." OFF) + mark_as_advanced(TRACY_ROCPROF_CALIBRATION) +endif() # handle incompatible combinations if(TRACY_MANUAL_LIFETIME AND NOT TRACY_DELAYED_INIT) diff --git a/manual/tracy.tex b/manual/tracy.tex index 2daa52fb..ab36b157 100644 --- a/manual/tracy.tex +++ b/manual/tracy.tex @@ -1707,6 +1707,35 @@ Unlike other GPU backends in Tracy, there is no need to call \texttt{TracyCUDACo To stop profiling, call the \texttt{TracyCUDAStopProfiling(ctx)} macro. +\subsubsection{ROCm} + +On Linux, if rocprofiler-sdk is installed, tracy can automatically trace GPU dispatches and collect +performance counter values. If CMake can't find rocprofiler-sdk, you can set the CMake variable +\texttt{rocprofiler-sdk\_DIR} to point it at the correct module directory. Use the +\texttt{TRACY\_ROCPROF\_COUNTERS} environment variable with the desired counters separated by commas +to control what values are collected. The results will appear for each dispatch in the tool tip and +zone detail window. Results are summed across dimensions. You can get a list of the counters +available for your hardware with this command: +\begin{lstlisting}[language=sh] +rocprofv3 -L +\end{lstlisting} + +\subparagraph{Troubleshooting} +\begin{itemize} +\item If you are taking very long captures, you may see drift between the GPU and + CPU timelines. This may be mitigated by setting the CMake variable + \texttt{TRACY\_ROCPROF\_CALIBRATION}, which will refresh the time synchronization about every + second. +\item The timeline drift may also be affected by network time synchronization, in which case the + drift will be reduced by disabling that, with the advantage that there is no application performance + cost. +\item On some GPUs, you will need to change the the performance level to see non-zero results from + some counters. Use this command: +\begin{lstlisting}[language=sh] +sudo amd-smi set -g 0 -l stable_std +\end{lstlisting} +\end{itemize} + \subsubsection{Multiple zones in one scope} Putting more than one GPU zone macro in a single scope features the same issue as with the \texttt{ZoneScoped} macros, described in section~\ref{multizone} (but this time the variable name is \texttt{\_\_\_tracy\_gpu\_zone}). diff --git a/profiler/src/profiler/TracyView.hpp b/profiler/src/profiler/TracyView.hpp index 73bb4ed0..5792b17f 100644 --- a/profiler/src/profiler/TracyView.hpp +++ b/profiler/src/profiler/TracyView.hpp @@ -46,7 +46,8 @@ constexpr const char* GpuContextNames[] = { "Direct3D 11", "Metal", "Custom", - "CUDA" + "CUDA", + "Rocprof" }; struct MemoryPage; diff --git a/profiler/src/profiler/TracyView_ZoneInfo.cpp b/profiler/src/profiler/TracyView_ZoneInfo.cpp index e4c68c79..29f592a0 100644 --- a/profiler/src/profiler/TracyView_ZoneInfo.cpp +++ b/profiler/src/profiler/TracyView_ZoneInfo.cpp @@ -1580,6 +1580,21 @@ void View::DrawGpuInfoWindow() TextFocused( "Delay to execution:", TimeToString( AdjustGpuTime( ev.GpuStart(), begin, drift ) - ev.CpuStart() ) ); } + if( ctx->notes.contains( ev.query_id ) ) + { + for( auto& p : ctx->notes.at( ev.query_id ) ) + { + if( ctx->noteNames.count( p.first ) ) + { + TextFocused( m_worker.GetString( ctx->noteNames.at( p.first ) ), RealToString( p.second ) ); + } + else + { + TextFocused( RealToString( p.first ), RealToString( p.second ) ); + } + } + } + ImGui::Separator(); std::vector zoneTrace; @@ -2047,6 +2062,21 @@ void View::ZoneTooltip( const GpuEvent& ev ) TextFocused( "Delay to execution:", TimeToString( AdjustGpuTime( ev.GpuStart(), begin, drift ) - ev.CpuStart() ) ); } + if( ctx->notes.contains( ev.query_id ) ) + { + for( auto& p : ctx->notes.at( ev.query_id ) ) + { + if( ctx->noteNames.count( p.first ) ) + { + TextFocused( m_worker.GetString( ctx->noteNames.at( p.first ) ), RealToString( p.second ) ); + } + else + { + TextFocused( RealToString( p.first ), RealToString( p.second ) ); + } + } + } + ImGui::EndTooltip(); } diff --git a/public/TracyClient.cpp b/public/TracyClient.cpp index 6224f48b..e9a01848 100644 --- a/public/TracyClient.cpp +++ b/public/TracyClient.cpp @@ -32,6 +32,10 @@ #include "client/TracyOverride.cpp" #include "client/TracyKCore.cpp" +#ifdef TRACY_ROCPROF +# include "client/TracyRocprof.cpp" +#endif + #if defined(TRACY_HAS_CALLSTACK) # if TRACY_HAS_CALLSTACK == 2 || TRACY_HAS_CALLSTACK == 3 || TRACY_HAS_CALLSTACK == 4 || TRACY_HAS_CALLSTACK == 6 # include "libbacktrace/alloc.cpp" diff --git a/public/client/TracyProfiler.cpp b/public/client/TracyProfiler.cpp index 943b6490..e1b9d50c 100644 --- a/public/client/TracyProfiler.cpp +++ b/public/client/TracyProfiler.cpp @@ -2383,6 +2383,10 @@ static void FreeAssociatedMemory( const QueueItem& item ) tracy_free( (void*)ptr ); break; #endif + case QueueType::GpuAnnotationName: + ptr = MemRead( &item.gpuAnnotationNameFat.ptr ); + tracy_free( (void*)ptr ); + break; #ifdef TRACY_ON_DEMAND case QueueType::MessageAppInfo: case QueueType::GpuContextName: @@ -2598,6 +2602,12 @@ Profiler::DequeueStatus Profiler::Dequeue( moodycamel::ConsumerToken& token ) tracy_free_fast( (void*)ptr ); #endif break; + case QueueType::GpuAnnotationName: + ptr = MemRead( &item->gpuAnnotationNameFat.ptr ); + size = MemRead( &item->gpuAnnotationNameFat.size ); + SendSingleString( (const char*)ptr, size ); + tracy_free_fast( (void*)ptr ); + break; case QueueType::PlotDataInt: case QueueType::PlotDataFloat: case QueueType::PlotDataDouble: @@ -2956,6 +2966,14 @@ Profiler::DequeueStatus Profiler::DequeueSerial() #endif break; } + case QueueType::GpuAnnotationName: + { + ptr = MemRead( &item->gpuAnnotationNameFat.ptr ); + uint16_t size = MemRead( &item->gpuAnnotationNameFat.size ); + SendSingleString( (const char*)ptr, size ); + tracy_free_fast( (void*)ptr ); + break; + } #ifdef TRACY_FIBERS case QueueType::ZoneBegin: case QueueType::ZoneBeginCallstack: diff --git a/public/client/TracyRocprof.cpp b/public/client/TracyRocprof.cpp new file mode 100644 index 00000000..370e42ec --- /dev/null +++ b/public/client/TracyRocprof.cpp @@ -0,0 +1,556 @@ +#include "../server/tracy_robin_hood.h" +#include "TracyProfiler.hpp" +#include "TracyThread.hpp" +#include "tracy/TracyC.h" +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#define ROCPROFILER_CALL( result, msg ) \ + { \ + rocprofiler_status_t CHECKSTATUS = result; \ + if( CHECKSTATUS != ROCPROFILER_STATUS_SUCCESS ) \ + { \ + std::string status_msg = rocprofiler_get_status_string( CHECKSTATUS ); \ + std::cerr << "[" #result "][" << __FILE__ << ":" << __LINE__ << "] " << msg << " failed with error code " \ + << CHECKSTATUS << ": " << status_msg << std::endl; \ + std::stringstream errmsg{}; \ + errmsg << "[" #result "][" << __FILE__ << ":" << __LINE__ << "] " << msg " failure (" << status_msg \ + << ")"; \ + throw std::runtime_error( errmsg.str() ); \ + } \ + } + +namespace +{ + +using kernel_symbol_data_t = rocprofiler_callback_tracing_code_object_kernel_symbol_register_data_t; + +struct DispatchData +{ + int64_t launch_start; + int64_t launch_end; + uint32_t thread_id; + uint16_t query_id; +}; + +struct ToolData +{ + uint32_t version; + const char* runtime_version; + uint32_t priority; + rocprofiler_client_id_t client_id; + uint8_t context_id; + bool init; + uint64_t query_id; + int64_t previous_cpu_time; + tracy::unordered_map client_kernels; + tracy::unordered_map dispatch_data; + tracy::unordered_set counter_names = { "SQ_WAVES", "GL2C_MISS", "GL2C_HIT" }; + std::unique_ptr cal_thread; + std::mutex mut{}; +}; + +using namespace tracy; + +rocprofiler_context_id_t& get_client_ctx() +{ + static rocprofiler_context_id_t ctx{ 0 }; + return ctx; +} + +const char* CTX_NAME = "rocprofv3"; + +uint8_t gpu_context_allocate( ToolData* data ) +{ + + timespec ts; + clock_gettime( CLOCK_BOOTTIME, &ts ); + uint64_t cpu_timestamp = Profiler::GetTime(); + uint64_t gpu_timestamp = ( (uint64_t)ts.tv_sec * 1000000000 ) + ts.tv_nsec; + float timestamp_period = 1.0f; + data->previous_cpu_time = cpu_timestamp; + + // Allocate the process-unique GPU context ID. There's a max of 255 available; + // if we are recreating devices a lot we may exceed that. Don't do that, or + // wrap around and get weird (but probably still usable) numbers. + uint8_t context_id = tracy::GetGpuCtxCounter().fetch_add( 1, std::memory_order_relaxed ); + if( context_id >= 255 ) + { + context_id %= 255; + } + + uint8_t context_flags = 0; +#ifdef TRACY_ROCPROF_CALIBRATION + // Tell tracy we'll be passing calibrated timestamps and not to mess with + // the times. We'll periodically send GpuCalibration events in case the + // times drift. + context_flags |= tracy::GpuContextCalibration; +#endif + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuNewContext ); + tracy::MemWrite( &item->gpuNewContext.cpuTime, cpu_timestamp ); + tracy::MemWrite( &item->gpuNewContext.gpuTime, gpu_timestamp ); + memset( &item->gpuNewContext.thread, 0, sizeof( item->gpuNewContext.thread ) ); + tracy::MemWrite( &item->gpuNewContext.period, timestamp_period ); + tracy::MemWrite( &item->gpuNewContext.context, context_id ); + tracy::MemWrite( &item->gpuNewContext.flags, context_flags ); + tracy::MemWrite( &item->gpuNewContext.type, tracy::GpuContextType::Rocprof ); + tracy::Profiler::QueueSerialFinish(); + } + + // Send the name of the context along. + // NOTE: Tracy will unconditionally free the name so we must clone it here. + // Since internally Tracy will use its own rpmalloc implementation we must + // make sure we allocate from the same source. + size_t name_length = strlen( CTX_NAME ); + char* cloned_name = (char*)tracy::tracy_malloc( name_length ); + memcpy( cloned_name, CTX_NAME, name_length ); + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuContextName ); + tracy::MemWrite( &item->gpuContextNameFat.context, context_id ); + tracy::MemWrite( &item->gpuContextNameFat.ptr, (uint64_t)cloned_name ); + tracy::MemWrite( &item->gpuContextNameFat.size, name_length ); + tracy::Profiler::QueueSerialFinish(); + } + + return context_id; +} + +uint64_t kernel_src_loc( ToolData* data, uint64_t kernel_id ) +{ + uint64_t src_loc = 0; + auto _lk = std::unique_lock{ data->mut }; + rocprofiler_kernel_id_t kid = kernel_id; + if( data->client_kernels.count( kid ) ) + { + auto& sym_data = data->client_kernels[kid]; + const char* name = sym_data.kernel_name; + size_t name_len = strlen( name ); + uint32_t line = 0; + src_loc = tracy::Profiler::AllocSourceLocation( line, NULL, 0, name, name_len, NULL, 0 ); + } + return src_loc; +} + +void record_interval( ToolData* data, rocprofiler_timestamp_t start_timestamp, rocprofiler_timestamp_t end_timestamp, + uint64_t src_loc, rocprofiler_dispatch_id_t dispatch_id ) +{ + + uint16_t query_id = 0; + uint8_t context_id = data->context_id; + + { + auto _lk = std::unique_lock{ data->mut }; + query_id = data->query_id; + data->query_id++; + if( dispatch_id != UINT64_MAX ) + { + DispatchData& dispatch_data = data->dispatch_data[dispatch_id]; + dispatch_data.query_id = query_id; + dispatch_data.thread_id = tracy::GetThreadHandle(); + } + } + + uint64_t cpu_start_time = 0, cpu_end_time = 0; + if( dispatch_id == UINT64_MAX ) + { + cpu_start_time = tracy::Profiler::GetTime(); + cpu_end_time = tracy::Profiler::GetTime(); + } + else + { + auto _lk = std::unique_lock{ data->mut }; + DispatchData& dispatch_data = data->dispatch_data[dispatch_id]; + cpu_start_time = dispatch_data.launch_start; + cpu_end_time = dispatch_data.launch_end; + } + + if( src_loc != 0 ) + { + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuZoneBeginAllocSrcLocSerial ); + tracy::MemWrite( &item->gpuZoneBegin.cpuTime, cpu_start_time ); + tracy::MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)src_loc ); + tracy::MemWrite( &item->gpuZoneBegin.thread, tracy::GetThreadHandle() ); + tracy::MemWrite( &item->gpuZoneBegin.queryId, query_id ); + tracy::MemWrite( &item->gpuZoneBegin.context, context_id ); + tracy::Profiler::QueueSerialFinish(); + } + } + else + { + static const ___tracy_source_location_data src_loc = { NULL, NULL, NULL, 0, 0 }; + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuZoneBeginSerial ); + tracy::MemWrite( &item->gpuZoneBegin.cpuTime, cpu_start_time ); + tracy::MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)&src_loc ); + tracy::MemWrite( &item->gpuZoneBegin.thread, tracy::GetThreadHandle() ); + tracy::MemWrite( &item->gpuZoneBegin.queryId, query_id ); + tracy::MemWrite( &item->gpuZoneBegin.context, context_id ); + tracy::Profiler::QueueSerialFinish(); + } + } + + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuTime ); + tracy::MemWrite( &item->gpuTime.gpuTime, start_timestamp ); + tracy::MemWrite( &item->gpuTime.queryId, query_id ); + tracy::MemWrite( &item->gpuTime.context, context_id ); + tracy::Profiler::QueueSerialFinish(); + } + + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuZoneEndSerial ); + tracy::MemWrite( &item->gpuZoneEnd.cpuTime, cpu_end_time ); + tracy::MemWrite( &item->gpuZoneEnd.thread, tracy::GetThreadHandle() ); + tracy::MemWrite( &item->gpuZoneEnd.queryId, query_id ); + tracy::MemWrite( &item->gpuZoneEnd.context, context_id ); + tracy::Profiler::QueueSerialFinish(); + } + + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuTime ); + tracy::MemWrite( &item->gpuTime.gpuTime, end_timestamp ); + tracy::MemWrite( &item->gpuTime.queryId, query_id ); + tracy::MemWrite( &item->gpuTime.context, context_id ); + tracy::Profiler::QueueSerialFinish(); + } +} + +void record_callback( rocprofiler_dispatch_counting_service_data_t dispatch_data, + rocprofiler_record_counter_t* record_data, size_t record_count, + rocprofiler_user_data_t /*user_data*/, void* callback_data ) +{ + assert( callback_data != nullptr ); + ToolData* data = static_cast( callback_data ); + if( !data->init ) return; + + std::unordered_map sums; + for( size_t i = 0; i < record_count; ++i ) + { + auto _counter_id = rocprofiler_counter_id_t{}; + ROCPROFILER_CALL( rocprofiler_query_record_counter_id( record_data[i].id, &_counter_id ), + "query record counter id" ); + sums[_counter_id.handle] += record_data[i].counter_value; + } + + uint16_t query_id = 0; + uint32_t thread_id = 0; + { + auto _lk = std::unique_lock{ data->mut }; + // An assumption is made here that the counter values are supplied after the dispatch + // complete callback. + assert( data->dispatch_data.count( dispatch_data.dispatch_info.dispatch_id ) ); + DispatchData& ddata = data->dispatch_data[dispatch_data.dispatch_info.dispatch_id]; + query_id = ddata.query_id; + thread_id = ddata.thread_id; + } + + for( auto& p : sums ) + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuZoneAnnotation ); + tracy::MemWrite( &item->zoneAnnotation.noteId, p.first ); + tracy::MemWrite( &item->zoneAnnotation.queryId, query_id ); + tracy::MemWrite( &item->zoneAnnotation.thread, thread_id ); + tracy::MemWrite( &item->zoneAnnotation.value, p.second ); + tracy::MemWrite( &item->zoneAnnotation.context, data->context_id ); + tracy::Profiler::QueueSerialFinish(); + } +} + +/** + * Callback from rocprofiler when an kernel dispatch is enqueued into the HSA queue. + * rocprofiler_counter_config_id_t* is a return to specify what counters to collect + * for this dispatch (dispatch_packet). + */ +void dispatch_callback( rocprofiler_dispatch_counting_service_data_t dispatch_data, + rocprofiler_profile_config_id_t* config, rocprofiler_user_data_t* /*user_data*/, + void* callback_data ) +{ + assert( callback_data != nullptr ); + ToolData* data = static_cast( callback_data ); + if( !data->init ) return; + + /** + * This simple example uses the same profile counter set for all agents. + * We store this in a cache to prevent constructing many identical profile counter + * sets. We first check the cache to see if we have already constructed a counter" + * set for the agent. If we have, return it. Otherwise, construct a new profile counter + * set. + */ + static std::shared_mutex m_mutex = {}; + static std::unordered_map profile_cache = {}; + + auto search_cache = [&]() + { + if( auto pos = profile_cache.find( dispatch_data.dispatch_info.agent_id.handle ); pos != profile_cache.end() ) + { + *config = pos->second; + return true; + } + return false; + }; + + { + auto rlock = std::shared_lock{ m_mutex }; + if( search_cache() ) return; + } + + auto wlock = std::unique_lock{ m_mutex }; + if( search_cache() ) return; + + // GPU Counter IDs + std::vector gpu_counters; + + // Iterate through the agents and get the counters available on that agent + ROCPROFILER_CALL( + rocprofiler_iterate_agent_supported_counters( + dispatch_data.dispatch_info.agent_id, + []( rocprofiler_agent_id_t, rocprofiler_counter_id_t* counters, size_t num_counters, void* user_data ) + { + std::vector* vec = + static_cast*>( user_data ); + for( size_t i = 0; i < num_counters; i++ ) + { + vec->push_back( counters[i] ); + } + return ROCPROFILER_STATUS_SUCCESS; + }, + static_cast( &gpu_counters ) ), + "Could not fetch supported counters" ); + + std::vector collect_counters; + collect_counters.reserve( data->counter_names.size() ); + // Look for the counters contained in counters_to_collect in gpu_counters + for( auto& counter : gpu_counters ) + { + rocprofiler_counter_info_v0_t info; + ROCPROFILER_CALL( + rocprofiler_query_counter_info( counter, ROCPROFILER_COUNTER_INFO_VERSION_0, static_cast( &info ) ), + "Could not query info" ); + if( data->counter_names.count( std::string( info.name ) ) > 0 ) + { + collect_counters.push_back( counter ); + + size_t name_length = strlen( info.name ); + char* cloned_name = (char*)tracy::tracy_malloc( name_length ); + memcpy( cloned_name, info.name, name_length ); + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuAnnotationName ); + tracy::MemWrite( &item->gpuAnnotationNameFat.context, data->context_id ); + tracy::MemWrite( &item->gpuAnnotationNameFat.noteId, counter.handle ); + tracy::MemWrite( &item->gpuAnnotationNameFat.ptr, (uint64_t)cloned_name ); + tracy::MemWrite( &item->gpuAnnotationNameFat.size, name_length ); + tracy::Profiler::QueueSerialFinish(); + } + } + } + + // Create a colleciton profile for the counters + rocprofiler_profile_config_id_t profile = { .handle = 0 }; + ROCPROFILER_CALL( rocprofiler_create_profile_config( dispatch_data.dispatch_info.agent_id, collect_counters.data(), + collect_counters.size(), &profile ), + "Could not construct profile cfg" ); + + profile_cache.emplace( dispatch_data.dispatch_info.agent_id.handle, profile ); + // Return the profile to collect those counters for this dispatch + *config = profile; +} + +void tool_callback_tracing_callback( rocprofiler_callback_tracing_record_t record, rocprofiler_user_data_t* user_data, + void* callback_data ) +{ + assert( callback_data != nullptr ); + ToolData* data = static_cast( callback_data ); + if( !data->init ) return; + + if( record.kind == ROCPROFILER_CALLBACK_TRACING_CODE_OBJECT && + record.operation == ROCPROFILER_CODE_OBJECT_DEVICE_KERNEL_SYMBOL_REGISTER ) + { + auto* sym_data = static_cast( record.payload ); + + if( record.phase == ROCPROFILER_CALLBACK_PHASE_LOAD ) + { + auto _lk = std::unique_lock{ data->mut }; + data->client_kernels.emplace( sym_data->kernel_id, *sym_data ); + } + else if( record.phase == ROCPROFILER_CALLBACK_PHASE_UNLOAD ) + { + auto _lk = std::unique_lock{ data->mut }; + data->client_kernels.erase( sym_data->kernel_id ); + } + } + else if( record.kind == ROCPROFILER_CALLBACK_TRACING_KERNEL_DISPATCH ) + { + auto* rdata = static_cast( record.payload ); + if( record.operation == ROCPROFILER_KERNEL_DISPATCH_ENQUEUE ) + { + if( record.phase == ROCPROFILER_CALLBACK_PHASE_ENTER ) + { + auto _lk = std::unique_lock{ data->mut }; + data->dispatch_data[rdata->dispatch_info.dispatch_id].launch_start = tracy::Profiler::GetTime(); + } + else if( record.phase == ROCPROFILER_CALLBACK_PHASE_EXIT ) + { + auto _lk = std::unique_lock{ data->mut }; + data->dispatch_data[rdata->dispatch_info.dispatch_id].launch_end = tracy::Profiler::GetTime(); + } + } + else if( record.operation == ROCPROFILER_KERNEL_DISPATCH_COMPLETE ) + { + uint64_t src_loc = kernel_src_loc( data, rdata->dispatch_info.kernel_id ); + record_interval( data, rdata->start_timestamp, rdata->end_timestamp, src_loc, + rdata->dispatch_info.dispatch_id ); + } + } + else if( record.kind == ROCPROFILER_CALLBACK_TRACING_MEMORY_COPY && + record.operation != ROCPROFILER_MEMORY_COPY_NONE && record.phase == ROCPROFILER_CALLBACK_PHASE_EXIT ) + { + auto* rdata = static_cast( record.payload ); + const char* name = nullptr; + switch( record.operation ) + { + case ROCPROFILER_MEMORY_COPY_DEVICE_TO_DEVICE: + name = "DeviceToDeviceCopy"; + break; + case ROCPROFILER_MEMORY_COPY_DEVICE_TO_HOST: + name = "DeviceToHostCopy"; + break; + case ROCPROFILER_MEMORY_COPY_HOST_TO_DEVICE: + name = "HostToDeviceCopy"; + break; + case ROCPROFILER_MEMORY_COPY_HOST_TO_HOST: + name = "HostToHostCopy"; + break; + } + size_t name_len = strlen( name ); + uint64_t src_loc = tracy::Profiler::AllocSourceLocation( 0, NULL, 0, name, name_len, NULL, 0 ); + record_interval( data, rdata->start_timestamp, rdata->end_timestamp, src_loc, UINT64_MAX ); + } +} + +void calibration_thread( void* ptr ) +{ + while( !TracyIsStarted ) + ; + ToolData* data = static_cast( ptr ); + data->context_id = gpu_context_allocate( data ); + const char* user_counters = GetEnvVar( "TRACY_ROCPROF_COUNTERS" ); + if( user_counters ) + { + data->counter_names.clear(); + std::stringstream ss( user_counters ); + std::string counter; + while( std::getline( ss, counter, ',' ) ) data->counter_names.insert( counter ); + } + data->init = true; + +#ifdef TRACY_ROCPROF_CALIBRATION + while( data->init ) + { + sleep( 1 ); + + timespec ts; + // HSA performs a linear interpolation of GPU time to CLOCK_BOOTTIME. However, this is + // subject to network time updates and can drift relative to tracy's clock. + clock_gettime( CLOCK_BOOTTIME, &ts ); + int64_t cpu_timestamp = Profiler::GetTime(); + int64_t gpu_timestamp = ts.tv_nsec + ts.tv_sec * 1e9L; + + if( cpu_timestamp > data->previous_cpu_time ) + { + auto* item = tracy::Profiler::QueueSerial(); + tracy::MemWrite( &item->hdr.type, tracy::QueueType::GpuCalibration ); + tracy::MemWrite( &item->gpuCalibration.gpuTime, gpu_timestamp ); + tracy::MemWrite( &item->gpuCalibration.cpuTime, cpu_timestamp ); + tracy::MemWrite( &item->gpuCalibration.cpuDelta, cpu_timestamp - data->previous_cpu_time ); + tracy::MemWrite( &item->gpuCalibration.context, data->context_id ); + tracy::Profiler::QueueSerialFinish(); + data->previous_cpu_time = cpu_timestamp; + } + } +#endif +} + +int tool_init( rocprofiler_client_finalize_t fini_func, void* user_data ) +{ + ToolData* data = static_cast( user_data ); + data->cal_thread = std::make_unique( calibration_thread, data ); + + ROCPROFILER_CALL( rocprofiler_create_context( &get_client_ctx() ), "context creation failed" ); + + ROCPROFILER_CALL( rocprofiler_configure_callback_dispatch_counting_service( get_client_ctx(), dispatch_callback, + user_data, record_callback, user_data ), + "Could not setup counting service" ); + + rocprofiler_tracing_operation_t ops[] = { ROCPROFILER_CODE_OBJECT_DEVICE_KERNEL_SYMBOL_REGISTER }; + ROCPROFILER_CALL( rocprofiler_configure_callback_tracing_service( get_client_ctx(), + ROCPROFILER_CALLBACK_TRACING_CODE_OBJECT, ops, 1, + tool_callback_tracing_callback, user_data ), + "callback tracing service failed to configure" ); + + rocprofiler_tracing_operation_t ops2[] = { ROCPROFILER_KERNEL_DISPATCH_COMPLETE, + ROCPROFILER_KERNEL_DISPATCH_ENQUEUE }; + ROCPROFILER_CALL( + rocprofiler_configure_callback_tracing_service( get_client_ctx(), ROCPROFILER_CALLBACK_TRACING_KERNEL_DISPATCH, + ops2, 2, tool_callback_tracing_callback, user_data ), + "callback tracing service failed to configure" ); + + ROCPROFILER_CALL( rocprofiler_configure_callback_tracing_service( get_client_ctx(), + ROCPROFILER_CALLBACK_TRACING_MEMORY_COPY, nullptr, + 0, tool_callback_tracing_callback, user_data ), + "callback tracing service failed to configure" ); + + ROCPROFILER_CALL( rocprofiler_start_context( get_client_ctx() ), "start context" ); + return 0; +} + +void tool_fini( void* tool_data_v ) +{ + rocprofiler_stop_context( get_client_ctx() ); + + ToolData* data = static_cast( tool_data_v ); + data->init = false; + data->cal_thread.reset(); +} +} + +extern "C" +{ + rocprofiler_tool_configure_result_t* rocprofiler_configure( uint32_t version, const char* runtime_version, + uint32_t priority, rocprofiler_client_id_t* client_id ) + { + // If not the first tool to register, indicate that the tool doesn't want to do anything + if( priority > 0 ) return nullptr; + + // (optional) Provide a name for this tool to rocprofiler + client_id->name = "Tracy"; + + // (optional) create configure data + static ToolData data = ToolData{ version, runtime_version, priority, *client_id, 0, false, 0, 0 }; + + // construct configure result + static auto cfg = rocprofiler_tool_configure_result_t{ sizeof( rocprofiler_tool_configure_result_t ), + &tool_init, &tool_fini, static_cast( &data ) }; + + return &cfg; + } +} diff --git a/public/common/TracyProtocol.hpp b/public/common/TracyProtocol.hpp index d23b1d08..ff38686f 100644 --- a/public/common/TracyProtocol.hpp +++ b/public/common/TracyProtocol.hpp @@ -9,7 +9,7 @@ namespace tracy constexpr unsigned Lz4CompressBound( unsigned isize ) { return isize + ( isize / 255 ) + 16; } -enum : uint32_t { ProtocolVersion = 75 }; +enum : uint32_t { ProtocolVersion = 76 }; enum : uint16_t { BroadcastVersion = 3 }; using lz4sz_t = uint32_t; diff --git a/public/common/TracyQueue.hpp b/public/common/TracyQueue.hpp index daef3ec1..765c83c7 100644 --- a/public/common/TracyQueue.hpp +++ b/public/common/TracyQueue.hpp @@ -61,6 +61,7 @@ enum class QueueType : uint8_t ThreadWakeup, GpuTime, GpuContextName, + GpuAnnotationName, CallstackFrameSize, SymbolInformation, ExternalNameMetadata, @@ -111,6 +112,7 @@ enum class QueueType : uint8_t SecondStringData, MemNamePayload, ThreadGroupHint, + GpuZoneAnnotation, StringData, ThreadName, PlotName, @@ -331,7 +333,7 @@ struct QueuePlotDataInt : public QueuePlotDataBase int64_t val; }; -struct QueuePlotDataFloat : public QueuePlotDataBase +struct QueuePlotDataFloat : public QueuePlotDataBase { float val; }; @@ -406,7 +408,8 @@ enum class GpuContextType : uint8_t Direct3D11, Metal, Custom, - CUDA + CUDA, + Rocprof }; enum GpuContextFlags : uint8_t @@ -446,6 +449,15 @@ struct QueueGpuZoneEnd uint8_t context; }; +struct QueueGpuZoneAnnotation +{ + int64_t noteId; + double value; + uint32_t thread; + uint16_t queryId; + uint8_t context; +}; + struct QueueGpuTime { int64_t gpuTime; @@ -467,7 +479,7 @@ struct QueueGpuTimeSync int64_t cpuTime; uint8_t context; }; - + struct QueueGpuContextName { uint8_t context; @@ -479,6 +491,18 @@ struct QueueGpuContextNameFat : public QueueGpuContextName uint16_t size; }; +struct QueueGpuAnnotationName +{ + int64_t noteId; + uint8_t context; +}; + +struct QueueGpuAnnotationNameFat : public QueueGpuAnnotationName +{ + uint64_t ptr; + uint16_t size; +}; + struct QueueMemNamePayload { uint64_t name; @@ -756,6 +780,8 @@ struct QueueItem QueueGpuTimeSync gpuTimeSync; QueueGpuContextName gpuContextName; QueueGpuContextNameFat gpuContextNameFat; + QueueGpuAnnotationName gpuAnnotationName; + QueueGpuAnnotationNameFat gpuAnnotationNameFat; QueueMemAlloc memAlloc; QueueMemFree memFree; QueueMemDiscard memDiscard; @@ -789,6 +815,7 @@ struct QueueItem QueueSourceCodeNotAvailable sourceCodeNotAvailable; QueueFiberEnter fiberEnter; QueueFiberLeave fiberLeave; + QueueGpuZoneAnnotation zoneAnnotation; }; }; #pragma pack( pop ) @@ -849,6 +876,7 @@ static constexpr size_t QueueDataSize[] = { sizeof( QueueHeader ) + sizeof( QueueThreadWakeup ), sizeof( QueueHeader ) + sizeof( QueueGpuTime ), sizeof( QueueHeader ) + sizeof( QueueGpuContextName ), + sizeof( QueueHeader ) + sizeof( QueueGpuAnnotationName ), sizeof( QueueHeader ) + sizeof( QueueCallstackFrameSize ), sizeof( QueueHeader ) + sizeof( QueueSymbolInformation ), sizeof( QueueHeader ), // ExternalNameMetadata - not for wire transfer @@ -900,6 +928,7 @@ static constexpr size_t QueueDataSize[] = { sizeof( QueueHeader ), // second string data sizeof( QueueHeader ) + sizeof( QueueMemNamePayload ), sizeof( QueueHeader ) + sizeof( QueueThreadGroupHint ), + sizeof( QueueHeader ) + sizeof( QueueGpuZoneAnnotation ), // GPU zone annotation // keep all QueueStringTransfer below sizeof( QueueHeader ) + sizeof( QueueStringTransfer ), // string data sizeof( QueueHeader ) + sizeof( QueueStringTransfer ), // thread name diff --git a/public/common/TracyVersion.hpp b/public/common/TracyVersion.hpp index 6da3a6f6..7d704c50 100644 --- a/public/common/TracyVersion.hpp +++ b/public/common/TracyVersion.hpp @@ -7,7 +7,7 @@ namespace Version { enum { Major = 0 }; enum { Minor = 12 }; -enum { Patch = 3 }; +enum { Patch = 4 }; } } diff --git a/server/TracyEvent.hpp b/server/TracyEvent.hpp index 33e81f24..b2276dee 100644 --- a/server/TracyEvent.hpp +++ b/server/TracyEvent.hpp @@ -412,6 +412,7 @@ struct GpuEvent uint64_t _gpuStart_child1; uint64_t _gpuEnd_child2; Int24 callstack; + uint16_t query_id; }; enum { GpuEventSize = sizeof( GpuEvent ) }; @@ -774,6 +775,8 @@ struct GpuCtxData uint32_t overflowMul; StringIdx name; unordered_flat_map threadData; + unordered_flat_map noteNames; + unordered_flat_map> notes; short_ptr query[64*1024]; }; diff --git a/server/TracyWorker.cpp b/server/TracyWorker.cpp index 93866cdf..99e859a7 100644 --- a/server/TracyWorker.cpp +++ b/server/TracyWorker.cpp @@ -741,7 +741,7 @@ Worker::Worker( FileRead& f, EventType::Type eventMask, bool bgTasks, bool allow { m_data.stringData.reserve_exact( sz, m_slab ); } - + for( uint64_t i=0; i(); uint8_t calibration; f.Read7( ctx->thread, calibration, ctx->count, ctx->period, ctx->type, ctx->name, ctx->overflow ); + uint64_t notesz; + f.Read( notesz ); + for( uint64_t i = 0; i < notesz; i++ ) + { + decltype( ctx->noteNames )::key_type key; + decltype( ctx->noteNames )::mapped_type value; + f.Read2( key, value ); + ctx->noteNames[key] = value; + } ctx->hasCalibration = calibration; ctx->hasPeriod = ctx->period != 1.f; m_data.gpuCnt += ctx->count; @@ -1122,6 +1131,26 @@ Worker::Worker( FileRead& f, EventType::Type eventMask, bool bgTasks, bool allow ReadTimeline( f, td->second.timeline, tsz, refTime, refGpuTime, childIdx ); } } + + f.Read( notesz ); + ctx->notes.reserve( notesz ); + for( uint64_t i = 0; i < notesz; i++ ) + { + uint16_t query_id; + f.Read( query_id ); + auto& notes = ctx->notes[query_id]; + uint64_t note_count; + f.Read( note_count ); + notes.reserve( note_count ); + for( uint64_t i = 0; i < note_count; i++ ) + { + int64_t id; + double value; + f.Read2( id, value ); + notes[id] = value; + } + } + m_data.gpuData[i] = ctx; } @@ -4614,6 +4643,12 @@ bool Worker::Process( const QueueItem& ev ) case QueueType::GpuContextName: ProcessGpuContextName( ev.gpuContextName ); break; + case QueueType::GpuAnnotationName: + ProcessGpuAnnotationName( ev.gpuAnnotationName ); + break; + case QueueType::GpuZoneAnnotation: + ProcessGpuZoneAnnotation( ev.zoneAnnotation ); + break; case QueueType::MemAlloc: ProcessMemAlloc( ev.memAlloc ); break; @@ -5750,6 +5785,7 @@ void Worker::ProcessGpuZoneBeginImplCommon( GpuEvent* zone, const QueueGpuZoneBe zone->SetGpuEnd( -1 ); zone->callstack.SetVal( 0 ); zone->SetChild( -1 ); + zone->query_id = ev.queryId; uint64_t ztid; if( ctx->thread == 0 ) @@ -5973,7 +6009,7 @@ void Worker::ProcessGpuCalibration( const QueueGpuCalibration& ev ) ctx->calibratedGpuTime = gpuTime; ctx->calibratedCpuTime = TscTime( ev.cpuTime ); } - + void Worker::ProcessGpuTimeSync( const QueueGpuTimeSync& ev ) { auto ctx = m_gpuCtxMap[ev.context]; @@ -6005,6 +6041,26 @@ void Worker::ProcessGpuContextName( const QueueGpuContextName& ev ) ctx->name = StringIdx( idx ); } +void Worker::ProcessGpuAnnotationName( const QueueGpuAnnotationName& ev ) +{ + auto ctx = m_gpuCtxMap[ev.context]; + assert( ctx ); + const auto idx = GetSingleStringIdx(); + ctx->noteNames[ev.noteId] = StringIdx( idx ); +} + +void Worker::ProcessGpuZoneAnnotation( const QueueGpuZoneAnnotation& ev ) +{ + auto ctx = m_gpuCtxMap[ev.context]; + assert( ctx ); + auto note = ctx->notes.find( ev.queryId ); + if( note == ctx->notes.end() ) { + note = ctx->notes.emplace( ev.queryId, decltype(ctx->notes)::mapped_type{} ).first; + note->second.reserve( ctx->noteNames.size() ); + } + note->second[ev.noteId] = ev.value; +} + MemEvent* Worker::ProcessMemAllocImpl( MemData& memdata, const QueueMemAlloc& ev ) { if( memdata.active.find( ev.ptr ) != memdata.active.end() ) @@ -7782,6 +7838,7 @@ void Worker::ReadTimeline( FileRead& f, Vector>& _vec, uint6 refGpuTime += tgpu; zone->SetCpuEnd( refTime ); zone->SetGpuEnd( refGpuTime ); + f.Read( zone->query_id ); } while( ++zone != end ); } @@ -8119,6 +8176,13 @@ void Worker::Write( FileWrite& f, bool fiDict ) f.Write( &ctx->type, sizeof( ctx->type ) ); f.Write( &ctx->name, sizeof( ctx->name ) ); f.Write( &ctx->overflow, sizeof( ctx->overflow ) ); + sz = ctx->noteNames.size(); + f.Write( &sz, sizeof( sz ) ); + for( auto& p : ctx->noteNames ) + { + f.Write( &p.first, sizeof( p.first ) ); + f.Write( &p.second, sizeof( p.second ) ); + } sz = ctx->threadData.size(); f.Write( &sz, sizeof( sz ) ); for( auto& td : ctx->threadData ) @@ -8129,6 +8193,20 @@ void Worker::Write( FileWrite& f, bool fiDict ) f.Write( &tid, sizeof( tid ) ); WriteTimeline( f, td.second.timeline, refTime, refGpuTime ); } + + sz = ctx->notes.size(); + f.Write( &sz, sizeof( sz ) ); + for( auto& notes : ctx->notes ) + { + f.Write( ¬es.first, sizeof( notes.first ) ); + sz = notes.second.size(); + f.Write( &sz, sizeof( sz ) ); + for( auto& note : notes.second ) + { + f.Write( ¬e.first, sizeof( note.first ) ); + f.Write( ¬e.second, sizeof( note.second ) ); + } + } } sz = m_data.plots.Data().size(); @@ -8509,6 +8587,7 @@ void Worker::WriteTimelineImpl( FileWrite& f, const V& vec, int64_t& refTime, in WriteTimeOffset( f, refTime, v.CpuEnd() ); WriteTimeOffset( f, refGpuTime, v.GpuEnd() ); + f.Write( &v.query_id, sizeof( v.query_id ) ); } } diff --git a/server/TracyWorker.hpp b/server/TracyWorker.hpp index d01a2193..aeee7d58 100644 --- a/server/TracyWorker.hpp +++ b/server/TracyWorker.hpp @@ -741,6 +741,8 @@ private: tracy_force_inline void ProcessGpuCalibration( const QueueGpuCalibration& ev ); tracy_force_inline void ProcessGpuTimeSync( const QueueGpuTimeSync& ev ); tracy_force_inline void ProcessGpuContextName( const QueueGpuContextName& ev ); + tracy_force_inline void ProcessGpuAnnotationName( const QueueGpuAnnotationName& ev ); + tracy_force_inline void ProcessGpuZoneAnnotation( const QueueGpuZoneAnnotation& ev ); tracy_force_inline MemEvent* ProcessMemAlloc( const QueueMemAlloc& ev ); tracy_force_inline MemEvent* ProcessMemAllocNamed( const QueueMemAlloc& ev ); tracy_force_inline MemEvent* ProcessMemFree( const QueueMemFree& ev );