diff --git a/manual/tracy.md b/manual/tracy.md index 19267b5f..8b7a0825 100644 --- a/manual/tracy.md +++ b/manual/tracy.md @@ -1566,6 +1566,8 @@ Tracy provides bindings for profiling OpenGL, Vulkan, Direct3D 11, Direct3D 12, Note that the CPU and GPU timers may be unsynchronized unless you create a calibrated context, but the availability of calibrated contexts is limited. You can try to correct the desynchronization of uncalibrated contexts in the profiler's options (section [5.4](#options)). +GPU contexts are identified by a process-wide unique id, assigned when the context is created (the context creation macros do this automatically). Ids are 8 bits wide (0--255) and are allocated sequentially, so a process can create at most 256 GPU contexts. + > [!TIP] > **Check the scope** > @@ -2203,7 +2205,7 @@ Moreover, there are two sets of functions described below. The standard set send A GPU context can be created with the `___tracy_emit_gpu_new_context` function (or the serialized variant). You'll need to specify: -- `context` -- a unique context id. +- `context` -- a unique context id (8-bit, 0--255; see section [3.10](#gpuprofiling)). - `gpuTime` -- an initial GPU timestamp. diff --git a/manual/tracy.tex b/manual/tracy.tex index 485c6c24..8c048e22 100644 --- a/manual/tracy.tex +++ b/manual/tracy.tex @@ -1776,6 +1776,8 @@ Tracy provides bindings for profiling OpenGL, Vulkan, Direct3D 11, Direct3D 12, Note that the CPU and GPU timers may be unsynchronized unless you create a calibrated context, but the availability of calibrated contexts is limited. You can try to correct the desynchronization of uncalibrated contexts in the profiler's options (section~\ref{options}). +GPU contexts are identified by a process-wide unique id, assigned when the context is created (the context creation macros do this automatically). Ids are 8 bits wide (0--255) and are allocated sequentially, so a process can create at most 256 GPU contexts. + \begin{bclogo}[ noborder=true, couleur=black!5, @@ -2480,7 +2482,7 @@ Moreover, there are two sets of functions described below. The standard set send A GPU context can be created with the \texttt{\_\_\_tracy\_emit\_gpu\_new\_context} function (or the serialized variant). You'll need to specify: \begin{itemize} -\item \texttt{context} -- a unique context id. +\item \texttt{context} -- a unique context id (8-bit, 0--255; see section~\ref{gpuprofiling}). \item \texttt{gpuTime} -- an initial GPU timestamp. \item \texttt{period} -- the timestamp period of the GPU. \item \texttt{flags} -- the flags to use. diff --git a/public/client/TracyProfiler.cpp b/public/client/TracyProfiler.cpp index 3d0f95c3..9eaee7c5 100644 --- a/public/client/TracyProfiler.cpp +++ b/public/client/TracyProfiler.cpp @@ -1240,7 +1240,7 @@ struct ProfilerData moodycamel::ConcurrentQueue queue; Profiler profiler; std::atomic lockCounter { 0 }; - std::atomic gpuCtxCounter { 0 }; + std::atomic gpuCtxCounter { 0 }; std::atomic threadNameData { nullptr }; }; @@ -1385,7 +1385,7 @@ TRACY_API Profiler& MANGLED_NAME_BASED_ON_CONFIG(GetProfiler)() { return GetProf TRACY_API moodycamel::ConcurrentQueue& GetQueue() { return GetProfilerData().queue; } TRACY_API int64_t GetInitTime() { return GetProfilerData().initTime; } TRACY_API std::atomic& GetLockCounter() { return GetProfilerData().lockCounter; } -TRACY_API std::atomic& GetGpuCtxCounter() { return GetProfilerData().gpuCtxCounter; } +TRACY_API std::atomic& GetGpuCtxCounter() { return GetProfilerData().gpuCtxCounter; } TRACY_API GpuCtxWrapper& GetGpuCtx() { return GetProfilerThreadData().gpuCtx; } TRACY_API uint32_t GetThreadHandle() { return detail::GetThreadHandleImpl(); } std::atomic& GetThreadNameData() { return GetProfilerData().threadNameData; } @@ -1471,7 +1471,7 @@ static EarlyMainThreadTokenRepair init_order(104) s_earlyMainThreadTokenRepair; # endif std::atomic init_order(104) s_lockCounter( 0 ); -std::atomic init_order(104) s_gpuCtxCounter( 0 ); +std::atomic init_order(104) s_gpuCtxCounter( 0 ); thread_local GpuCtxWrapper init_order(104) s_gpuCtx { nullptr }; @@ -1490,7 +1490,7 @@ TRACY_API Profiler& MANGLED_NAME_BASED_ON_CONFIG(GetProfiler)() { return s_profi TRACY_API moodycamel::ConcurrentQueue& GetQueue() { return s_queue; } TRACY_API int64_t GetInitTime() { return s_initTime.val; } TRACY_API std::atomic& GetLockCounter() { return s_lockCounter; } -TRACY_API std::atomic& GetGpuCtxCounter() { return s_gpuCtxCounter; } +TRACY_API std::atomic& GetGpuCtxCounter() { return s_gpuCtxCounter; } TRACY_API GpuCtxWrapper& GetGpuCtx() { return s_gpuCtx; } TRACY_API uint32_t GetThreadHandle() { return s_threadHandle.val; } @@ -1501,6 +1501,18 @@ TRACY_API LuaZoneState& GetLuaZoneState() { return s_luaZoneState; } # endif #endif +TRACY_API int32_t NextGpuContextId() +{ + const auto id = GetGpuCtxCounter().fetch_add( 1, std::memory_order_relaxed ); + if( id > UINT8_MAX ) + { + Profiler::LogString( MessageSourceType::Tracy, MessageSeverity::Error, 0, 0, "Tracy: more than 256 GPU contexts in this process; gpu context ids are 8-bit" ); + assert( false ); + return InvalidGpuContextId; + } + return int32_t( id ); +} + TRACY_API bool ProfilerAvailable() { return s_instance != nullptr; } TRACY_API bool ProfilerAllocatorAvailable() { return !RpThreadShutdown; } diff --git a/public/client/TracyProfiler.hpp b/public/client/TracyProfiler.hpp index 877b522f..d7f3cf39 100644 --- a/public/client/TracyProfiler.hpp +++ b/public/client/TracyProfiler.hpp @@ -95,7 +95,8 @@ TRACY_API moodycamel::ConcurrentQueue::ExplicitProducer* GetToken(); TRACY_API Profiler& MANGLED_NAME_BASED_ON_CONFIG(GetProfiler)(); tracy_force_inline Profiler& GetProfiler() { return MANGLED_NAME_BASED_ON_CONFIG(GetProfiler)(); } TRACY_API std::atomic& GetLockCounter(); -TRACY_API std::atomic& GetGpuCtxCounter(); +TRACY_API std::atomic& GetGpuCtxCounter(); +TRACY_API int32_t NextGpuContextId(); TRACY_API GpuCtxWrapper& GetGpuCtx(); TRACY_API uint32_t GetThreadHandle(); TRACY_API bool ProfilerAvailable(); diff --git a/public/client/TracyRocprof.cpp b/public/client/TracyRocprof.cpp index c2449dbd..d9f2e749 100644 --- a/public/client/TracyRocprof.cpp +++ b/public/client/TracyRocprof.cpp @@ -79,14 +79,9 @@ uint8_t gpu_context_allocate( ToolData* data ) float timestamp_period = 1.0f; data->previous_cpu_time = cpu_timestamp; - // Allocate the process-unique GPU context ID. There's a max of 255 available; - // if we are recreating devices a lot we may exceed that. Don't do that, or - // wrap around and get weird (but probably still usable) numbers. - uint8_t context_id = tracy::GetGpuCtxCounter().fetch_add( 1, std::memory_order_relaxed ); - if( context_id >= 255 ) - { - context_id %= 255; - } + // Allocate the process-unique GPU context ID. There's a max of 256 available + // (ids are 8-bit); if we are recreating devices a lot we may exceed that. + uint8_t context_id = uint8_t( tracy::NextGpuContextId() ); uint8_t context_flags = 0; #ifdef TRACY_ROCPROF_CALIBRATION diff --git a/public/common/TracyQueue.hpp b/public/common/TracyQueue.hpp index d383ea94..76dde707 100644 --- a/public/common/TracyQueue.hpp +++ b/public/common/TracyQueue.hpp @@ -536,6 +536,8 @@ enum GpuContextFlags : uint8_t GpuContextCalibration = 1 << 0 }; +constexpr int32_t InvalidGpuContextId = -1; + struct QueueGpuNewContext { int64_t cpuTime; diff --git a/public/tracy/TracyCUDA.hpp b/public/tracy/TracyCUDA.hpp index 2473f187..df578205 100644 --- a/public/tracy/TracyCUDA.hpp +++ b/public/tracy/TracyCUDA.hpp @@ -576,7 +576,7 @@ namespace tracy auto item = Profiler::QueueSerial(); tracyMemWrite(item->hdr.type, QueueType::GpuContextName); - tracyMemWrite(item->gpuContextNameFat.context, m_tracyGpuContext); + tracyMemWrite(item->gpuContextNameFat.context, (uint8_t)m_tracyGpuContext); tracyMemWrite(item->gpuContextNameFat.ptr, (uint64_t)ptr); tracyMemWrite(item->gpuContextNameFat.size, len); SubmitQueueItem(item); @@ -623,7 +623,7 @@ namespace tracy tracyMemWrite(item->gpuCalibration.gpuTime, (int64_t)tCUpti); tracyMemWrite(item->gpuCalibration.cpuTime, tTracy); tracyMemWrite(item->gpuCalibration.cpuDelta, deltaTicksCUpti); - tracyMemWrite(item->gpuCalibration.context, m_tracyGpuContext); + tracyMemWrite(item->gpuCalibration.context, (uint8_t)m_tracyGpuContext); Profiler::QueueSerialFinish(); } #endif @@ -644,8 +644,8 @@ namespace tracy //uint32_t timelineId = tracy::GetThreadHandle(); uint32_t timelineId = tracyTimelineId(cudaContextId, cudaStreamId); uint16_t queryId = m_queryIdGen.fetch_add(2); - tracyAnnounceGpuTimestamp(apiStart, apiEnd, queryId, m_tracyGpuContext, pSrcLoc, timelineId); - tracySubmitGpuTimestamp(gpuStart, gpuEnd, queryId, m_tracyGpuContext); + tracyAnnounceGpuTimestamp(apiStart, apiEnd, queryId, (uint8_t)m_tracyGpuContext, pSrcLoc, timelineId); + tracySubmitGpuTimestamp(gpuStart, gpuEnd, queryId, (uint8_t)m_tracyGpuContext); } void OnEventsProcessed() { @@ -1410,17 +1410,17 @@ namespace tracy }; - CUDACtx(uint8_t gpuContextID = 255) + CUDACtx(int32_t gpuContextID = InvalidGpuContextId) { ZoneScoped; - if (gpuContextID != 255) { + if (gpuContextID != InvalidGpuContextId) { m_tracyGpuContext = gpuContextID; return; } - m_tracyGpuContext = GetGpuCtxCounter().fetch_add(1, std::memory_order_relaxed); - assert(m_tracyGpuContext != 255); + m_tracyGpuContext = NextGpuContextId(); + assert(m_tracyGpuContext != InvalidGpuContextId); TracyTimestamp tTracy; CUptiTimestamp tCUpti; @@ -1434,7 +1434,7 @@ namespace tracy tracyMemWrite(item->gpuNewContext.thread, (uint32_t)0); tracyMemWrite(item->gpuNewContext.period, 1.0f); tracyMemWrite(item->gpuNewContext.type, GpuContextType::CUDA); - tracyMemWrite(item->gpuNewContext.context, m_tracyGpuContext); + tracyMemWrite(item->gpuNewContext.context, (uint8_t)m_tracyGpuContext); #if TRACY_CUDA_CALIBRATED_CONTEXT tracyMemWrite(item->gpuNewContext.flags, GpuContextCalibration); #else @@ -1470,7 +1470,7 @@ namespace tracy CUDACtx* ctx = nullptr; std::mutex m; int ref_count = 0; - uint8_t ctx_id = 255; + int32_t ctx_id = InvalidGpuContextId; static Singleton& Get() { static Singleton singleton; return singleton; @@ -1481,7 +1481,7 @@ namespace tracy ProfilerStats stats = {}; #endif - uint8_t m_tracyGpuContext = 255; + int32_t m_tracyGpuContext = InvalidGpuContextId; static constexpr size_t cacheline = 64; alignas(cacheline) std::atomic m_queryIdGen = 0; }; diff --git a/public/tracy/TracyD3D11.hpp b/public/tracy/TracyD3D11.hpp index 8516b106..a63d6604 100644 --- a/public/tracy/TracyD3D11.hpp +++ b/public/tracy/TracyD3D11.hpp @@ -121,7 +121,7 @@ public: } // ready to roll - m_contextId = GetGpuCtxCounter().fetch_add(1); + m_contextId = NextGpuContextId(); m_immediateDevCtx->Begin(m_disjointQuery); m_previousCheckpoint = m_nextCheckpoint = 0; @@ -131,7 +131,7 @@ public: MemWrite( &item->gpuNewContext.gpuTime, tgpu ); MemWrite( &item->gpuNewContext.thread, uint32_t(0) ); // #TODO: why not GetThreadHandle()? MemWrite( &item->gpuNewContext.period, 1.0f ); - MemWrite( &item->gpuNewContext.context, m_contextId); + MemWrite( &item->gpuNewContext.context, uint8_t(m_contextId)); MemWrite( &item->gpuNewContext.flags, GpuContextFlags(0) ); MemWrite( &item->gpuNewContext.type, GpuContextType::Direct3D11 ); @@ -167,7 +167,7 @@ public: auto item = Profiler::QueueSerial(); MemWrite( &item->hdr.type, QueueType::GpuContextName ); - MemWrite( &item->gpuContextNameFat.context, m_contextId ); + MemWrite( &item->gpuContextNameFat.context, uint8_t(m_contextId) ); MemWrite( &item->gpuContextNameFat.ptr, (uint64_t)ptr ); MemWrite( &item->gpuContextNameFat.size, len ); #ifdef TRACY_ON_DEMAND @@ -240,7 +240,7 @@ public: MemWrite(&item->hdr.type, QueueType::GpuTime); MemWrite(&item->gpuTime.gpuTime, static_cast(timestamp)); MemWrite(&item->gpuTime.queryId, static_cast(k)); - MemWrite(&item->gpuTime.context, m_contextId); + MemWrite(&item->gpuTime.context, uint8_t(m_contextId)); Profiler::QueueSerialFinish(); } @@ -287,7 +287,7 @@ private: YieldThread(); // busy-wait :-( attempt to reduce power usage with _mm_pause() & friends... } - tracy_force_inline uint8_t GetContextId() const + tracy_force_inline int32_t GetContextId() const { return m_contextId; } @@ -298,7 +298,7 @@ private: ID3D11Query* m_queries[MaxQueries]; ID3D11Query* m_disjointQuery = nullptr; - uint8_t m_contextId = 255; // NOTE: apparently, 255 means invalid id; is this documented anywhere? + int32_t m_contextId = InvalidGpuContextId; uintptr_t m_queryCounter = 0; @@ -380,7 +380,7 @@ public: MemWrite( &item->gpuZoneEnd.cpuTime, Profiler::GetTime() ); MemWrite( &item->gpuZoneEnd.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneEnd.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneEnd.context, m_ctx->GetContextId() ); + MemWrite( &item->gpuZoneEnd.context, uint8_t(m_ctx->GetContextId()) ); Profiler::QueueSerialFinish(); } @@ -407,7 +407,7 @@ private: MemWrite( &item->gpuZoneBegin.srcloc, sourceLocation ); MemWrite( &item->gpuZoneBegin.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, m_ctx->GetContextId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t(m_ctx->GetContextId()) ); Profiler::QueueSerialFinish(); } @@ -433,7 +433,6 @@ static inline void DestroyD3D11Context( D3D11Ctx* ctx ) tracy_free( ctx ); } } - #undef TracyD3D11Panic using TracyD3D11Ctx = tracy::D3D11Ctx*; diff --git a/public/tracy/TracyD3D12.hpp b/public/tracy/TracyD3D12.hpp index d1919370..0865b113 100644 --- a/public/tracy/TracyD3D12.hpp +++ b/public/tracy/TracyD3D12.hpp @@ -76,7 +76,7 @@ namespace tracy { friend class D3D12ZoneScope; - uint8_t m_contextId = 255; // 255 represents "invalid id" + int32_t m_contextId = InvalidGpuContextId; std::mutex m_collectionMutex; @@ -128,7 +128,7 @@ namespace tracy MemWrite(&item->gpuCalibration.gpuTime, int64_t(gpuTimestamp)); MemWrite(&item->gpuCalibration.cpuTime, int64_t(cpuTimestamp)); MemWrite(&item->gpuCalibration.cpuDelta, cpuDeltaNS); - MemWrite(&item->gpuCalibration.context, GetId()); + MemWrite(&item->gpuCalibration.context, static_cast(GetId())); SubmitQueueItem(item); } } @@ -239,7 +239,7 @@ namespace tracy cpuTimestamp = Profiler::GetTime(); // All setup/init checks completed: ready to create the context. - m_contextId = GetGpuCtxCounter().fetch_add(1); + m_contextId = NextGpuContextId(); ZoneValue(m_contextId); auto* item = Profiler::QueueSerial(); @@ -288,7 +288,7 @@ namespace tracy m_queue->Release(); } - tracy_force_inline uint8_t GetId() const + tracy_force_inline int32_t GetId() const { return m_contextId; } @@ -300,7 +300,7 @@ namespace tracy auto item = Profiler::QueueSerial(); MemWrite( &item->hdr.type, QueueType::GpuContextName ); - MemWrite( &item->gpuContextNameFat.context, GetId()); + MemWrite( &item->gpuContextNameFat.context, static_cast(GetId())); MemWrite( &item->gpuContextNameFat.ptr, (uint64_t)ptr ); MemWrite( &item->gpuContextNameFat.size, len ); SubmitQueueItem(item); @@ -458,7 +458,7 @@ namespace tracy MemWrite(&item->hdr.type, QueueType::GpuTime); MemWrite(&item->gpuTime.gpuTime, static_cast(gpuTimestamp)); MemWrite(&item->gpuTime.queryId, static_cast(queryId)); - MemWrite(&item->gpuTime.context, GetId()); + MemWrite(&item->gpuTime.context, static_cast(GetId())); Profiler::QueueSerialFinish(); m_shadowBuffer[queryId] = gpuTimestamp; TracyD3D12Debug( @@ -608,7 +608,7 @@ namespace tracy MemWrite( &item->gpuZoneBegin.srcloc, srcLocationAddr ); MemWrite( &item->gpuZoneBegin.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneBegin.queryId, static_cast( m_queryId ) ); - MemWrite( &item->gpuZoneBegin.context, m_ctx->GetId() ); + MemWrite( &item->gpuZoneBegin.context, static_cast(m_ctx->GetId()) ); Profiler::QueueSerialFinish(); } @@ -669,7 +669,7 @@ namespace tracy MemWrite(&item->gpuZoneEnd.cpuTime, Profiler::GetTime()); MemWrite(&item->gpuZoneEnd.thread, GetThreadHandle()); MemWrite(&item->gpuZoneEnd.queryId, static_cast(queryId)); - MemWrite(&item->gpuZoneEnd.context, m_ctx->GetId()); + MemWrite( &item->gpuZoneEnd.context, static_cast(m_ctx->GetId()) ); Profiler::QueueSerialFinish(); #ifdef TRACY_ON_DEMAND } @@ -695,12 +695,6 @@ namespace tracy { auto* ctx = static_cast(tracy_malloc(sizeof(D3D12QueueCtx))); new (ctx) D3D12QueueCtx{ device, queue }; - // constructor may have failed: - if (ctx->GetId() == 255) - { - DestroyD3D12Context(ctx); - return nullptr; - } return ctx; } diff --git a/public/tracy/TracyMetal.hmm b/public/tracy/TracyMetal.hmm index 519423e4..7afc1536 100644 --- a/public/tracy/TracyMetal.hmm +++ b/public/tracy/TracyMetal.hmm @@ -125,11 +125,6 @@ public: ZoneScopedNC("tracy::MetalCtx::Create", Color::Red4); auto ctx = static_cast(tracy_malloc(sizeof(MetalCtx))); new (ctx) MetalCtx(device); - if (ctx->m_contextId == 255) - { - TracyMetalPanic({assert(false);} return nullptr, "ERROR: unable to create context."); - Destroy(ctx); - } return ctx; } @@ -147,7 +142,7 @@ public: auto* item = Profiler::QueueSerial(); MemWrite( &item->hdr.type, QueueType::GpuContextName ); - MemWrite( &item->gpuContextNameFat.context, m_contextId ); + MemWrite( &item->gpuContextNameFat.context, static_cast(m_contextId) ); MemWrite( &item->gpuContextNameFat.ptr, (uint64_t)ptr ); MemWrite( &item->gpuContextNameFat.size, len ); SubmitQueueItem(item); @@ -259,7 +254,7 @@ public: MemWrite(&item->hdr.type, QueueType::GpuTime); MemWrite(&item->gpuTime.gpuTime, static_cast(t_start)); MemWrite(&item->gpuTime.queryId, static_cast(k)); - MemWrite(&item->gpuTime.context, m_contextId); + MemWrite(&item->gpuTime.context, static_cast(m_contextId)); Profiler::QueueSerialFinish(); } { @@ -267,7 +262,7 @@ public: MemWrite(&item->hdr.type, QueueType::GpuTime); MemWrite(&item->gpuTime.gpuTime, static_cast(t_end)); MemWrite(&item->gpuTime.queryId, static_cast(k+1)); - MemWrite(&item->gpuTime.context, m_contextId); + MemWrite(&item->gpuTime.context, static_cast(m_contextId)); Profiler::QueueSerialFinish(); } m_mostRecentTimestamp = (t_end > m_mostRecentTimestamp) ? t_end : m_mostRecentTimestamp; @@ -342,7 +337,7 @@ private: float period = 1.0f; - m_contextId = GetGpuCtxCounter().fetch_add(1); + m_contextId = NextGpuContextId(); auto* item = Profiler::QueueSerial(); MemWrite(&item->hdr.type, QueueType::GpuNewContext); @@ -350,7 +345,7 @@ private: MemWrite(&item->gpuNewContext.gpuTime, int64_t(gpuTimestamp)); MemWrite(&item->gpuNewContext.thread, uint32_t(0)); // TODO: why not GetThreadHandle()? MemWrite(&item->gpuNewContext.period, period); - MemWrite(&item->gpuNewContext.context, m_contextId); + MemWrite(&item->gpuNewContext.context, static_cast(m_contextId)); //MemWrite(&item->gpuNewContext.flags, GpuContextCalibration); MemWrite(&item->gpuNewContext.flags, GpuContextFlags(0)); MemWrite(&item->gpuNewContext.type, GpuContextType::Metal); @@ -421,7 +416,7 @@ private: return Query{ buffer, idx }; } - tracy_force_inline uint8_t GetContextId() const + tracy_force_inline int32_t GetContextId() const { return m_contextId; } @@ -463,7 +458,7 @@ private: return counterSampleBuffer; } - uint8_t m_contextId = 255; + int32_t m_contextId = InvalidGpuContextId; id m_device = nil; id m_counterSampleBuffers [2] = {}; @@ -602,7 +597,7 @@ private: MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)srcloc ); MemWrite( &item->gpuZoneBegin.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, ctx->GetContextId() ); + MemWrite( &item->gpuZoneBegin.context, static_cast(ctx->GetContextId()) ); Profiler::QueueSerialFinish(); TracyMetalDebugMasked(1<<2, TracyAllocN((void*)(uintptr_t)queryId, 1, "TracyMetalGpuZone")); @@ -615,7 +610,7 @@ private: MemWrite( &item->gpuZoneEnd.cpuTime, Profiler::GetTime() ); MemWrite( &item->gpuZoneEnd.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneEnd.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneEnd.context, ctx->GetContextId() ); + MemWrite( &item->gpuZoneEnd.context, static_cast(ctx->GetContextId()) ); Profiler::QueueSerialFinish(); TracyMetalDebugMasked(1<<2, TracyAllocN((void*)(uintptr_t)queryId, 1, "TracyMetalGpuZone")); diff --git a/public/tracy/TracyOpenCL.hpp b/public/tracy/TracyOpenCL.hpp index a0ce52f6..a5a4375f 100644 --- a/public/tracy/TracyOpenCL.hpp +++ b/public/tracy/TracyOpenCL.hpp @@ -79,12 +79,12 @@ namespace tracy { static constexpr size_t QueryCount = 64 * 1024; OpenCLCtx(cl_context context, cl_device_id device) - : m_contextId(GetGpuCtxCounter().fetch_add(1, std::memory_order_relaxed)) + : m_contextId(NextGpuContextId()) , m_head(0) , m_tail(0) { int64_t tcpu, tgpu; - TRACY_CL_ASSERT(m_contextId != 255); + TRACY_CL_ASSERT(m_contextId != InvalidGpuContextId); cl_int err = CL_SUCCESS; cl_command_queue queue = clCreateCommandQueue(context, device, CL_QUEUE_PROFILING_ENABLE, &err); @@ -200,7 +200,7 @@ namespace tracy { } } - tracy_force_inline uint8_t GetId() const + tracy_force_inline int32_t GetId() const { return m_contextId; } @@ -222,7 +222,7 @@ namespace tracy { private: - unsigned int m_contextId; + int32_t m_contextId; EventInfo m_query[QueryCount]; unsigned int m_head; // index at which a new event should be inserted @@ -251,7 +251,7 @@ namespace tracy { MemWrite(&item->gpuZoneBegin.srcloc, (uint64_t)srcLoc); MemWrite(&item->gpuZoneBegin.thread, GetThreadHandle()); MemWrite(&item->gpuZoneBegin.queryId, (uint16_t)m_beginQueryId); - MemWrite(&item->gpuZoneBegin.context, ctx->GetId()); + MemWrite(&item->gpuZoneBegin.context, uint8_t(ctx->GetId())); Profiler::QueueSerialFinish(); } @@ -276,7 +276,7 @@ namespace tracy { MemWrite(&item->gpuZoneBegin.srcloc, (uint64_t)srcLoc); MemWrite(&item->gpuZoneBegin.thread, GetThreadHandle()); MemWrite(&item->gpuZoneBegin.queryId, (uint16_t)m_beginQueryId); - MemWrite(&item->gpuZoneBegin.context, ctx->GetId()); + MemWrite(&item->gpuZoneBegin.context, uint8_t(ctx->GetId())); Profiler::QueueSerialFinish(); } @@ -300,7 +300,7 @@ namespace tracy { MemWrite(&item->gpuZoneBegin.srcloc, srcloc); MemWrite(&item->gpuZoneBegin.thread, GetThreadHandle()); MemWrite(&item->gpuZoneBegin.queryId, (uint16_t)m_beginQueryId); - MemWrite(&item->gpuZoneBegin.context, ctx->GetId()); + MemWrite(&item->gpuZoneBegin.context, uint8_t(ctx->GetId())); Profiler::QueueSerialFinish(); } @@ -324,7 +324,7 @@ namespace tracy { MemWrite(&item->gpuZoneBegin.srcloc, srcloc); MemWrite(&item->gpuZoneBegin.thread, GetThreadHandle()); MemWrite(&item->gpuZoneBegin.queryId, (uint16_t)m_beginQueryId); - MemWrite(&item->gpuZoneBegin.context, ctx->GetId()); + MemWrite(&item->gpuZoneBegin.context, uint8_t(ctx->GetId())); Profiler::QueueSerialFinish(); } @@ -346,7 +346,7 @@ namespace tracy { MemWrite(&item->gpuZoneEnd.cpuTime, Profiler::GetTime()); MemWrite(&item->gpuZoneEnd.thread, GetThreadHandle()); MemWrite(&item->gpuZoneEnd.queryId, (uint16_t)queryId); - MemWrite(&item->gpuZoneEnd.context, m_ctx->GetId()); + MemWrite(&item->gpuZoneEnd.context, uint8_t(m_ctx->GetId())); Profiler::QueueSerialFinish(); } diff --git a/public/tracy/TracyOpenGL.hpp b/public/tracy/TracyOpenGL.hpp index b07d17e7..dba13c06 100644 --- a/public/tracy/TracyOpenGL.hpp +++ b/public/tracy/TracyOpenGL.hpp @@ -114,14 +114,14 @@ class GpuCtx public: GpuCtx() - : m_context( GetGpuCtxCounter().fetch_add( 1, std::memory_order_relaxed ) ) + : m_context( NextGpuContextId() ) , m_head( 0 ) , m_tail( 0 ) , m_supportsQueryBufferObject( false ) { ZoneScopedC( Color::Red4 ); - assert( m_context != 255 ); + assert( m_context != InvalidGpuContextId ); if( !CheckFeature( "GL_ARB_timer_query" ) && !CheckFeature( "GL_EXT_disjoint_timer_query" ) ) { @@ -169,7 +169,7 @@ public: MemWrite( &item->gpuNewContext.gpuTime, tgpu ); MemWrite( &item->gpuNewContext.thread, thread ); MemWrite( &item->gpuNewContext.period, period ); - MemWrite( &item->gpuNewContext.context, m_context ); + MemWrite( &item->gpuNewContext.context, uint8_t( m_context ) ); #ifdef TRACY_OPENGL_AUTO_CALIBRATION MemWrite( &item->gpuNewContext.flags, GpuContextFlags( GpuContextCalibration ) ); #else @@ -190,7 +190,7 @@ public: memcpy( ptr, name, len ); TracyLfqPrepare( QueueType::GpuContextName ); - MemWrite( &item->gpuContextNameFat.context, m_context ); + MemWrite( &item->gpuContextNameFat.context, uint8_t( m_context ) ); MemWrite( &item->gpuContextNameFat.ptr, (uint64_t)ptr ); MemWrite( &item->gpuContextNameFat.size, len ); #ifdef TRACY_ON_DEMAND @@ -227,7 +227,7 @@ public: TracyLfqPrepare( QueueType::GpuTime ); MemWrite( &item->gpuTime.gpuTime, (int64_t)time ); MemWrite( &item->gpuTime.queryId, (uint16_t)m_tail ); - MemWrite( &item->gpuTime.context, m_context ); + MemWrite( &item->gpuTime.context, uint8_t( m_context ) ); TracyLfqCommit; m_tail = ( m_tail + 1 ) % QueryCount; @@ -311,7 +311,7 @@ private: MemWrite( &item->gpuCalibration.gpuTime, tgpu ); MemWrite( &item->gpuCalibration.cpuTime, refCpu ); MemWrite( &item->gpuCalibration.cpuDelta, delta ); - MemWrite( &item->gpuCalibration.context, m_context ); + MemWrite( &item->gpuCalibration.context, uint8_t( m_context ) ); TracyLfqCommit; } #endif @@ -329,13 +329,13 @@ private: return m_query[id]; } - tracy_force_inline uint8_t GetId() const + tracy_force_inline int32_t GetId() const { return m_context; } unsigned int m_query[QueryCount]; - uint8_t m_context; + int32_t m_context; unsigned int m_head; unsigned int m_tail; @@ -367,7 +367,7 @@ public: MemWrite( &item->gpuZoneBegin.cpuTime, Profiler::GetTime() ); memset( &item->gpuZoneBegin.thread, 0, sizeof( item->gpuZoneBegin.thread ) ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, GetGpuCtx().ptr->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( GetGpuCtx().ptr->GetId() ) ); MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)srcloc ); TracyLfqCommit; } @@ -395,7 +395,7 @@ public: #endif MemWrite( &item->gpuZoneBegin.cpuTime, Profiler::GetTime() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, GetGpuCtx().ptr->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( GetGpuCtx().ptr->GetId() ) ); MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)srcloc ); TracyLfqCommit; } @@ -418,7 +418,7 @@ public: MemWrite( &item->gpuZoneBegin.cpuTime, Profiler::GetTime() ); memset( &item->gpuZoneBegin.thread, 0, sizeof( item->gpuZoneBegin.thread ) ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, GetGpuCtx().ptr->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( GetGpuCtx().ptr->GetId() ) ); MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)srcloc ); TracyLfqCommit; } @@ -447,7 +447,7 @@ public: const auto srcloc = Profiler::AllocSourceLocation( line, source, sourceSz, function, functionSz, name, nameSz ); MemWrite( &item->gpuZoneBegin.cpuTime, Profiler::GetTime() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, GetGpuCtx().ptr->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( GetGpuCtx().ptr->GetId() ) ); MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)srcloc ); TracyLfqCommit; } @@ -466,7 +466,7 @@ public: MemWrite( &item->gpuZoneEnd.cpuTime, Profiler::GetTime() ); memset( &item->gpuZoneEnd.thread, 0, sizeof( item->gpuZoneEnd.thread ) ); MemWrite( &item->gpuZoneEnd.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneEnd.context, GetGpuCtx().ptr->GetId() ); + MemWrite( &item->gpuZoneEnd.context, uint8_t( GetGpuCtx().ptr->GetId() ) ); TracyLfqCommit; } diff --git a/public/tracy/TracyVulkan.hpp b/public/tracy/TracyVulkan.hpp index 5c17e41a..36fc49e2 100644 --- a/public/tracy/TracyVulkan.hpp +++ b/public/tracy/TracyVulkan.hpp @@ -105,7 +105,7 @@ public: #endif : m_device( device ) , m_timeDomain( VK_TIME_DOMAIN_DEVICE_EXT ) - , m_context( GetGpuCtxCounter().fetch_add( 1, std::memory_order_relaxed ) ) + , m_context( NextGpuContextId() ) , m_head( 0 ) , m_tail( 0 ) , m_oldCnt( 0 ) @@ -114,7 +114,7 @@ public: , m_vkGetCalibratedTimestampsEXT( vkGetCalibratedTimestampsEXT ) #endif { - assert( m_context != 255 ); + assert( m_context != InvalidGpuContextId ); #if defined TRACY_VK_USE_SYMBOL_TABLE PopulateSymbolTable(instance, instanceProcAddr, deviceProcAddr); @@ -190,7 +190,7 @@ public: #endif : m_device( device ) , m_timeDomain( VK_TIME_DOMAIN_DEVICE_EXT ) - , m_context( GetGpuCtxCounter().fetch_add(1, std::memory_order_relaxed) ) + , m_context( NextGpuContextId() ) , m_head( 0 ) , m_tail( 0 ) , m_oldCnt( 0 ) @@ -199,7 +199,7 @@ public: , m_vkGetCalibratedTimestampsEXT( vkGetCalibratedTimestampsEXT ) #endif { - assert( m_context != 255); + assert( m_context != InvalidGpuContextId); #if defined TRACY_VK_USE_SYMBOL_TABLE PopulateSymbolTable(instance, instanceProcAddr, deviceProcAddr); @@ -242,7 +242,7 @@ public: auto item = Profiler::QueueSerial(); MemWrite( &item->hdr.type, QueueType::GpuContextName ); - MemWrite( &item->gpuContextNameFat.context, m_context ); + MemWrite( &item->gpuContextNameFat.context, uint8_t( m_context ) ); MemWrite( &item->gpuContextNameFat.ptr, (uint64_t)ptr ); MemWrite( &item->gpuContextNameFat.size, len ); #ifdef TRACY_ON_DEMAND @@ -309,7 +309,7 @@ public: MemWrite( &item->hdr.type, QueueType::GpuTime ); MemWrite( &item->gpuTime.gpuTime, m_res[idx * 2] ); MemWrite( &item->gpuTime.queryId, uint16_t( wrappedTail + idx ) ); - MemWrite( &item->gpuTime.context, m_context ); + MemWrite( &item->gpuTime.context, uint8_t( m_context ) ); Profiler::QueueSerialFinish(); } @@ -327,7 +327,7 @@ public: MemWrite( &item->gpuCalibration.gpuTime, tgpu ); MemWrite( &item->gpuCalibration.cpuTime, refCpu ); MemWrite( &item->gpuCalibration.cpuDelta, delta ); - MemWrite( &item->gpuCalibration.context, m_context ); + MemWrite( &item->gpuCalibration.context, uint8_t( m_context ) ); Profiler::QueueSerialFinish(); } } @@ -345,7 +345,7 @@ public: return id % m_queryCount; } - tracy_force_inline uint8_t GetId() const + tracy_force_inline int32_t GetId() const { return m_context; } @@ -473,7 +473,7 @@ private: MemWrite( &item->gpuNewContext.gpuTime, tgpu ); memset( &item->gpuNewContext.thread, 0, sizeof( item->gpuNewContext.thread ) ); MemWrite( &item->gpuNewContext.period, period ); - MemWrite( &item->gpuNewContext.context, m_context ); + MemWrite( &item->gpuNewContext.context, uint8_t( m_context ) ); MemWrite( &item->gpuNewContext.flags, GpuContextFlags( flags ) ); MemWrite( &item->gpuNewContext.type, GpuContextType::Vulkan ); @@ -517,7 +517,7 @@ private: int64_t m_qpcToNs; #endif int64_t m_prevCalibration; - uint8_t m_context; + int32_t m_context; std::atomic m_head; uint64_t m_tail; @@ -553,7 +553,7 @@ public: MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)srcloc ); MemWrite( &item->gpuZoneBegin.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, ctx->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( ctx->GetId() ) ); Profiler::QueueSerialFinish(); } @@ -587,7 +587,7 @@ public: MemWrite( &item->gpuZoneBegin.srcloc, (uint64_t)srcloc ); MemWrite( &item->gpuZoneBegin.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, ctx->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( ctx->GetId() ) ); Profiler::QueueSerialFinish(); } @@ -613,7 +613,7 @@ public: MemWrite( &item->gpuZoneBegin.srcloc, srcloc ); MemWrite( &item->gpuZoneBegin.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, ctx->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( ctx->GetId() ) ); Profiler::QueueSerialFinish(); } @@ -648,7 +648,7 @@ public: MemWrite( &item->gpuZoneBegin.srcloc, srcloc ); MemWrite( &item->gpuZoneBegin.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneBegin.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneBegin.context, ctx->GetId() ); + MemWrite( &item->gpuZoneBegin.context, uint8_t( ctx->GetId() ) ); Profiler::QueueSerialFinish(); } @@ -667,7 +667,7 @@ public: MemWrite( &item->gpuZoneEnd.cpuTime, Profiler::GetTime() ); MemWrite( &item->gpuZoneEnd.thread, GetThreadHandle() ); MemWrite( &item->gpuZoneEnd.queryId, uint16_t( queryId ) ); - MemWrite( &item->gpuZoneEnd.context, m_ctx->GetId() ); + MemWrite( &item->gpuZoneEnd.context, uint8_t( m_ctx->GetId() ) ); Profiler::QueueSerialFinish(); } diff --git a/public/tracy/TracyWebGPU.hpp b/public/tracy/TracyWebGPU.hpp index 9b25f263..928bfc78 100644 --- a/public/tracy/TracyWebGPU.hpp +++ b/public/tracy/TracyWebGPU.hpp @@ -96,7 +96,7 @@ namespace tracy { friend class WebGPUZoneScope; - uint8_t m_contextId = 255; // 255 represents "invalid id" + int32_t m_contextId = InvalidGpuContextId; std::mutex m_collectionMutex; @@ -528,7 +528,7 @@ namespace tracy m_shadowBuffer.resize(m_queryLimit, gpuTimestamp); // All setup completed: register the context. - m_contextId = GetGpuCtxCounter().fetch_add(1); + m_contextId = NextGpuContextId(); ZoneValue(m_contextId); auto* item = Profiler::QueueSerial(); @@ -560,7 +560,7 @@ namespace tracy if (m_instance) { wgpuInstanceRelease(m_instance); m_instance = nullptr; } } - tracy_force_inline uint8_t GetId() const + tracy_force_inline int32_t GetId() const { return m_contextId; } @@ -572,7 +572,7 @@ namespace tracy auto item = Profiler::QueueSerial(); MemWrite(&item->hdr.type, QueueType::GpuContextName); - MemWrite(&item->gpuContextNameFat.context, GetId()); + MemWrite(&item->gpuContextNameFat.context, static_cast(GetId())); MemWrite(&item->gpuContextNameFat.ptr, (uint64_t)ptr); MemWrite(&item->gpuContextNameFat.size, len); SubmitQueueItem(item); @@ -682,7 +682,7 @@ namespace tracy MemWrite(&item->hdr.type, QueueType::GpuTime); MemWrite(&item->gpuTime.gpuTime, static_cast(gpuTimestamp)); MemWrite(&item->gpuTime.queryId, static_cast(queryId)); - MemWrite(&item->gpuTime.context, GetId()); + MemWrite(&item->gpuTime.context, static_cast(GetId())); Profiler::QueueSerialFinish(); m_shadowBuffer[queryId] = gpuTimestamp; } @@ -808,7 +808,7 @@ namespace tracy MemWrite(&item->gpuZoneBegin.srcloc, srcLocationAddr); MemWrite(&item->gpuZoneBegin.thread, GetThreadHandle()); MemWrite(&item->gpuZoneBegin.queryId, static_cast(m_queryId)); - MemWrite(&item->gpuZoneBegin.context, m_ctx->GetId()); + MemWrite(&item->gpuZoneBegin.context, static_cast(m_ctx->GetId())); Profiler::QueueSerialFinish(); } @@ -901,7 +901,7 @@ namespace tracy MemWrite(&item->gpuZoneEnd.cpuTime, Profiler::GetTime()); MemWrite(&item->gpuZoneEnd.thread, GetThreadHandle()); MemWrite(&item->gpuZoneEnd.queryId, static_cast(queryId)); - MemWrite(&item->gpuZoneEnd.context, m_ctx->GetId()); + MemWrite(&item->gpuZoneEnd.context, static_cast(m_ctx->GetId())); Profiler::QueueSerialFinish(); #ifdef TRACY_ON_DEMAND } @@ -923,11 +923,6 @@ namespace tracy { auto* ctx = static_cast(tracy_malloc(sizeof(WebGPUQueueCtx))); new (ctx) WebGPUQueueCtx{ instance, device, queue }; - if (ctx->GetId() == 255) - { - DestroyWebGPUContext(ctx); - return nullptr; - } return ctx; }