diff --git a/NEW_RELEASE_NOTES.md b/NEW_RELEASE_NOTES.md index a629390438..c690c6a623 100644 --- a/NEW_RELEASE_NOTES.md +++ b/NEW_RELEASE_NOTES.md @@ -12,3 +12,4 @@ appropriate header in [RELEASE_NOTES.md](./RELEASE_NOTES.md). - vulkan: support sRGB swap chain - Add new `getMaxAutomaticInstances()` API on `Engine` to get max supported automatic instances. - UiHelper: fix jank when a `TextureView` is resized (fixes b\282220665) +- backend: parallel shader compilation support. This breaks and improves the recent `Material::compile` API. diff --git a/filament/backend/CMakeLists.txt b/filament/backend/CMakeLists.txt index de86153b06..c11bbc4ed7 100644 --- a/filament/backend/CMakeLists.txt +++ b/filament/backend/CMakeLists.txt @@ -82,6 +82,8 @@ if (FILAMENT_SUPPORTS_OPENGL AND NOT FILAMENT_USE_EXTERNAL_GLES3 AND NOT FILAMEN src/opengl/OpenGLPlatform.cpp src/opengl/OpenGLTimerQuery.cpp src/opengl/OpenGLTimerQuery.h + src/opengl/ShaderCompilerService.cpp + src/opengl/ShaderCompilerService.h ) if (EGL) list(APPEND SRCS src/opengl/platforms/PlatformEGL.cpp) diff --git a/filament/backend/include/backend/DriverEnums.h b/filament/backend/include/backend/DriverEnums.h index 302525ad7c..6e5a142e3f 100644 --- a/filament/backend/include/backend/DriverEnums.h +++ b/filament/backend/include/backend/DriverEnums.h @@ -296,6 +296,14 @@ enum class Precision : uint8_t { DEFAULT }; +/** + * Shader compiler priority queue + */ +enum class CompilerPriorityQueue : uint8_t { + HIGH, + LOW +}; + //! Texture sampler type enum class SamplerType : uint8_t { SAMPLER_2D, //!< 2D texture diff --git a/filament/backend/include/backend/Program.h b/filament/backend/include/backend/Program.h index 1d700d3d98..2a491959e9 100644 --- a/filament/backend/include/backend/Program.h +++ b/filament/backend/include/backend/Program.h @@ -71,6 +71,8 @@ public: ~Program() noexcept; + Program& priorityQueue(CompilerPriorityQueue priorityQueue) noexcept; + // sets the material name and variant for diagnostic purposes only Program& diagnostics(utils::CString const& name, utils::Invocable&& logger); @@ -138,6 +140,8 @@ public: uint64_t getCacheId() const noexcept { return mCacheId; } + CompilerPriorityQueue getPriorityQueue() const noexcept { return mPriorityQueue; } + private: friend utils::io::ostream& operator<<(utils::io::ostream& out, const Program& builder); @@ -150,6 +154,7 @@ private: utils::FixedCapacityVector mSpecializationConstants; utils::FixedCapacityVector> mAttributes; std::array mBindingUniformInfo; + CompilerPriorityQueue mPriorityQueue = CompilerPriorityQueue::HIGH; }; } // namespace filament::backend diff --git a/filament/backend/include/backend/platforms/OpenGLPlatform.h b/filament/backend/include/backend/platforms/OpenGLPlatform.h index 29994cda50..c41dce4360 100644 --- a/filament/backend/include/backend/platforms/OpenGLPlatform.h +++ b/filament/backend/include/backend/platforms/OpenGLPlatform.h @@ -267,6 +267,27 @@ public: * @return Transformed image. */ virtual AcquiredImage transformAcquiredImage(AcquiredImage source) noexcept; + + // -------------------------------------------------------------------------------------------- + + /** + * Returns true if additional OpenGL contexts can be created. Default: false. + * @return true if additional OpenGL contexts can be created. + * @see createContext + */ + virtual bool isExtraContextSupported() const noexcept; + + /** + * Creates an OpenGL context with the same configuration than the main context and makes it + * current to the current thread. Must not be called from the main driver thread. + * createContext() is only supported if isExtraContextSupported() returns true. + * These additional contexts will be automatically terminated in terminate. + * + * @param shared whether the new context is shared with the main context. + * @see isExtraContextSupported() + * @see terminate() + */ + virtual void createContext(bool shared); }; } // namespace filament diff --git a/filament/backend/include/backend/platforms/PlatformCocoaGL.h b/filament/backend/include/backend/platforms/PlatformCocoaGL.h index 97188852a3..df03bcbf29 100644 --- a/filament/backend/include/backend/platforms/PlatformCocoaGL.h +++ b/filament/backend/include/backend/platforms/PlatformCocoaGL.h @@ -50,6 +50,9 @@ protected: // -------------------------------------------------------------------------------------------- // OpenGLPlatform Interface + bool isExtraContextSupported() const noexcept override; + void createContext(bool shared) override; + void terminate() noexcept override; SwapChain* createSwapChain(void* nativewindow, uint64_t flags) noexcept override; diff --git a/filament/backend/include/backend/platforms/PlatformCocoaTouchGL.h b/filament/backend/include/backend/platforms/PlatformCocoaTouchGL.h index 5eb3b63b06..cbdd6a066a 100644 --- a/filament/backend/include/backend/platforms/PlatformCocoaTouchGL.h +++ b/filament/backend/include/backend/platforms/PlatformCocoaTouchGL.h @@ -47,6 +47,9 @@ public: uint32_t createDefaultRenderTarget() noexcept override; + bool isExtraContextSupported() const noexcept override; + void createContext(bool shared) override; + SwapChain* createSwapChain(void* nativewindow, uint64_t flags) noexcept override; SwapChain* createSwapChain(uint32_t width, uint32_t height, uint64_t flags) noexcept override; void destroySwapChain(SwapChain* swapChain) noexcept override; diff --git a/filament/backend/include/backend/platforms/PlatformEGL.h b/filament/backend/include/backend/platforms/PlatformEGL.h index 36d287c3e4..3a507c66fa 100644 --- a/filament/backend/include/backend/platforms/PlatformEGL.h +++ b/filament/backend/include/backend/platforms/PlatformEGL.h @@ -38,6 +38,8 @@ class PlatformEGL : public OpenGLPlatform { public: PlatformEGL() noexcept; + bool isExtraContextSupported() const noexcept override; + void createContext(bool shared) override; protected: @@ -124,6 +126,8 @@ protected: EGLSurface mCurrentReadSurface = EGL_NO_SURFACE; EGLSurface mEGLDummySurface = EGL_NO_SURFACE; EGLConfig mEGLConfig = EGL_NO_CONFIG_KHR; + Config mContextAttribs; + std::vector mAdditionalContexts; // supported extensions detected at runtime struct { diff --git a/filament/backend/include/backend/platforms/PlatformWGL.h b/filament/backend/include/backend/platforms/PlatformWGL.h index d18a20b2cd..6c16c30518 100644 --- a/filament/backend/include/backend/platforms/PlatformWGL.h +++ b/filament/backend/include/backend/platforms/PlatformWGL.h @@ -23,9 +23,10 @@ #include "utils/unwindows.h" #include - #include +#include + namespace filament::backend { /** @@ -46,6 +47,9 @@ protected: void terminate() noexcept override; + bool isExtraContextSupported() const noexcept override; + void createContext(bool shared) override; + SwapChain* createSwapChain(void* nativewindow, uint64_t flags) noexcept override; SwapChain* createSwapChain(uint32_t width, uint32_t height, uint64_t flags) noexcept override; void destroySwapChain(SwapChain* swapChain) noexcept override; @@ -57,6 +61,8 @@ protected: HWND mHWnd = NULL; HDC mWhdc = NULL; PIXELFORMATDESCRIPTOR mPfd = {}; + std::vector mAdditionalContexts; + std::vector mAttribs; }; } // namespace filament::backend diff --git a/filament/backend/src/Program.cpp b/filament/backend/src/Program.cpp index 5e7a944a00..8fd03f3e83 100644 --- a/filament/backend/src/Program.cpp +++ b/filament/backend/src/Program.cpp @@ -39,6 +39,11 @@ Program& Program::operator=(Program&& rhs) noexcept { Program::~Program() noexcept = default; +Program& Program::priorityQueue(CompilerPriorityQueue priorityQueue) noexcept { + mPriorityQueue = priorityQueue; + return *this; +} + Program& Program::diagnostics(CString const& name, Invocable&& logger) { mName = name; diff --git a/filament/backend/src/metal/MetalDriver.mm b/filament/backend/src/metal/MetalDriver.mm index ba30cff170..813e210095 100644 --- a/filament/backend/src/metal/MetalDriver.mm +++ b/filament/backend/src/metal/MetalDriver.mm @@ -977,7 +977,9 @@ void MetalDriver::updateSamplerGroup(Handle sbh, BufferDescripto void MetalDriver::compilePrograms(CallbackHandler* handler, CallbackHandler::Callback callback, void* user) { - scheduleCallback(handler, user, callback); + if (callback) { + scheduleCallback(handler, user, callback); + } } void MetalDriver::beginRenderPass(Handle rth, diff --git a/filament/backend/src/noop/NoopDriver.cpp b/filament/backend/src/noop/NoopDriver.cpp index ff83d9a1a2..5e0eee835c 100644 --- a/filament/backend/src/noop/NoopDriver.cpp +++ b/filament/backend/src/noop/NoopDriver.cpp @@ -262,7 +262,9 @@ void NoopDriver::updateSamplerGroup(Handle sbh, void NoopDriver::compilePrograms(CallbackHandler* handler, CallbackHandler::Callback callback, void* user) { - scheduleCallback(handler, user, callback); + if (callback) { + scheduleCallback(handler, user, callback); + } } void NoopDriver::beginRenderPass(Handle rth, const RenderPassParams& params) { diff --git a/filament/backend/src/opengl/OpenGLBlobCache.cpp b/filament/backend/src/opengl/OpenGLBlobCache.cpp index 8720639f54..48155dc23f 100644 --- a/filament/backend/src/opengl/OpenGLBlobCache.cpp +++ b/filament/backend/src/opengl/OpenGLBlobCache.cpp @@ -41,6 +41,7 @@ GLuint OpenGLBlobCache::retrieve(BlobCacheKey* outKey, Platform& platform, GLuint programId = 0; +#ifndef FILAMENT_SILENCE_NOT_SUPPORTED_BY_ES2 BlobCacheKey key{ program.getCacheId(), program.getSpecializationConstants() }; // FIXME: use a static buffer to avoid systematic allocation @@ -77,12 +78,14 @@ GLuint OpenGLBlobCache::retrieve(BlobCacheKey* outKey, Platform& platform, using std::swap; swap(*outKey, key); } +#endif return programId; } void OpenGLBlobCache::insert(Platform& platform, BlobCacheKey const& key, GLuint program) noexcept { +#ifndef FILAMENT_SILENCE_NOT_SUPPORTED_BY_ES2 SYSTRACE_CALL(); if (platform.hasBlobFunc()) { SYSTRACE_CONTEXT(); @@ -102,6 +105,21 @@ void OpenGLBlobCache::insert(Platform& platform, } } } +#endif +} + +void OpenGLBlobCache::insert(Platform& platform, BlobCacheKey const& key, + GLenum format, void* data, GLsizei programBinarySize) noexcept { + SYSTRACE_CALL(); + if (platform.hasBlobFunc()) { + if (programBinarySize) { + size_t const size = sizeof(Blob) + programBinarySize; + std::unique_ptr blob{ (Blob*)malloc(size), &::free }; + blob->format = format; + memcpy(blob->data, data, programBinarySize); + platform.insertBlob(key.data(), key.size(), blob.get(), size); + } + } } } // namespace filament::backend diff --git a/filament/backend/src/opengl/OpenGLBlobCache.h b/filament/backend/src/opengl/OpenGLBlobCache.h index cc78c429b9..5569fa2034 100644 --- a/filament/backend/src/opengl/OpenGLBlobCache.h +++ b/filament/backend/src/opengl/OpenGLBlobCache.h @@ -34,6 +34,9 @@ public: static void insert(Platform& platform, BlobCacheKey const& key, GLuint program) noexcept; + static void insert(Platform& platform, BlobCacheKey const& key, + GLenum format, void* data, GLsizei programBinarySize) noexcept; + private: struct Blob; }; diff --git a/filament/backend/src/opengl/OpenGLContext.cpp b/filament/backend/src/opengl/OpenGLContext.cpp index bf92dbb3d9..7e8a8f9f15 100644 --- a/filament/backend/src/opengl/OpenGLContext.cpp +++ b/filament/backend/src/opengl/OpenGLContext.cpp @@ -115,6 +115,9 @@ OpenGLContext::OpenGLContext() noexcept { procs.invalidateFramebuffer = glInvalidateFramebuffer; #endif // BACKEND_OPENGL_LEVEL_GLES30 + // no-op if not supported + procs.maxShaderCompilerThreadsKHR = +[](GLuint) {}; + #ifdef BACKEND_OPENGL_VERSION_GLES initExtensionsGLES(); if (state.major == 3) { @@ -173,6 +176,8 @@ OpenGLContext::OpenGLContext() noexcept { procs.invalidateFramebuffer = glDiscardFramebufferEXT; + procs.maxShaderCompilerThreadsKHR = glMaxShaderCompilerThreadsKHR; + mFeatureLevel = FeatureLevel::FEATURE_LEVEL_0; } #endif // IOS @@ -197,6 +202,8 @@ OpenGLContext::OpenGLContext() noexcept { bugs.allow_read_only_ancillary_feedback_loop = true; assert_invariant(gets.max_texture_image_units >= 16); assert_invariant(gets.max_combined_texture_image_units >= 32); + + procs.maxShaderCompilerThreadsKHR = glMaxShaderCompilerThreadsARB; #endif #ifdef GL_EXT_texture_filter_anisotropic diff --git a/filament/backend/src/opengl/OpenGLContext.h b/filament/backend/src/opengl/OpenGLContext.h index 2ba722e6fe..0e1e587b25 100644 --- a/filament/backend/src/opengl/OpenGLContext.h +++ b/filament/backend/src/opengl/OpenGLContext.h @@ -415,6 +415,8 @@ public: void (* getQueryObjectui64v)(GLuint id, GLenum pname, GLuint64* params); void (* invalidateFramebuffer)(GLenum target, GLsizei numAttachments, const GLenum *attachments); + + void (* maxShaderCompilerThreadsKHR)(GLuint count); } procs{}; private: diff --git a/filament/backend/src/opengl/OpenGLDriver.cpp b/filament/backend/src/opengl/OpenGLDriver.cpp index 05e5c4ecab..0fb3b9430b 100644 --- a/filament/backend/src/opengl/OpenGLDriver.cpp +++ b/filament/backend/src/opengl/OpenGLDriver.cpp @@ -50,13 +50,15 @@ #define HAS_MAPBUFFERS 1 #endif -#define DEBUG_MARKER_NONE 0 -#define DEBUG_MARKER_OPENGL 1 +#define DEBUG_MARKER_NONE 0x00 // no debug marker +#define DEBUG_MARKER_OPENGL 0x01 // markers in the gl command queue (req. driver support) +#define DEBUG_MARKER_BACKEND 0x02 // markers on the backend side (systrace) +#define DEBUG_MARKER_ALL 0x03 // all markers // set to the desired debug marker level #define DEBUG_MARKER_LEVEL DEBUG_MARKER_NONE -#if DEBUG_MARKER_LEVEL == DEBUG_MARKER_OPENGL +#if DEBUG_MARKER_LEVEL > DEBUG_MARKER_NONE # define DEBUG_MARKER() \ DebugMarker _debug_marker(*this, __func__); #else @@ -165,9 +167,11 @@ OpenGLDriver::DebugMarker::~DebugMarker() noexcept { // ------------------------------------------------------------------------------------------------ OpenGLDriver::OpenGLDriver(OpenGLPlatform* platform, const Platform::DriverConfig& driverConfig) noexcept - : mHandleAllocator("Handles", driverConfig.handleArenaSize), - mSamplerMap(32), - mPlatform(*platform) { + : mPlatform(*platform), + mContext(), + mShaderCompilerService(*this), + mHandleAllocator("Handles", driverConfig.handleArenaSize), + mSamplerMap(32) { std::fill(mSamplerBindings.begin(), mSamplerBindings.end(), nullptr); @@ -208,6 +212,8 @@ OpenGLDriver::OpenGLDriver(OpenGLPlatform* platform, const Platform::DriverConfi mTimerQueryImpl = new TimerQueryFallback(); mFrameTimeSupported = false; } + + mShaderCompilerService.init(); } OpenGLDriver::~OpenGLDriver() noexcept { // NOLINT(modernize-use-equals-default) @@ -243,6 +249,8 @@ void OpenGLDriver::terminate() { delete mTimerQueryImpl; + mShaderCompilerService.terminate(); + mPlatform.terminate(); } @@ -1455,7 +1463,6 @@ void OpenGLDriver::destroyProgram(Handle ph) { DEBUG_MARKER() if (ph) { OpenGLProgram* p = handle_cast(ph); - cancelRunAtNextPassOp(p); destruct(ph, p); } } @@ -2602,18 +2609,16 @@ SyncStatus OpenGLDriver::getSyncStatus(Handle sh) { void OpenGLDriver::compilePrograms(CallbackHandler* handler, CallbackHandler::Callback callback, void* user) { - // TODO: this works because currently `executeRenderPassOps` is only used for compiling - // materials. If that changed, we'd have to only execute the callbacks related to - // material compilation. - executeRenderPassOps(); - scheduleCallback(handler, user, callback); + if (callback) { + getShaderCompilerService().notifyWhenAllProgramsAreReady(handler, callback, user); + } } void OpenGLDriver::beginRenderPass(Handle rth, const RenderPassParams& params) { DEBUG_MARKER() - executeRenderPassOps(); + getShaderCompilerService().tick(); auto& gl = mContext; @@ -2954,29 +2959,34 @@ void OpenGLDriver::insertEventMarker(char const* string, uint32_t len) { } void OpenGLDriver::pushGroupMarker(char const* string, uint32_t len) { + #ifdef GL_EXT_debug_marker - auto& gl = mContext; - if (UTILS_LIKELY(gl.ext.EXT_debug_marker)) { +#if DEBUG_MARKER_LEVEL & DEBUG_MARKER_OPENGL + if (UTILS_LIKELY(mContext.ext.EXT_debug_marker)) { glPushGroupMarkerEXT(GLsizei(len ? len : strlen(string)), string); - } else -#endif - { - SYSTRACE_CONTEXT(); - SYSTRACE_NAME_BEGIN(string); } +#endif +#endif + +#if DEBUG_MARKER_LEVEL & DEBUG_MARKER_BACKEND + SYSTRACE_CONTEXT(); + SYSTRACE_NAME_BEGIN(string); +#endif } void OpenGLDriver::popGroupMarker(int) { #ifdef GL_EXT_debug_marker - auto& gl = mContext; - if (UTILS_LIKELY(gl.ext.EXT_debug_marker)) { +#if DEBUG_MARKER_LEVEL & DEBUG_MARKER_OPENGL + if (UTILS_LIKELY(mContext.ext.EXT_debug_marker)) { glPopGroupMarkerEXT(); - } else -#endif - { - SYSTRACE_CONTEXT(); - SYSTRACE_NAME_END(); } +#endif +#endif + +#if DEBUG_MARKER_LEVEL & DEBUG_MARKER_BACKEND + SYSTRACE_CONTEXT(); + SYSTRACE_NAME_END(); +#endif } void OpenGLDriver::startCapture(int) { @@ -3213,25 +3223,6 @@ void OpenGLDriver::executeEveryNowAndThenOps() noexcept { } } -void OpenGLDriver::runAtNextRenderPass(void* token, std::function fn) noexcept { - assert_invariant(mRunAtNextRenderPassOps.find(token) == mRunAtNextRenderPassOps.end()); - mRunAtNextRenderPassOps[token] = std::move(fn); -} - -void OpenGLDriver::cancelRunAtNextPassOp(void* token) noexcept { - mRunAtNextRenderPassOps.erase(token); -} - -void OpenGLDriver::executeRenderPassOps() noexcept { - auto& ops = mRunAtNextRenderPassOps; - if (!ops.empty()) { - for (auto& item: ops) { - item.second(); - } - ops.clear(); - } -} - // ------------------------------------------------------------------------------------------------ // Rendering ops // ------------------------------------------------------------------------------------------------ @@ -3240,6 +3231,7 @@ void OpenGLDriver::tick(int) { DEBUG_MARKER() executeGpuCommandsCompleteOps(); executeEveryNowAndThenOps(); + getShaderCompilerService().tick(); } void OpenGLDriver::beginFrame( @@ -3549,7 +3541,7 @@ void OpenGLDriver::draw(PipelineState state, Handle rph, uint } void OpenGLDriver::dispatchCompute(Handle program, math::uint3 workGroupCount) { - executeRenderPassOps(); + getShaderCompilerService().tick(); OpenGLProgram* p = handle_cast(program); diff --git a/filament/backend/src/opengl/OpenGLDriver.h b/filament/backend/src/opengl/OpenGLDriver.h index 24f3cb7180..efb512b526 100644 --- a/filament/backend/src/opengl/OpenGLDriver.h +++ b/filament/backend/src/opengl/OpenGLDriver.h @@ -20,6 +20,7 @@ #include "DriverBase.h" #include "GLUtils.h" #include "OpenGLContext.h" +#include "ShaderCompilerService.h" #include "private/backend/Driver.h" #include "private/backend/HandleAllocator.h" @@ -209,10 +210,16 @@ public: OpenGLDriver& operator=(OpenGLDriver const&) = delete; private: + OpenGLPlatform& mPlatform; OpenGLContext mContext; + ShaderCompilerService mShaderCompilerService; OpenGLContext& getContext() noexcept { return mContext; } + ShaderCompilerService& getShaderCompilerService() noexcept { + return mShaderCompilerService; + } + ShaderModel getShaderModel() const noexcept final; /* @@ -273,6 +280,7 @@ private: } friend class OpenGLProgram; + friend class ShaderCompilerService; /* Extension management... */ @@ -373,8 +381,6 @@ private: void detachStream(GLTexture* t) noexcept; void replaceStream(GLTexture* t, GLStream* stream) noexcept; - OpenGLPlatform& mPlatform; - void updateTextureLodRange(GLTexture* texture, int8_t targetLevel) noexcept; // tasks executed on the main thread after the fence signaled @@ -387,11 +393,6 @@ private: void executeEveryNowAndThenOps() noexcept; std::vector> mEveryNowAndThenOps; - void runAtNextRenderPass(void* token, std::function fn) noexcept; - void executeRenderPassOps() noexcept; - void cancelRunAtNextPassOp(void* token) noexcept; - tsl::robin_map> mRunAtNextRenderPassOps; - // timer query implementation OpenGLTimerQueryInterface* mTimerQueryImpl = nullptr; bool mFrameTimeSupported = false; diff --git a/filament/backend/src/opengl/OpenGLPlatform.cpp b/filament/backend/src/opengl/OpenGLPlatform.cpp index e44566d05f..4297479d97 100644 --- a/filament/backend/src/opengl/OpenGLPlatform.cpp +++ b/filament/backend/src/opengl/OpenGLPlatform.cpp @@ -109,4 +109,11 @@ TargetBufferFlags OpenGLPlatform::getPreservedFlags(UTILS_UNUSED SwapChain* swap return TargetBufferFlags::NONE; } +bool OpenGLPlatform::isExtraContextSupported() const noexcept { + return false; +} + +void OpenGLPlatform::createContext(bool) { +} + } // namespace filament::backend diff --git a/filament/backend/src/opengl/OpenGLProgram.cpp b/filament/backend/src/opengl/OpenGLProgram.cpp index 69d0f85f16..69ba1b6a3e 100644 --- a/filament/backend/src/opengl/OpenGLProgram.cpp +++ b/filament/backend/src/opengl/OpenGLProgram.cpp @@ -16,377 +16,88 @@ #include "OpenGLProgram.h" -#include "OpenGLBlobCache.h" -#include "OpenGLDriver.h" - #include "BlobCacheKey.h" +#include "OpenGLDriver.h" +#include "ShaderCompilerService.h" #include #include #include -#include #include #include -#include - namespace filament::backend { using namespace filament::math; using namespace utils; using namespace backend; -static void logCompilationError(utils::io::ostream& out, - ShaderStage shaderType, const char* name, - GLuint shaderId, CString const& sourceCode) noexcept; +struct OpenGLProgram::LazyInitializationData { + Program::UniformBlockInfo uniformBlockInfo; + Program::SamplerGroupInfo samplerGroupInfo; + std::array bindingUniformInfo; +}; -static void logProgramLinkError(utils::io::ostream& out, - const char* name, GLuint program) noexcept; -static inline std::string to_string(bool b) noexcept { - return b ? "true" : "false"; -} - -static inline std::string to_string(int i) noexcept { - return std::to_string(i); -} - -static inline std::string to_string(float f) noexcept { - return "float(" + std::to_string(f) + ")"; -} - -OpenGLProgram::OpenGLProgram() noexcept - : mInitialized(false), mValid(true), mLazyInitializationData(nullptr) { -} +OpenGLProgram::OpenGLProgram() noexcept = default; OpenGLProgram::OpenGLProgram(OpenGLDriver& gld, Program&& program) noexcept - : HwProgram(std::move(program.getName())), - mInitialized(false), mValid(true), - mLazyInitializationData{ new(LazyInitializationData) } { + : HwProgram(std::move(program.getName())) { - OpenGLContext& context = gld.getContext(); - - mLazyInitializationData->samplerGroupInfo = std::move(program.getSamplerGroupInfo()); + auto* const lazyInitializationData = new(std::nothrow) LazyInitializationData(); + lazyInitializationData->samplerGroupInfo = std::move(program.getSamplerGroupInfo()); if (UTILS_UNLIKELY(gld.getContext().isES2())) { - mLazyInitializationData->bindingUniformInfo = std::move(program.getBindingUniformInfo()); - mLazyInitializationData->attributes = std::move(program.getAttributes()); + lazyInitializationData->bindingUniformInfo = std::move(program.getBindingUniformInfo()); } else { - mLazyInitializationData->uniformBlockInfo = std::move(program.getUniformBlockBindings()); + lazyInitializationData->uniformBlockInfo = std::move(program.getUniformBlockBindings()); } - BlobCacheKey key; - gl.program = OpenGLBlobCache::retrieve(&key, gld.mPlatform, program); - if (!gl.program) { - // this cannot fail because we check compilation status after linking the program - // shaders[] is filled with id of shader stages present. - OpenGLProgram::compileShaders(context, - std::move(program.getShadersSource()), - program.getSpecializationConstants(), - gl.shaders, - mLazyInitializationData->shaderSourceCode); + ShaderCompilerService& compiler = gld.getShaderCompilerService(); + mToken = compiler.createProgram(name, std::move(program)); - gld.runAtNextRenderPass(this, [this, &gld, &context, key = std::move(key)]() { - // by this point we must not have a GL program - assert_invariant(!gl.program); - // we also can't be in the initialized state - assert_invariant(!mInitialized); - // we must have our lazy initialization data - assert_invariant(mLazyInitializationData); - // link the program, this also cannot fail because status is checked later. - gl.program = OpenGLProgram::linkProgram(context, - mLazyInitializationData, gl.shaders); - - if (key) { - // attempt to cache - OpenGLBlobCache::insert(gld.mPlatform, key, gl.program); - } - }); - } + ShaderCompilerService::setUserData(mToken, lazyInitializationData); } OpenGLProgram::~OpenGLProgram() noexcept { - if (!mInitialized) { - // mLazyInitializationData is aliased with mIndicesRuns - delete mLazyInitializationData; + if (mToken) { + // if the token is non-nullptr it means the program has not been used, and + // we need to clean-up. + assert_invariant(gl.program == 0); + + LazyInitializationData* const lazyInitializationData = + (LazyInitializationData *)ShaderCompilerService::getUserData(mToken); + delete lazyInitializationData; + + ShaderCompilerService::terminate(mToken); } + delete [] mUniformsRecords; const GLuint program = gl.program; - UTILS_NOUNROLL - for (GLuint const shader: gl.shaders) { - if (shader) { - if (program) { - glDetachShader(program, shader); - } - glDeleteShader(shader); - } - } if (program) { glDeleteProgram(program); } } -/* - * Compile shaders in the ShaderSource. This cannot fail because compilation failures are not - * checked until after the program is linked. - * This always returns the GL shader IDs or zero a shader stage is not present. - */ -void OpenGLProgram::compileShaders(OpenGLContext& context, - Program::ShaderSource shadersSource, - utils::FixedCapacityVector const& specializationConstants, - GLuint shaderIds[Program::SHADER_TYPE_COUNT], - UTILS_UNUSED_IN_RELEASE std::array& outShaderSourceCode) noexcept { +void OpenGLProgram::initialize(OpenGLDriver& gld) { SYSTRACE_CALL(); - auto appendSpecConstantString = +[](std::string& s, Program::SpecializationConstant const& sc) { - s += "#define SPIRV_CROSS_CONSTANT_ID_" + std::to_string(sc.id) + ' '; - s += std::visit([](auto&& arg) { return to_string(arg); }, sc.value); - s += '\n'; - return s; - }; + assert_invariant(gl.program == 0); + assert_invariant(mToken); - std::string specializationConstantString; - for (auto const& sc : specializationConstants) { - appendSpecConstantString(specializationConstantString, sc); + LazyInitializationData* const lazyInitializationData = + (LazyInitializationData *)ShaderCompilerService::getUserData(mToken); + + ShaderCompilerService& compiler = gld.getShaderCompilerService(); + gl.program = compiler.getProgram(mToken); + + assert_invariant(mToken == nullptr); + if (gl.program) { + assert_invariant(lazyInitializationData); + initializeProgramState(gld.getContext(), gl.program, *lazyInitializationData); + delete lazyInitializationData; } - if (!specializationConstantString.empty()) { - specializationConstantString += '\n'; - } - - // build all shaders - UTILS_NOUNROLL - for (size_t i = 0; i < Program::SHADER_TYPE_COUNT; i++) { - const ShaderStage stage = static_cast(i); - GLenum glShaderType{}; - switch (stage) { - case ShaderStage::VERTEX: - glShaderType = GL_VERTEX_SHADER; - break; - case ShaderStage::FRAGMENT: - glShaderType = GL_FRAGMENT_SHADER; - break; - case ShaderStage::COMPUTE: -#if defined(BACKEND_OPENGL_LEVEL_GLES31) - glShaderType = GL_COMPUTE_SHADER; -#else - continue; -#endif - break; - } - - if (UTILS_LIKELY(!shadersSource[i].empty())) { - Program::ShaderBlob& shader = shadersSource[i]; - - // remove GOOGLE_cpp_style_line_directive - std::string_view const source = process_GOOGLE_cpp_style_line_directive(context, - reinterpret_cast(shader.data()), shader.size()); - - // add support for ARB_shading_language_packing if needed - auto const packingFunctions = process_ARB_shading_language_packing(context); - - // split shader source, so we can insert the specification constants and the packing functions - auto const [prolog, body] = splitShaderSource(source); - - const std::array sources = { - prolog.data(), - specializationConstantString.c_str(), - packingFunctions.data(), - body.data() - }; - - const std::array lengths = { - (GLint)prolog.length(), - (GLint)specializationConstantString.length(), - (GLint)packingFunctions.length(), - (GLint)body.length() - 1 // null terminated - }; - - GLuint const shaderId = glCreateShader(glShaderType); - glShaderSource(shaderId, sources.size(), sources.data(), lengths.data()); - glCompileShader(shaderId); - -#ifndef NDEBUG - // for debugging we return the original shader source (without the modifications we - // made here), otherwise the line numbers wouldn't match. - outShaderSourceCode[i] = { source.data(), source.length() }; -#endif - - shaderIds[i] = shaderId; - } - } -} - -// If usages of the Google-style line directive are present, remove them, as some -// drivers don't allow the quotation marks. This happens in-place. -std::string_view OpenGLProgram::process_GOOGLE_cpp_style_line_directive(OpenGLContext& context, - char* source, size_t len) noexcept { - if (!context.ext.GOOGLE_cpp_style_line_directive) { - if (UTILS_UNLIKELY(requestsGoogleLineDirectivesExtension({ source, len }))) { - removeGoogleLineDirectives(source, len); // length is unaffected - } - } - return { source, len }; -} - -// Tragically, OpenGL 4.1 doesn't support unpackHalf2x16 (appeared in 4.2) and -// macOS doesn't support GL_ARB_shading_language_packing -std::string_view OpenGLProgram::process_ARB_shading_language_packing(OpenGLContext& context) noexcept { - using namespace std::literals; -#ifdef BACKEND_OPENGL_VERSION_GL - if (!context.isAtLeastGL<4, 2>() && !context.ext.ARB_shading_language_packing) { - return R"( - -// these don't handle denormals, NaNs or inf -float u16tofp32(highp uint v) { - v <<= 16u; - highp uint s = v & 0x80000000u; - highp uint n = v & 0x7FFFFFFFu; - highp uint nz = n == 0u ? 0u : 0xFFFFFFFF; - return uintBitsToFloat(s | ((((n >> 3u) + (0x70u << 23))) & nz)); -} -vec2 unpackHalf2x16(highp uint v) { - return vec2(u16tofp32(v&0xFFFFu), u16tofp32(v>>16u)); -} -uint fp32tou16(float val) { - uint f32 = floatBitsToUint(val); - uint f16 = 0u; - uint sign = (f32 >> 16) & 0x8000u; - int exponent = int((f32 >> 23) & 0xFFu) - 127; - uint mantissa = f32 & 0x007FFFFFu; - if (exponent > 15) { - f16 = sign | (0x1Fu << 10); - } else if (exponent > -15) { - exponent += 15; - mantissa >>= 13; - f16 = sign | uint(exponent << 10) | mantissa; - } else { - f16 = sign; - } - return f16; -} -highp uint packHalf2x16(vec2 v) { - highp uint x = fp32tou16(v.x); - highp uint y = fp32tou16(v.y); - return (y << 16) | x; -} -)"sv; - } -#endif // BACKEND_OPENGL_VERSION_GL - return ""sv; -} - -// split shader source code in two, the first section goes from the start to the line after the -// last #extension, and the 2nd part goes from there to the end. -std::array OpenGLProgram::splitShaderSource(std::string_view source) noexcept { - auto start = source.find("#version"); - assert_invariant(start != std::string_view::npos); - - auto pos = source.rfind("\n#extension"); - if (pos == std::string_view::npos) { - pos = start; - } else { - ++pos; - } - - auto eol = source.find('\n', pos) + 1; - assert_invariant(eol != std::string_view::npos); - - std::string_view const version = source.substr(start, eol - start); - std::string_view const body = source.substr(version.length(), source.length() - version.length()); - return { version, body }; -} - -/* - * Create a program from the given shader IDs and links it. This cannot fail because errors - * are checked later. This always returns a valid GL program ID (which doesn't mean the - * program itself is valid). - */ -GLuint OpenGLProgram::linkProgram(OpenGLContext& context, - LazyInitializationData* const lazyInitializationData, - const GLuint shaderIds[Program::SHADER_TYPE_COUNT]) noexcept { - - SYSTRACE_CALL(); - - GLuint const program = glCreateProgram(); - for (size_t i = 0; i < Program::SHADER_TYPE_COUNT; i++) { - if (shaderIds[i]) { - glAttachShader(program, shaderIds[i]); - } - } - - if (UTILS_UNLIKELY(context.isES2())) { - for (auto const& [ name, loc ] : lazyInitializationData->attributes) { - glBindAttribLocation(program, loc, name.c_str()); - } - } - - glLinkProgram(program); - return program; -} - -/* - * Checks a program link status and logs errors and frees resources on failure. - * Returns true on success. - */ -bool OpenGLProgram::checkProgramStatus(const char* name, - GLuint& program, GLuint shaderIds[Program::SHADER_TYPE_COUNT], - std::array const& shaderSourceCode) noexcept { - - SYSTRACE_CALL(); - - GLint status; - glGetProgramiv(program, GL_LINK_STATUS, &status); - if (UTILS_LIKELY(status == GL_TRUE)) { - return true; - } - - // only if the link fails, we check the compilation status - UTILS_NOUNROLL - for (size_t i = 0; i < Program::SHADER_TYPE_COUNT; i++) { - const ShaderStage type = static_cast(i); - const GLuint shader = shaderIds[i]; - if (shader) { - glGetShaderiv(shader, GL_COMPILE_STATUS, &status); - if (status != GL_TRUE) { - logCompilationError(slog.e, type, name, shader, shaderSourceCode[i]); - } - glDetachShader(program, shader); - glDeleteShader(shader); - shaderIds[i] = 0; - } - } - // log the link error as well - logProgramLinkError(slog.e, name, program); - glDeleteProgram(program); - program = 0; - return false; -} - -void OpenGLProgram::initialize(OpenGLContext& context) { - // by this point we must have a GL program - assert_invariant(gl.program); - // we also can't be in the initialized state - assert_invariant(!mInitialized); - // we must have our lazy initialization data - assert_invariant(mLazyInitializationData); - - // we must copy mLazyInitializationData locally because it is aliased with mIndicesRuns - auto* const pInitializationData = mLazyInitializationData; - - // check status of program linking and shader compilation, logs error and free all resources - // in case of error. - mValid = OpenGLProgram::checkProgramStatus(name.c_str_safe(), - gl.program, gl.shaders, pInitializationData->shaderSourceCode); - - if (UTILS_LIKELY(mValid)) { - initializeProgramState(context, gl.program, *pInitializationData); - } - - // and destroy all temporary init data - delete pInitializationData; - // mInitialized means mLazyInitializationData is no more valid - mInitialized = true; } /* @@ -592,64 +303,5 @@ void OpenGLProgram::setRec709ColorSpace(bool rec709) const noexcept { glUniform1i(mRec709Location, rec709); } -UTILS_NOINLINE -void logCompilationError(io::ostream& out, ShaderStage shaderType, - const char* name, GLuint shaderId, - UTILS_UNUSED_IN_RELEASE CString const& sourceCode) noexcept { - - auto to_string = [](ShaderStage type) -> const char* { - switch (type) { - case ShaderStage::VERTEX: return "vertex"; - case ShaderStage::FRAGMENT: return "fragment"; - case ShaderStage::COMPUTE: return "compute"; - } - }; - - { // scope for the temporary string storage - GLint length = 0; - glGetShaderiv(shaderId, GL_INFO_LOG_LENGTH, &length); - - CString infoLog(length); - glGetShaderInfoLog(shaderId, length, nullptr, infoLog.data()); - - out << "Compilation error in " << to_string(shaderType) << " shader \"" << name << "\":\n" - << "\"" << infoLog.c_str() << "\"" - << io::endl; - } - -#ifndef NDEBUG - std::string_view const shader{ sourceCode.data(), sourceCode.size() }; - size_t lc = 1; - size_t start = 0; - std::string line; - while (true) { - size_t const end = shader.find('\n', start); - if (end == std::string::npos) { - line = shader.substr(start); - } else { - line = shader.substr(start, end - start); - } - out << lc++ << ": " << line.c_str() << '\n'; - if (end == std::string::npos) { - break; - } - start = end + 1; - } - out << io::endl; -#endif -} - -UTILS_NOINLINE -void logProgramLinkError(io::ostream& out, char const* name, GLuint program) noexcept { - GLint length = 0; - glGetProgramiv(program, GL_INFO_LOG_LENGTH, &length); - - CString infoLog(length); - glGetProgramInfoLog(program, length, nullptr, infoLog.data()); - - out << "Link error in \"" << name << "\":\n" - << "\"" << infoLog.c_str() << "\"" - << io::endl; -} } // namespace filament::backend diff --git a/filament/backend/src/opengl/OpenGLProgram.h b/filament/backend/src/opengl/OpenGLProgram.h index a89005b974..5d183f6da1 100644 --- a/filament/backend/src/opengl/OpenGLProgram.h +++ b/filament/backend/src/opengl/OpenGLProgram.h @@ -18,22 +18,23 @@ #define TNT_FILAMENT_BACKEND_OPENGL_OPENGLPROGRAM_H #include "DriverBase.h" -#include "OpenGLDriver.h" -#include "private/backend/Driver.h" -#include "backend/Program.h" +#include "OpenGLContext.h" +#include "ShaderCompilerService.h" + +#include +#include #include -#include - -#include +#include #include #include - namespace filament::backend { +class OpenGLDriver; + class OpenGLProgram : public HwProgram { public: @@ -41,11 +42,11 @@ public: OpenGLProgram(OpenGLDriver& gld, Program&& program) noexcept; ~OpenGLProgram() noexcept; - bool isValid() const noexcept { return mValid; } + bool isValid() const noexcept { return mToken || gl.program != 0; } void use(OpenGLDriver* const gld, OpenGLContext& context) noexcept { - if (UTILS_UNLIKELY(!mInitialized)) { - initialize(context); + if (UTILS_UNLIKELY(!gl.program)) { + initialize(*gld); } context.useProgram(gl.program); @@ -67,7 +68,6 @@ public: } struct { - GLuint shaders[Program::SHADER_TYPE_COUNT] = {}; GLuint program = 0; } gl; // 12 bytes @@ -77,69 +77,29 @@ public: private: // keep these away from of other class attributes - struct LazyInitializationData { - Program::UniformBlockInfo uniformBlockInfo; - Program::SamplerGroupInfo samplerGroupInfo; - std::array bindingUniformInfo; - utils::FixedCapacityVector> attributes; - std::array shaderSourceCode; - }; + struct LazyInitializationData; - static void compileShaders(OpenGLContext& context, - Program::ShaderSource shadersSource, - utils::FixedCapacityVector const& specializationConstants, - GLuint shaderIds[Program::SHADER_TYPE_COUNT], - std::array& outShaderSourceCode) noexcept; - - static std::string_view process_GOOGLE_cpp_style_line_directive(OpenGLContext& context, - char* source, size_t len) noexcept; - - static std::string_view process_ARB_shading_language_packing(OpenGLContext& context) noexcept; - - static std::array splitShaderSource(std::string_view source) noexcept; - - static GLuint linkProgram(OpenGLContext& context, - LazyInitializationData* lazyInitializationData, - const GLuint shaderIds[Program::SHADER_TYPE_COUNT]) noexcept; - - static bool checkProgramStatus(const char* name, - GLuint& program, GLuint shaderIds[Program::SHADER_TYPE_COUNT], - std::array const& shaderSourceCode) noexcept; - - void initialize(OpenGLContext& context); + void initialize(OpenGLDriver& gld); void initializeProgramState(OpenGLContext& context, GLuint program, LazyInitializationData& lazyInitializationData) noexcept; void updateSamplers(OpenGLDriver* gld) const noexcept; + ShaderCompilerService::program_token_t mToken{}; // number of bindings actually used by this program uint8_t mUsedBindingsCount = 0u; - // whether lazy initialization has been performed - bool mInitialized : 1; - // whether lazy initialization was successful - bool mValid : 1; - UTILS_UNUSED uint8_t padding[2] = {}; + UTILS_UNUSED uint8_t padding[3] = {}; + std::array mUsedSamplerBindingPoints; // 4 bytes - // only for ES2 + // only needed for ES2 using LocationInfo = utils::FixedCapacityVector; struct UniformsRecord { Program::UniformInfo uniforms; LocationInfo locations; mutable uint16_t age = std::numeric_limits::max(); }; - - union { - // when mInitialized == true: - // information about each USED sampler buffer per binding (no gaps) - std::array mUsedSamplerBindingPoints; // 4 bytes - // when mInitialized == false: - // lazy initialization data pointer - LazyInitializationData* mLazyInitializationData; - }; - - // only needed for ES2 UniformsRecord const* mUniformsRecords = nullptr; GLint mRec709Location = -1; }; diff --git a/filament/backend/src/opengl/ShaderCompilerService.cpp b/filament/backend/src/opengl/ShaderCompilerService.cpp new file mode 100644 index 0000000000..7f052d3b06 --- /dev/null +++ b/filament/backend/src/opengl/ShaderCompilerService.cpp @@ -0,0 +1,900 @@ +/* + * Copyright (C) 2023 The Android Open Source Project + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "ShaderCompilerService.h" + +#include "BlobCacheKey.h" +#include "OpenGLBlobCache.h" +#include "OpenGLDriver.h" + +#include + +#include +#include + +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +namespace filament::backend { + +using namespace utils; + +// ------------------------------------------------------------------------------------------------ + +static void logCompilationError(utils::io::ostream& out, + ShaderStage shaderType, const char* name, + GLuint shaderId, CString const& sourceCode) noexcept; + +static void logProgramLinkError(utils::io::ostream& out, + const char* name, GLuint program) noexcept; + +static inline std::string to_string(bool b) noexcept { + return b ? "true" : "false"; +} + +static inline std::string to_string(int i) noexcept { + return std::to_string(i); +} + +static inline std::string to_string(float f) noexcept { + return "float(" + std::to_string(f) + ")"; +} + +// ------------------------------------------------------------------------------------------------ + +struct ShaderCompilerService::ProgramToken { + struct ProgramBinary { + GLenum format{}; + GLuint program{}; + std::vector blob; + }; + + ProgramToken(ShaderCompilerService& compiler, utils::CString const& name) noexcept + : compiler(compiler), name(name) { + } + ShaderCompilerService& compiler; + utils::CString const& name; + utils::FixedCapacityVector> attributes; + std::array shaderSourceCode; + void* user = nullptr; + struct { + std::array shaders{}; + GLuint program = 0; + } gl; // 12 bytes + + BlobCacheKey key; + std::future binary; + CompilerPriorityQueue priorityQueue = CompilerPriorityQueue::HIGH; + bool canceled = false; +}; + +void ShaderCompilerService::setUserData(const program_token_t& token, void* user) noexcept { + token->user = user; +} + +void* ShaderCompilerService::getUserData(const program_token_t& token) noexcept { + return token->user; +} + +// ------------------------------------------------------------------------------------------------ + +void ShaderCompilerService::CompilerThreadPool::init( + bool useSharedContexts, uint32_t threadCount, OpenGLPlatform& platform) noexcept { + + for (size_t i = 0; i < threadCount; i++) { + mCompilerThreads.emplace_back([this, useSharedContexts, &platform]() { + // give the thread a name + JobSystem::setThreadName("CompilerThreadPool"); + + // create a gl context current to this thread + platform.createContext(useSharedContexts); + + // process jobs from the queue until we're asked to exit + while (!mExitRequested) { + std::unique_lock lock(mQueueLock); + mQueueCondition.wait(lock, [this]() { + return mExitRequested || + mUrgentJob || + (!std::all_of( std::begin(mQueues), std::end(mQueues), + [](auto&& q) { return q.empty(); })); + }); + if (!mExitRequested) { + Job job{ std::move(mUrgentJob) }; + if (!job) { + // use the first queue that's not empty + auto& queue = [this]() -> auto& { + for (auto& q: mQueues) { + if (!q.empty()) { + return q; + } + } + return mQueues[0]; // we should never end-up here. + }(); + assert_invariant(!queue.empty()); + std::swap(job, queue.front().second); + queue.pop_front(); + } + + // execute the job without holding any locks + lock.unlock(); + job(); + } + } + }); + + } +} + +auto ShaderCompilerService::CompilerThreadPool::dequeue(program_token_t const& token) -> Job { + auto& q = mQueues[size_t(token->priorityQueue)]; + auto pos = std::find_if(q.begin(), q.end(), [&token](auto&& item) { + return item.first == token; + }); + Job job; + if (pos != q.end()) { + std::swap(job, pos->second); + q.erase(pos); + } + return job; +} + +void ShaderCompilerService::CompilerThreadPool::makeUrgent(program_token_t const& token) { + std::unique_lock const lock(mQueueLock); + assert_invariant(!mUrgentJob); + Job job{ dequeue(token) }; + std::swap(job, mUrgentJob); + mQueueCondition.notify_one(); +} + +void ShaderCompilerService::CompilerThreadPool::queue(program_token_t const& token, Job&& job) { + std::unique_lock const lock(mQueueLock); + mQueues[size_t(token->priorityQueue)].emplace_back(token, std::move(job)); + mQueueCondition.notify_one(); +} + +void ShaderCompilerService::CompilerThreadPool::exit() noexcept { + std::unique_lock lock(mQueueLock); + mExitRequested = true; + mQueueCondition.notify_all(); + lock.unlock(); + for (auto& thread: mCompilerThreads) { + if (thread.joinable()) { + thread.join(); + } + } +} + +// ------------------------------------------------------------------------------------------------ + +ShaderCompilerService::ShaderCompilerService(OpenGLDriver& driver) + : mDriver(driver), + KHR_parallel_shader_compile(driver.getContext().ext.KHR_parallel_shader_compile) { +} + +ShaderCompilerService::~ShaderCompilerService() noexcept = default; + +void ShaderCompilerService::init() noexcept { + // If we have KHR_parallel_shader_compile, we always use it, it should be more resource + // friendly. + if (!KHR_parallel_shader_compile) { + // - on Adreno there is a single compiler object. We can't use a pool > 1 + // also glProgramBinary blocks if other threads are compiling. + // - on Mali shader compilation can be multithreaded, but program linking happens on + // a single service thread, so we don't bother using more than one thread either. + // - on desktop we could use more threads, tbd. + if (mDriver.mPlatform.isExtraContextSupported()) { + mShaderCompilerThreadCount = 1; + mCompilerThreadPool.init(mUseSharedContext, + mShaderCompilerThreadCount, mDriver.mPlatform); + } + } +} + +void ShaderCompilerService::terminate() noexcept { + // FIXME: could we have some user callbacks pending here? + mCompilerThreadPool.exit(); +} + +ShaderCompilerService::program_token_t ShaderCompilerService::createProgram( + utils::CString const& name, Program&& program) { + auto& gl = mDriver.getContext(); + + auto token = std::make_shared(*this, name); + + if (UTILS_UNLIKELY(gl.isES2())) { + token->attributes = std::move(program.getAttributes()); + } + + token->gl.program = OpenGLBlobCache::retrieve(&token->key, mDriver.mPlatform, program); + if (!token->gl.program) { + if (mShaderCompilerThreadCount) { + // set the future in the token and pass the promise to the worker thread + std::promise promise; + token->binary = promise.get_future(); + token->priorityQueue = program.getPriorityQueue(); + // queue a compile job + mCompilerThreadPool.queue(token, + [this, &gl, promise = std::move(promise), + program = std::move(program), token]() mutable { + + // compile the shaders + std::array shaders{}; + std::array shaderSourceCode; + compileShaders(gl, + std::move(program.getShadersSource()), + program.getSpecializationConstants(), + shaders, + shaderSourceCode); + + // link the program + GLuint const glProgram = linkProgram(gl, shaders, token->attributes); + + // immediately destroy the shaders + for (GLuint const shader: shaders) { + if (shader) { + glDetachShader(glProgram, shader); + glDeleteShader(shader); + } + } + + ProgramToken::ProgramBinary binary; + if (UTILS_LIKELY(mUseSharedContext)) { + // We need to query the link status here to guarantee that the + // program is compiled and linked now (we don't want this to be + // deferred to later). We don't care about the result at this point. + GLint status; + glGetProgramiv(glProgram, GL_LINK_STATUS, &status); + binary.program = glProgram; + if (token->key) { + // Attempt to cache. This calls glGetProgramBinary. + OpenGLBlobCache::insert(mDriver.mPlatform, + token->key, token->gl.program); + } + } +#ifndef FILAMENT_SILENCE_NOT_SUPPORTED_BY_ES2 + else { + // retrieve the program binary + GLsizei programBinarySize = 0; + glGetProgramiv(glProgram, GL_PROGRAM_BINARY_LENGTH, &programBinarySize); + assert_invariant(programBinarySize); + if (programBinarySize) { + binary.blob.resize(programBinarySize); + glGetProgramBinary(glProgram, programBinarySize, + &programBinarySize, &binary.format, binary.blob.data()); + } + // and we can destroy the program + glDeleteProgram(glProgram); + if (token->key) { + // attempt to cache + OpenGLBlobCache::insert(mDriver.mPlatform, token->key, + binary.format, + binary.blob.data(), GLsizei(binary.blob.size())); + } + } +#endif + // we don't need to check for success here, it'll be done on the + // main thread side. + promise.set_value(binary); + }); + } else + { + // this cannot fail because we check compilation status after linking the program + // shaders[] is filled with id of shader stages present. + compileShaders(gl, + std::move(program.getShadersSource()), + program.getSpecializationConstants(), + token->gl.shaders, + token->shaderSourceCode); + + } + + runAtNextTick(token, [this, token]() { + if (mShaderCompilerThreadCount) { + if (!token->gl.program) { + // TODO: see if we could completely eliminate this callback here + // and instead just rely on token->gl.program being atomically + // set by the compiler thread. + assert_invariant(token->binary.valid()); + // we're using the compiler thread, check if the program is ready, no-op if not. + using namespace std::chrono_literals; + if (token->binary.wait_for(0s) != std::future_status::ready) { + return false; + } + // program binary is ready, retrieve it without blocking + getProgramFromCompilerPool(const_cast(token)); + } + } else { + if (KHR_parallel_shader_compile) { + // don't attempt to link this program if all shaders are not done compiling + GLint status; + if (token->gl.program) { + glGetProgramiv(token->gl.program, GL_COMPLETION_STATUS, &status); + if (status == GL_FALSE) { + return false; + } + } else { + for (auto shader: token->gl.shaders) { + if (shader) { + glGetShaderiv(shader, GL_COMPLETION_STATUS, &status); + if (status == GL_FALSE) { + return false; + } + } + } + } + } + + if (!token->gl.program) { + // link the program, this also cannot fail because status is checked later. + token->gl.program = linkProgram(mDriver.getContext(), + token->gl.shaders, token->attributes); + if (KHR_parallel_shader_compile) { + // wait until the link finishes... + return false; + } + } + } + + assert_invariant(token->gl.program); + + if (token->key && !mShaderCompilerThreadCount) { + // TODO: technically we don't have to cache right now. Is it advantageous to + // do this later, maybe depending on CPU usage? + // attempt to cache if we don't have a thread pool (otherwise it's done + // by the pool). + OpenGLBlobCache::insert(mDriver.mPlatform, token->key, token->gl.program); + } + + return true; + }); + } + + return token; +} + +bool ShaderCompilerService::isProgramReady( + const ShaderCompilerService::program_token_t& token) const noexcept { + + assert_invariant(token); + + if (!token->gl.program) { + return false; + } + + if (KHR_parallel_shader_compile) { + GLint status = GL_FALSE; + glGetProgramiv(token->gl.program, GL_COMPLETION_STATUS, &status); + return (bool)status; + } + + // If gl.program is set, this means the program was linked. Some drivers may defer the link + // in which case we might block in getProgram() when we check the program status. + // Unfortunately, this is nothing we can do about that. + return bool(token->gl.program); +} + +GLuint ShaderCompilerService::getProgram(ShaderCompilerService::program_token_t& token) { + GLuint const program = initialize(token); + assert_invariant(token == nullptr); + assert_invariant(program); + return program; +} + +/* static*/ void ShaderCompilerService::terminate(program_token_t& token) { + assert_invariant(token); + + token->canceled = true; + + token->compiler.cancelTickOp(token); + + for (GLuint& shader: token->gl.shaders) { + if (shader) { + if (token->gl.program) { + glDetachShader(token->gl.program, shader); + } + glDeleteShader(shader); + shader = 0; + } + } + if (token->gl.program) { + glDeleteProgram(token->gl.program); + } + + token = nullptr; +} + +void ShaderCompilerService::tick() { + executeTickOps(); +} + +void ShaderCompilerService::notifyWhenAllProgramsAreReady(CallbackHandler* handler, + CallbackHandler::Callback callback, void* user) { + + if (KHR_parallel_shader_compile || mShaderCompilerThreadCount) { + // list all programs up to this point + utils::FixedCapacityVector, false> tokens; + tokens.reserve(mRunAtNextTickOps.size()); + for (auto& [token, _] : mRunAtNextTickOps) { + if (token) { + tokens.push_back(token); + } + } + + runAtNextTick(nullptr, [this, tokens = std::move(tokens), handler, user, callback]() { + for (auto const& token : tokens) { + assert_invariant(token); + if (!isProgramReady(token)) { + // one of the program is not ready, try next time + return false; + } + } + if (callback) { + // all programs are ready, we can call the callbacks + mDriver.scheduleCallback(handler, user, callback); + } + // and we're done + return true; + }); + + return; + } + + // we don't have KHR_parallel_shader_compile + + runAtNextTick(nullptr, [this, handler, user, callback]() { + mDriver.scheduleCallback(handler, user, callback); + return true; + }); + + // TODO: we could spread the compiles over several frames, the tick() below then is not + // needed here. We keep it for now as to not change the current behavior too much. + // this will block until all programs are linked + tick(); +} + +// ------------------------------------------------------------------------------------------------ + +void ShaderCompilerService::getProgramFromCompilerPool(program_token_t& token) noexcept { + ProgramToken::ProgramBinary const binary{ token->binary.get() }; + if (!token->canceled) { + if (UTILS_LIKELY(mUseSharedContext)) { + token->gl.program = binary.program; + } +#ifndef FILAMENT_SILENCE_NOT_SUPPORTED_BY_ES2 + else { + token->gl.program = glCreateProgram(); + glProgramBinary(token->gl.program, binary.format, + binary.blob.data(), GLsizei(binary.blob.size())); + } +#endif + } +} + +GLuint ShaderCompilerService::initialize(program_token_t& token) noexcept { + SYSTRACE_CALL(); + if (!token->gl.program) { + if (mShaderCompilerThreadCount) { + // Block until the program is ready. This could take a very long time. + assert_invariant(token->binary.valid()); + + // we need this program right now, so move it to the head of the queue. + mCompilerThreadPool.makeUrgent(token); + + if (!token->canceled) { + token->compiler.cancelTickOp(token); + } + + // block until we get the program from the pool + getProgramFromCompilerPool(token); + } else if (KHR_parallel_shader_compile) { + // we force the program link -- which might stall, either here or below in + // checkProgramStatus(), but we don't have a choice, we need to use the program now. + token->compiler.cancelTickOp(token); + token->gl.program = linkProgram(mDriver.getContext(), + token->gl.shaders, token->attributes); + } else { + // if we don't have a program yet, block until we get it. + tick(); + } + } + + // by this point we must have a GL program + assert_invariant(token->gl.program); + + GLuint program = 0; + + // check status of program linking and shader compilation, logs error and free all resources + // in case of error. + bool const success = checkProgramStatus(token); + if (UTILS_LIKELY(success)) { + program = token->gl.program; + // no need to keep the shaders around + UTILS_NOUNROLL + for (GLuint& shader: token->gl.shaders) { + if (shader) { + glDetachShader(program, shader); + glDeleteShader(shader); + shader = 0; + } + } + } + + // and destroy all temporary init data + token = nullptr; + + return program; +} + + +/* + * Compile shaders in the ShaderSource. This cannot fail because compilation failures are not + * checked until after the program is linked. + * This always returns the GL shader IDs or zero a shader stage is not present. + */ +void ShaderCompilerService::compileShaders(OpenGLContext& context, + Program::ShaderSource shadersSource, + utils::FixedCapacityVector const& specializationConstants, + std::array& outShaders, + UTILS_UNUSED_IN_RELEASE std::array& outShaderSourceCode) noexcept { + + SYSTRACE_CALL(); + + auto appendSpecConstantString = +[](std::string& s, Program::SpecializationConstant const& sc) { + s += "#define SPIRV_CROSS_CONSTANT_ID_" + std::to_string(sc.id) + ' '; + s += std::visit([](auto&& arg) { return to_string(arg); }, sc.value); + s += '\n'; + return s; + }; + + std::string specializationConstantString; + for (auto const& sc : specializationConstants) { + appendSpecConstantString(specializationConstantString, sc); + } + if (!specializationConstantString.empty()) { + specializationConstantString += '\n'; + } + + // build all shaders + UTILS_NOUNROLL + for (size_t i = 0; i < Program::SHADER_TYPE_COUNT; i++) { + const ShaderStage stage = static_cast(i); + GLenum glShaderType{}; + switch (stage) { + case ShaderStage::VERTEX: + glShaderType = GL_VERTEX_SHADER; + break; + case ShaderStage::FRAGMENT: + glShaderType = GL_FRAGMENT_SHADER; + break; + case ShaderStage::COMPUTE: +#if defined(BACKEND_OPENGL_LEVEL_GLES31) + glShaderType = GL_COMPUTE_SHADER; +#else + continue; +#endif + break; + } + + if (UTILS_LIKELY(!shadersSource[i].empty())) { + Program::ShaderBlob& shader = shadersSource[i]; + + // remove GOOGLE_cpp_style_line_directive + std::string_view const source = process_GOOGLE_cpp_style_line_directive(context, + reinterpret_cast(shader.data()), shader.size()); + + // add support for ARB_shading_language_packing if needed + auto const packingFunctions = process_ARB_shading_language_packing(context); + + // split shader source, so we can insert the specification constants and the packing functions + auto const [prolog, body] = splitShaderSource(source); + + const std::array sources = { + prolog.data(), + specializationConstantString.c_str(), + packingFunctions.data(), + body.data() + }; + + const std::array lengths = { + (GLint)prolog.length(), + (GLint)specializationConstantString.length(), + (GLint)packingFunctions.length(), + (GLint)body.length() - 1 // null terminated + }; + + GLuint const shaderId = glCreateShader(glShaderType); + glShaderSource(shaderId, sources.size(), sources.data(), lengths.data()); + glCompileShader(shaderId); + +#ifndef NDEBUG + // for debugging we return the original shader source (without the modifications we + // made here), otherwise the line numbers wouldn't match. + outShaderSourceCode[i] = { source.data(), source.length() }; +#endif + + outShaders[i] = shaderId; + } + } +} + +// If usages of the Google-style line directive are present, remove them, as some +// drivers don't allow the quotation marks. This happens in-place. +std::string_view ShaderCompilerService::process_GOOGLE_cpp_style_line_directive(OpenGLContext& context, + char* source, size_t len) noexcept { + if (!context.ext.GOOGLE_cpp_style_line_directive) { + if (UTILS_UNLIKELY(requestsGoogleLineDirectivesExtension({ source, len }))) { + removeGoogleLineDirectives(source, len); // length is unaffected + } + } + return { source, len }; +} + +// Tragically, OpenGL 4.1 doesn't support unpackHalf2x16 (appeared in 4.2) and +// macOS doesn't support GL_ARB_shading_language_packing +std::string_view ShaderCompilerService::process_ARB_shading_language_packing(OpenGLContext& context) noexcept { + using namespace std::literals; +#ifdef BACKEND_OPENGL_VERSION_GL + if (!context.isAtLeastGL<4, 2>() && !context.ext.ARB_shading_language_packing) { + return R"( + +// these don't handle denormals, NaNs or inf +float u16tofp32(highp uint v) { + v <<= 16u; + highp uint s = v & 0x80000000u; + highp uint n = v & 0x7FFFFFFFu; + highp uint nz = n == 0u ? 0u : 0xFFFFFFFF; + return uintBitsToFloat(s | ((((n >> 3u) + (0x70u << 23))) & nz)); +} +vec2 unpackHalf2x16(highp uint v) { + return vec2(u16tofp32(v&0xFFFFu), u16tofp32(v>>16u)); +} +uint fp32tou16(float val) { + uint f32 = floatBitsToUint(val); + uint f16 = 0u; + uint sign = (f32 >> 16) & 0x8000u; + int exponent = int((f32 >> 23) & 0xFFu) - 127; + uint mantissa = f32 & 0x007FFFFFu; + if (exponent > 15) { + f16 = sign | (0x1Fu << 10); + } else if (exponent > -15) { + exponent += 15; + mantissa >>= 13; + f16 = sign | uint(exponent << 10) | mantissa; + } else { + f16 = sign; + } + return f16; +} +highp uint packHalf2x16(vec2 v) { + highp uint x = fp32tou16(v.x); + highp uint y = fp32tou16(v.y); + return (y << 16) | x; +} +)"sv; + } +#endif // BACKEND_OPENGL_VERSION_GL + return ""sv; +} + +// split shader source code in two, the first section goes from the start to the line after the +// last #extension, and the 2nd part goes from there to the end. +std::array ShaderCompilerService::splitShaderSource(std::string_view source) noexcept { + auto start = source.find("#version"); + assert_invariant(start != std::string_view::npos); + + auto pos = source.rfind("\n#extension"); + if (pos == std::string_view::npos) { + pos = start; + } else { + ++pos; + } + + auto eol = source.find('\n', pos) + 1; + assert_invariant(eol != std::string_view::npos); + + std::string_view const version = source.substr(start, eol - start); + std::string_view const body = source.substr(version.length(), source.length() - version.length()); + return { version, body }; +} + +/* + * Create a program from the given shader IDs and links it. This cannot fail because errors + * are checked later. This always returns a valid GL program ID (which doesn't mean the + * program itself is valid). + */ +GLuint ShaderCompilerService::linkProgram(OpenGLContext& context, + std::array shaders, + utils::FixedCapacityVector> const& attributes) noexcept { + + SYSTRACE_CALL(); + + GLuint const program = glCreateProgram(); + for (auto shader : shaders) { + if (shader) { + glAttachShader(program, shader); + } + } + + if (UTILS_UNLIKELY(context.isES2())) { + for (auto const& [ name, loc ] : attributes) { + glBindAttribLocation(program, loc, name.c_str()); + } + } + + glLinkProgram(program); + + return program; +} + +// ------------------------------------------------------------------------------------------------ + +void ShaderCompilerService::runAtNextTick( + const program_token_t& token, std::function fn) noexcept { + // insert items in order of priority and at the end of the range + auto& ops = mRunAtNextTickOps; + using ContainerType = std::pair>; + auto const pos = std::lower_bound(ops.begin(), ops.end(), + token->priorityQueue, + [](ContainerType const& lhs, CompilerPriorityQueue priorityQueue) { + return lhs.first->priorityQueue < priorityQueue; + }); + ops.emplace(pos, token, std::move(fn)); + + SYSTRACE_CONTEXT(); + SYSTRACE_VALUE32("ShaderCompilerService Jobs", mRunAtNextTickOps.size()); +} + +void ShaderCompilerService::cancelTickOp(program_token_t token) noexcept { + // We do a linear search here, but this is rare, and we know the list is pretty small. + auto& ops = mRunAtNextTickOps; + auto pos = std::find_if(ops.begin(), ops.end(), + [&](const auto& item) { + return item.first == token; + }); + if (pos != ops.end()) { + ops.erase(pos); + } + SYSTRACE_CONTEXT(); + SYSTRACE_VALUE32("ShaderCompilerService Jobs", ops.size()); +} + +void ShaderCompilerService::executeTickOps() noexcept { + auto& ops = mRunAtNextTickOps; + auto it = ops.begin(); + while (it != ops.end()) { + bool const remove = it->second(); + if (remove) { + it = ops.erase(it); + } else { + ++it; + } + } + SYSTRACE_CONTEXT(); + SYSTRACE_VALUE32("ShaderCompilerService Jobs", ops.size()); +} + +// ------------------------------------------------------------------------------------------------ + +/* + * Checks a program link status and logs errors and frees resources on failure. + * Returns true on success. + */ +bool ShaderCompilerService::checkProgramStatus(program_token_t const& token) noexcept { + + SYSTRACE_CALL(); + + assert_invariant(token->gl.program); + + GLint status; + glGetProgramiv(token->gl.program, GL_LINK_STATUS, &status); + if (UTILS_LIKELY(status == GL_TRUE)) { + return true; + } + + // only if the link fails, we check the compilation status + UTILS_NOUNROLL + for (size_t i = 0; i < Program::SHADER_TYPE_COUNT; i++) { + const ShaderStage type = static_cast(i); + const GLuint shader = token->gl.shaders[i]; + if (shader) { + glGetShaderiv(shader, GL_COMPILE_STATUS, &status); + if (status != GL_TRUE) { + logCompilationError(slog.e, type, + token->name.c_str_safe(), shader, token->shaderSourceCode[i]); + } + glDetachShader(token->gl.program, shader); + glDeleteShader(shader); + token->gl.shaders[i] = 0; + } + } + // log the link error as well + logProgramLinkError(slog.e, token->name.c_str_safe(), token->gl.program); + glDeleteProgram(token->gl.program); + token->gl.program = 0; + return false; +} + +UTILS_NOINLINE +void logCompilationError(io::ostream& out, ShaderStage shaderType, + const char* name, GLuint shaderId, + UTILS_UNUSED_IN_RELEASE CString const& sourceCode) noexcept { + + auto to_string = [](ShaderStage type) -> const char* { + switch (type) { + case ShaderStage::VERTEX: return "vertex"; + case ShaderStage::FRAGMENT: return "fragment"; + case ShaderStage::COMPUTE: return "compute"; + } + }; + + { // scope for the temporary string storage + GLint length = 0; + glGetShaderiv(shaderId, GL_INFO_LOG_LENGTH, &length); + + CString infoLog(length); + glGetShaderInfoLog(shaderId, length, nullptr, infoLog.data()); + + out << "Compilation error in " << to_string(shaderType) << " shader \"" << name << "\":\n" + << "\"" << infoLog.c_str() << "\"" + << io::endl; + } + +#ifndef NDEBUG + std::string_view const shader{ sourceCode.data(), sourceCode.size() }; + size_t lc = 1; + size_t start = 0; + std::string line; + while (true) { + size_t const end = shader.find('\n', start); + if (end == std::string::npos) { + line = shader.substr(start); + } else { + line = shader.substr(start, end - start); + } + out << lc++ << ": " << line.c_str() << '\n'; + if (end == std::string::npos) { + break; + } + start = end + 1; + } + out << io::endl; +#endif +} + +UTILS_NOINLINE +void logProgramLinkError(io::ostream& out, char const* name, GLuint program) noexcept { + GLint length = 0; + glGetProgramiv(program, GL_INFO_LOG_LENGTH, &length); + + CString infoLog(length); + glGetProgramInfoLog(program, length, nullptr, infoLog.data()); + + out << "Link error in \"" << name << "\":\n" + << "\"" << infoLog.c_str() << "\"" + << io::endl; +} + + +} // namespace filament::backend diff --git a/filament/backend/src/opengl/ShaderCompilerService.h b/filament/backend/src/opengl/ShaderCompilerService.h new file mode 100644 index 0000000000..0fccf7525c --- /dev/null +++ b/filament/backend/src/opengl/ShaderCompilerService.h @@ -0,0 +1,154 @@ +/* + * Copyright (C) 2023 The Android Open Source Project + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef TNT_FILAMENT_BACKEND_OPENGL_SHADERCOMPILERSERVICE_H +#define TNT_FILAMENT_BACKEND_OPENGL_SHADERCOMPILERSERVICE_H + +#include "gl_headers.h" + +#include +#include + +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace filament::backend { + +class OpenGLDriver; +class OpenGLContext; +class OpenGLPlatform; +class Program; +class CallbackHandler; + +/* + * A class handling shader compilation that supports asynchronous compilation. + */ +class ShaderCompilerService { + struct ProgramToken; + +public: + using program_token_t = std::shared_ptr; + + explicit ShaderCompilerService(OpenGLDriver& driver); + + ShaderCompilerService(ShaderCompilerService const& rhs) = delete; + ShaderCompilerService(ShaderCompilerService&& rhs) = delete; + ShaderCompilerService& operator=(ShaderCompilerService const& rhs) = delete; + ShaderCompilerService& operator=(ShaderCompilerService&& rhs) = delete; + + ~ShaderCompilerService() noexcept; + + void init() noexcept; + void terminate() noexcept; + + // creates a program (compile + link) asynchronously if supported + program_token_t createProgram(utils::CString const& name, Program&& program); + + // Returns true if the program is linked (successfully or not). Guarantees that + // getProgram() won't block. Does not block. + bool isProgramReady(const program_token_t& token) const noexcept; + + // Return the GL program, blocks if necessary. The Token is destroyed and becomes invalid. + GLuint getProgram(program_token_t& token); + + // Must be called at least once per frame. + void tick(); + + // Destroys a valid token and all associated resources. Used to "cancel" a program compilation. + static void terminate(program_token_t& token); + + // stores a user data pointer in the token + static void setUserData(const program_token_t& token, void* user) noexcept; + + // retrieves the user data pointer stored in the token + static void* getUserData(const program_token_t& token) noexcept; + + // call the callback when all active programs are ready + void notifyWhenAllProgramsAreReady(CallbackHandler* handler, + CallbackHandler::Callback callback, void* user); + +private: + class CompilerThreadPool { + public: + using Job = utils::Invocable; + void init(bool useSharedContexts, uint32_t threadCount, OpenGLPlatform& platform) noexcept; + void exit() noexcept; + void queue(program_token_t const& token, Job&& job); + void makeUrgent(program_token_t const& token); + + private: + std::vector mCompilerThreads; + std::atomic_bool mExitRequested{ false }; + std::mutex mQueueLock; + std::condition_variable mQueueCondition; + std::array>, 2> mQueues; + Job mUrgentJob; + Job dequeue(program_token_t const& token); // lock must be held + }; + + OpenGLDriver& mDriver; + CompilerThreadPool mCompilerThreadPool; + + const bool KHR_parallel_shader_compile; + uint32_t mShaderCompilerThreadCount = 0u; + + // For now, we assume shared contexts are supported everywhere. If they are not, + // we don't use the shader compiler pool. However, the code supports it. + static constexpr bool mUseSharedContext = true; + + GLuint initialize(ShaderCompilerService::program_token_t& token) noexcept; + + void getProgramFromCompilerPool(program_token_t& token) noexcept; + + static void compileShaders( + OpenGLContext& context, + Program::ShaderSource shadersSource, + utils::FixedCapacityVector const& specializationConstants, + std::array& outShaders, + std::array& outShaderSourceCode) noexcept; + + static std::string_view process_GOOGLE_cpp_style_line_directive(OpenGLContext& context, + char* source, size_t len) noexcept; + + static std::string_view process_ARB_shading_language_packing(OpenGLContext& context) noexcept; + + static std::array splitShaderSource(std::string_view source) noexcept; + + static GLuint linkProgram(OpenGLContext& context, + std::array shaders, + utils::FixedCapacityVector> const& attributes) noexcept; + + static bool checkProgramStatus(program_token_t const& token) noexcept; + + void runAtNextTick(const program_token_t& token, std::function fn) noexcept; + void executeTickOps() noexcept; + void cancelTickOp(program_token_t token) noexcept; + // order of insertion is important + std::vector>> mRunAtNextTickOps; +}; + +} // namespace filament::backend + +#endif // TNT_FILAMENT_BACKEND_OPENGL_SHADERCOMPILERSERVICE_H diff --git a/filament/backend/src/opengl/gl_headers.h b/filament/backend/src/opengl/gl_headers.h index f00351fcb0..9bcb8ffdcc 100644 --- a/filament/backend/src/opengl/gl_headers.h +++ b/filament/backend/src/opengl/gl_headers.h @@ -174,6 +174,12 @@ using namespace glext; # define GL_ZERO_TO_ONE GL_ZERO_TO_ONE_EXT #endif +#ifdef GL_KHR_parallel_shader_compile +# define GL_COMPLETION_STATUS GL_COMPLETION_STATUS_KHR +#else +# define GL_COMPLETION_STATUS 0x91B1 +#endif + // we need GL_TEXTURE_CUBE_MAP_ARRAY defined, but we won't use it if the extension/feature // is not available. #if defined(GL_EXT_texture_cube_map_array) @@ -223,6 +229,14 @@ using namespace glext; # endif #endif +#else // All version OpenGL below + +#ifdef GL_ARB_parallel_shader_compile +# define GL_COMPLETION_STATUS GL_COMPLETION_STATUS_ARB +#else +# define GL_COMPLETION_STATUS 0x91B1 +#endif + #endif // GL_ES_VERSION_2_0 // This is just to simplify the implementation (i.e. so we don't have to have #ifdefs everywhere) diff --git a/filament/backend/src/opengl/platforms/PlatformCocoaGL.mm b/filament/backend/src/opengl/platforms/PlatformCocoaGL.mm index 34a76078ae..195d8d17e7 100644 --- a/filament/backend/src/opengl/platforms/PlatformCocoaGL.mm +++ b/filament/backend/src/opengl/platforms/PlatformCocoaGL.mm @@ -50,6 +50,7 @@ struct PlatformCocoaGLImpl { NSOpenGLContext* mGLContext = nullptr; CocoaGLSwapChain* mCurrentSwapChain = nullptr; std::vector mHeadlessSwapChains; + std::vector mAdditionalContexts; CVOpenGLTextureCacheRef mTextureCache = nullptr; std::unique_ptr mExternalImageSharedGl; void updateOpenGLContext(NSView *nsView, bool resetView, bool clearView); @@ -171,6 +172,23 @@ Driver* PlatformCocoaGL::createDriver(void* sharedContext, const Platform::Drive return OpenGLPlatform::createDefaultDriver(this, sharedContext, driverConfig); } +bool PlatformCocoaGL::isExtraContextSupported() const noexcept { + return true; +} + +void PlatformCocoaGL::createContext(bool shared) { + NSOpenGLPixelFormatAttribute pixelFormatAttributes[] = { + NSOpenGLPFAOpenGLProfile, NSOpenGLProfileVersion3_2Core, + NSOpenGLPFAAccelerated, (NSOpenGLPixelFormatAttribute) true, + 0, 0, + }; + NSOpenGLContext* const sharedContext = shared ? pImpl->mGLContext : nil; + NSOpenGLPixelFormat* const pixelFormat = [[NSOpenGLPixelFormat alloc] initWithAttributes:pixelFormatAttributes]; + NSOpenGLContext* const nsOpenGLContext = [[NSOpenGLContext alloc] initWithFormat:pixelFormat shareContext:sharedContext]; + [nsOpenGLContext makeCurrentContext]; + pImpl->mAdditionalContexts.push_back(nsOpenGLContext); +} + int PlatformCocoaGL::getOSVersion() const noexcept { return 0; } @@ -179,6 +197,9 @@ void PlatformCocoaGL::terminate() noexcept { CFRelease(pImpl->mTextureCache); pImpl->mExternalImageSharedGl.reset(); pImpl->mGLContext = nil; + for (auto& context : pImpl->mAdditionalContexts) { + context = nil; + } bluegl::unbind(); } diff --git a/filament/backend/src/opengl/platforms/PlatformCocoaTouchGL.mm b/filament/backend/src/opengl/platforms/PlatformCocoaTouchGL.mm index 0fc33c1369..0d375df103 100644 --- a/filament/backend/src/opengl/platforms/PlatformCocoaTouchGL.mm +++ b/filament/backend/src/opengl/platforms/PlatformCocoaTouchGL.mm @@ -40,6 +40,7 @@ struct PlatformCocoaTouchGLImpl { EAGLContext* mGLContext = nullptr; CAEAGLLayer* mCurrentGlLayer = nullptr; std::vector mHeadlessGlLayers; + std::vector mAdditionalContexts; CGRect mCurrentGlLayerRect; GLuint mDefaultFramebuffer = 0; GLuint mDefaultColorbuffer = 0; @@ -93,6 +94,20 @@ Driver* PlatformCocoaTouchGL::createDriver(void* const sharedGLContext, const Pl return OpenGLPlatform::createDefaultDriver(this, sharedGLContext, driverConfig); } +bool PlatformCocoaTouchGL::isExtraContextSupported() const noexcept { + return true; +} + +void PlatformCocoaTouchGL::createContext(bool shared) { + EAGLSharegroup* const sharegroup = shared ? pImpl->mGLContext.sharegroup : nil; + EAGLContext* const context = [[EAGLContext alloc] + initWithAPI:kEAGLRenderingAPIOpenGLES3 + sharegroup:sharegroup]; + ASSERT_POSTCONDITION(context, "Unable to create extra OpenGL ES context."); + [EAGLContext setCurrentContext:context]; + pImpl->mAdditionalContexts.push_back(context); +} + void PlatformCocoaTouchGL::terminate() noexcept { CFRelease(pImpl->mTextureCache); pImpl->mGLContext = nil; diff --git a/filament/backend/src/opengl/platforms/PlatformEGL.cpp b/filament/backend/src/opengl/platforms/PlatformEGL.cpp index f52cb83c5e..a674bccb80 100644 --- a/filament/backend/src/opengl/platforms/PlatformEGL.cpp +++ b/filament/backend/src/opengl/platforms/PlatformEGL.cpp @@ -27,6 +27,7 @@ #include #include +#include #ifndef EGL_CONTEXT_OPENGL_BACKWARDS_COMPATIBLE_ANGLE # define EGL_CONTEXT_OPENGL_BACKWARDS_COMPATIBLE_ANGLE 0x3483 @@ -225,6 +226,8 @@ Driver* PlatformEGL::createDriver(void* sharedContext, const Platform::DriverCon goto error; } + mContextAttribs = std::move(contextAttribs); + initializeGlExtensions(); // this is needed with older emulators/API levels on Android @@ -251,6 +254,26 @@ error: return nullptr; } +bool PlatformEGL::isExtraContextSupported() const noexcept { + return ext.egl.KHR_no_config_context; +} + +void PlatformEGL::createContext(bool shared) { + EGLContext context = eglCreateContext(mEGLDisplay, EGL_NO_CONFIG_KHR, + shared ? mEGLContext : EGL_NO_CONTEXT, mContextAttribs.data()); + + if (UTILS_UNLIKELY(context == EGL_NO_CONTEXT)) { + // eglCreateContext failed + logEglError("eglCreateContext"); + } + + assert_invariant(context != EGL_NO_CONTEXT); + + eglMakeCurrent(mEGLDisplay, EGL_NO_SURFACE, EGL_NO_SURFACE, context); + + mAdditionalContexts.push_back(context); +} + EGLBoolean PlatformEGL::makeCurrent(EGLSurface drawSurface, EGLSurface readSurface) noexcept { if (UTILS_UNLIKELY((drawSurface != mCurrentDrawSurface || readSurface != mCurrentReadSurface))) { mCurrentDrawSurface = drawSurface; @@ -264,6 +287,9 @@ void PlatformEGL::terminate() noexcept { eglMakeCurrent(mEGLDisplay, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT); eglDestroySurface(mEGLDisplay, mEGLDummySurface); eglDestroyContext(mEGLDisplay, mEGLContext); + for (auto context : mAdditionalContexts) { + eglDestroyContext(mEGLDisplay, context); + } eglTerminate(mEGLDisplay); eglReleaseThread(); } @@ -441,7 +467,7 @@ FenceStatus PlatformEGL::waitFence( #ifdef EGL_KHR_reusable_sync EGLSyncKHR sync = (EGLSyncKHR) fence; if (sync != EGL_NO_SYNC_KHR) { - EGLint status = eglClientWaitSyncKHR(mEGLDisplay, sync, 0, (EGLTimeKHR)timeout); + EGLint const status = eglClientWaitSyncKHR(mEGLDisplay, sync, 0, (EGLTimeKHR)timeout); if (status == EGL_CONDITION_SATISFIED_KHR) { return FenceStatus::CONDITION_SATISFIED; } diff --git a/filament/backend/src/opengl/platforms/PlatformWGL.cpp b/filament/backend/src/opengl/platforms/PlatformWGL.cpp index 4ff4a4fc0a..f2943ee091 100644 --- a/filament/backend/src/opengl/platforms/PlatformWGL.cpp +++ b/filament/backend/src/opengl/platforms/PlatformWGL.cpp @@ -74,10 +74,11 @@ struct WGLSwapChain { bool isHeadless = false; }; +static PFNWGLCREATECONTEXTATTRIBSARBPROC wglCreateContextAttribs = nullptr; + Driver* PlatformWGL::createDriver(void* const sharedGLContext, const Platform::DriverConfig& driverConfig) noexcept { int result = 0; - PFNWGLCREATECONTEXTATTRIBSARBPROC wglCreateContextAttribs = nullptr; int pixelFormat = 0; mPfd = { @@ -123,13 +124,15 @@ Driver* PlatformWGL::createDriver(void* const sharedGLContext, (PFNWGLCREATECONTEXTATTRIBSARBPROC) wglGetProcAddress("wglCreateContextAttribsARB"); // try all versions down, from GL 4.5 to 4.1 + + for (int minor = 5; minor >= 1; minor--) { - int const attribs[] = { + mAttribs = { WGL_CONTEXT_MAJOR_VERSION_ARB, 4, WGL_CONTEXT_MINOR_VERSION_ARB, minor, 0 }; - mContext = wglCreateContextAttribs(whdc, (HGLRC)sharedGLContext, attribs); + mContext = wglCreateContextAttribs(whdc, (HGLRC)sharedGLContext, mAttribs.data()); if (mContext) { break; } @@ -164,12 +167,25 @@ error: return NULL; } +bool PlatformWGL::isExtraContextSupported() const noexcept { + return true; +} + +void PlatformWGL::createContext(bool shared) { + HGLRC context = wglCreateContextAttribs(mWhdc, shared ? mContext : nullptr, mAttribs.data()); + wglMakeCurrent(mWhdc, context); + mAdditionalContexts.push_back(context); +} + void PlatformWGL::terminate() noexcept { wglMakeCurrent(NULL, NULL); if (mContext) { wglDeleteContext(mContext); mContext = NULL; } + for (auto& context : mAdditionalContexts) { + wglDeleteContext(mContext); + } if (mHWnd && mWhdc) { ReleaseDC(mHWnd, mWhdc); DestroyWindow(mHWnd); diff --git a/filament/backend/src/vulkan/VulkanDriver.cpp b/filament/backend/src/vulkan/VulkanDriver.cpp index 132609672c..b37a15117f 100644 --- a/filament/backend/src/vulkan/VulkanDriver.cpp +++ b/filament/backend/src/vulkan/VulkanDriver.cpp @@ -980,7 +980,9 @@ void VulkanDriver::updateSamplerGroup(Handle sbh, void VulkanDriver::compilePrograms(CallbackHandler* handler, CallbackHandler::Callback callback, void* user) { - scheduleCallback(handler, user, callback); + if (callback) { + scheduleCallback(handler, user, callback); + } } void VulkanDriver::beginRenderPass(Handle rth, const RenderPassParams& params) { diff --git a/filament/include/filament/Material.h b/filament/include/filament/Material.h index 1262059ca0..5b9314a042 100644 --- a/filament/include/filament/Material.h +++ b/filament/include/filament/Material.h @@ -152,11 +152,13 @@ public: friend class FMaterial; }; + using CompilerPriorityQueue = backend:: CompilerPriorityQueue; + /** * Asynchronously ensures that a subset of this Material's variants are compiled. After issuing * several Material::compile() calls in a row, it is recommended to call Engine::flush() * such that the backend can start the compilation work as soon as possible. - * The provided callback is guaranteed to be called on the main thread after all unfiltered + * The provided callback is guaranteed to be called on the main thread after all specified * variants of the material are compiled. This can take hundreds of milliseconds. * * If all the material's variants are already compiled, the callback will be scheduled as @@ -164,14 +166,31 @@ public: * many previous frames are enqueued in the backend. This also varies by backend. Therefore, * it is recommended to only call this method once per material shortly after creation. * - * @param handler Handler to dispatch the callback or nullptr for the default handler - * @param callback callback called on the main thread when the compilation is done on - * by backend. - * @param variantFilter Variants to filter-out from the compile command. + * @param priority Which priority queue to use, LOW or HIGH. + * @param variants Variants to include to the compile command. + * @param handler Handler to dispatch the callback or nullptr for the default handler + * @param callback callback called on the main thread when the compilation is done on + * by backend. */ - void compile(backend::CallbackHandler* handler, - utils::Invocable&& callback, - UserVariantFilterMask variantFilter) noexcept; + void compile(CompilerPriorityQueue priority, + UserVariantFilterMask variants, + backend::CallbackHandler* handler = nullptr, + utils::Invocable&& callback = {}) noexcept; + + inline void compile(CompilerPriorityQueue priority, + UserVariantFilterBit variants, + backend::CallbackHandler* handler = nullptr, + utils::Invocable&& callback = {}) noexcept { + compile(priority, UserVariantFilterMask(variants), handler, + std::forward>(callback)); + } + + inline void compile(CompilerPriorityQueue priority, + backend::CallbackHandler* handler = nullptr, + utils::Invocable&& callback = {}) noexcept { + compile(priority, UserVariantFilterBit::ALL, handler, + std::forward>(callback)); + } /** * Creates a new instance of this material. Material instances should be freed using diff --git a/filament/src/Material.cpp b/filament/src/Material.cpp index 19568947e9..fc4e198f4d 100644 --- a/filament/src/Material.cpp +++ b/filament/src/Material.cpp @@ -136,10 +136,10 @@ MaterialInstance const* Material::getDefaultInstance() const noexcept { return downcast(this)->getDefaultInstance(); } -void Material::compile(backend::CallbackHandler* handler, - utils::Invocable&& callback, - UserVariantFilterMask variantFilter) noexcept { - downcast(this)->compile(handler, std::move(callback), variantFilter); +void Material::compile(CompilerPriorityQueue priority, UserVariantFilterMask variantFilter, + backend::CallbackHandler* handler, utils::Invocable&& callback) noexcept { + downcast(this)->compile(priority, variantFilter, handler, std::move(callback)); } + } // namespace filament diff --git a/filament/src/MaterialParser.h b/filament/src/MaterialParser.h index 74dd59294d..61c8932249 100644 --- a/filament/src/MaterialParser.h +++ b/filament/src/MaterialParser.h @@ -117,7 +117,14 @@ public: bool getShader(filaflat::ShaderContent& shader, backend::ShaderModel shaderModel, Variant variant, backend::ShaderStage stage) noexcept; - filaflat::MaterialChunk const& getMaterialChunk() const noexcept { return mImpl.mMaterialChunk; } + bool hasShader(backend::ShaderModel model, + Variant variant, backend::ShaderStage stage) const noexcept { + return getMaterialChunk().hasShader(model, variant, stage); + } + + filaflat::MaterialChunk const& getMaterialChunk() const noexcept { + return mImpl.mMaterialChunk; + } private: struct MaterialParserDetails { diff --git a/filament/src/PostProcessManager.cpp b/filament/src/PostProcessManager.cpp index 92dba087e3..f2664d4502 100644 --- a/filament/src/PostProcessManager.cpp +++ b/filament/src/PostProcessManager.cpp @@ -2213,8 +2213,10 @@ void PostProcessManager::colorGradingPrepareSubpass(DriverApi& driver, mi->commit(driver); // load both variants - material.getMaterial(mEngine)->prepareProgram(Variant{Variant::type_t(PostProcessVariant::TRANSLUCENT)}); - material.getMaterial(mEngine)->prepareProgram(Variant{Variant::type_t(PostProcessVariant::OPAQUE)}); + material.getMaterial(mEngine)->prepareProgram( + Variant{ Variant::type_t(PostProcessVariant::TRANSLUCENT) }); + material.getMaterial(mEngine)->prepareProgram( + Variant{ Variant::type_t(PostProcessVariant::OPAQUE) }); } void PostProcessManager::colorGradingSubpass(DriverApi& driver, diff --git a/filament/src/details/Material.cpp b/filament/src/details/Material.cpp index 98b8b655d0..8b9b67412a 100644 --- a/filament/src/details/Material.cpp +++ b/filament/src/details/Material.cpp @@ -403,7 +403,7 @@ FMaterial::FMaterial(FEngine& engine, const Material::Builder& builder) filaflat::MaterialChunk const& materialChunk{ mMaterialParser->getMaterialChunk() }; auto variants = FixedCapacityVector::with_capacity(materialChunk.getShaderCount()); materialChunk.visitShaders([&variants]( - ShaderModel model, Variant variant, ShaderStage stage) { + ShaderModel, Variant variant, ShaderStage) { if (Variant::isValidDepthVariant(variant)) { variants.push_back(variant); } @@ -418,7 +418,7 @@ FMaterial::FMaterial(FEngine& engine, const Material::Builder& builder) if (UTILS_UNLIKELY(!mIsDefaultMaterial && !mHasCustomDepthShader)) { FMaterial const* const pDefaultMaterial = engine.getDefaultMaterial(); auto& cachedPrograms = mCachedPrograms; - for (Variant variant : pDefaultMaterial->mDepthVariants) { + for (Variant const variant : pDefaultMaterial->mDepthVariants) { pDefaultMaterial->prepareProgram(variant); cachedPrograms[variant.key] = pDefaultMaterial->getProgram(variant); } @@ -467,31 +467,39 @@ void FMaterial::terminate(FEngine& engine) { mDefaultInstance.terminate(engine); } -void FMaterial::compile(backend::CallbackHandler* handler, - utils::Invocable&& callback, - UserVariantFilterMask variantFilter) noexcept { +void FMaterial::compile(CompilerPriorityQueue priority, + UserVariantFilterMask variantSpec, + backend::CallbackHandler* handler, + utils::Invocable&& callback) noexcept { + + UserVariantFilterMask const variantFilter = + ~variantSpec & UserVariantFilterMask(UserVariantFilterBit::ALL); + auto const& variants = isVariantLit() ? VariantUtils::getLitVariants() : VariantUtils::getUnlitVariants(); for (auto const variant : variants) { if (!variantFilter || variant == Variant::filterUserVariant(variant, variantFilter)) { if (hasVariant(variant)) { - prepareProgram(variant); + prepareProgram(variant, priority); } } } - struct Callback { - Invocable f; - Material* m; - static void func(void* user) { - auto* const c = reinterpret_cast(user); - c->f(c->m); - delete c; - } - }; - - auto* const user = new Callback{ std::move(callback), this }; - mEngine.getDriverApi().compilePrograms(handler, &Callback::func, static_cast(user)); + if (callback) { + struct Callback { + Invocable f; + Material* m; + static void func(void* user) { + auto* const c = reinterpret_cast(user); + c->f(c->m); + delete c; + } + }; + auto* const user = new(std::nothrow) Callback{ std::move(callback), this }; + mEngine.getDriverApi().compilePrograms(handler, &Callback::func, static_cast(user)); + } else { + mEngine.getDriverApi().compilePrograms(nullptr, nullptr, nullptr); + } } FMaterialInstance* FMaterial::createInstance(const char* name) const noexcept { @@ -527,25 +535,25 @@ bool FMaterial::hasVariant(Variant variant) const noexcept { // TODO: implement MaterialDomain::COMPUTE return false; } - ShaderContent& vsBuilder = mEngine.getVertexShaderContent(); const ShaderModel sm = mEngine.getShaderModel(); - if (!mMaterialParser->getShader(vsBuilder, sm, vertexVariant, ShaderStage::VERTEX)) { + if (!mMaterialParser->hasShader(sm, vertexVariant, ShaderStage::VERTEX)) { return false; } - if (!mMaterialParser->getShader(vsBuilder, sm, fragmentVariant, ShaderStage::FRAGMENT)) { + if (!mMaterialParser->hasShader(sm, fragmentVariant, ShaderStage::FRAGMENT)) { return false; } return true; } -void FMaterial::prepareProgramSlow(Variant variant) const noexcept { +void FMaterial::prepareProgramSlow(Variant variant, + backend::CompilerPriorityQueue priorityQueue) const noexcept { assert_invariant(mEngine.hasFeatureLevel(mFeatureLevel)); switch (getMaterialDomain()) { case MaterialDomain::SURFACE: - getSurfaceProgramSlow(variant); + getSurfaceProgramSlow(variant, priorityQueue); break; case MaterialDomain::POST_PROCESS: - getPostProcessProgramSlow(variant); + getPostProcessProgramSlow(variant, priorityQueue); break; case MaterialDomain::COMPUTE: // TODO: implement MaterialDomain::COMPUTE @@ -553,7 +561,8 @@ void FMaterial::prepareProgramSlow(Variant variant) const noexcept { } } -void FMaterial::getSurfaceProgramSlow(Variant variant) const noexcept { +void FMaterial::getSurfaceProgramSlow(Variant variant, + CompilerPriorityQueue priorityQueue) const noexcept { // filterVariant() has already been applied in generateCommands(), shouldn't be needed here // if we're unlit, we don't have any bits that correspond to lit materials assert_invariant(variant == Variant::filterVariant(variant, isVariantLit()) ); @@ -564,11 +573,14 @@ void FMaterial::getSurfaceProgramSlow(Variant variant) const noexcept { Variant const fragmentVariant = Variant::filterVariantFragment(variant); Program pb{ getProgramWithVariants(variant, vertexVariant, fragmentVariant) }; + pb.priorityQueue(priorityQueue); createAndCacheProgram(std::move(pb), variant); } -void FMaterial::getPostProcessProgramSlow(Variant variant) const noexcept { +void FMaterial::getPostProcessProgramSlow(Variant variant, + CompilerPriorityQueue priorityQueue) const noexcept { Program pb{ getProgramWithVariants(variant, variant, variant) }; + pb.priorityQueue(priorityQueue); createAndCacheProgram(std::move(pb), variant); } diff --git a/filament/src/details/Material.h b/filament/src/details/Material.h index cf1bf9fe84..a537224906 100644 --- a/filament/src/details/Material.h +++ b/filament/src/details/Material.h @@ -63,9 +63,10 @@ public: return mSamplerInterfaceBlock; } - void compile(backend::CallbackHandler* handler, - utils::Invocable&& callback, - UserVariantFilterMask variantFilter) noexcept; + void compile(CompilerPriorityQueue priority, + UserVariantFilterMask variantFilter, + backend::CallbackHandler* handler, + utils::Invocable&& callback) noexcept; // Create an instance of this material FMaterialInstance* createInstance(const char* name) const noexcept; @@ -88,10 +89,11 @@ public: // prepareProgram creates the program for the material's given variant at the backend level. // Must be called outside of backend render pass. // Must be called before getProgram() below. - void prepareProgram(Variant variant) const noexcept { + void prepareProgram(Variant variant, + backend::CompilerPriorityQueue priorityQueue = CompilerPriorityQueue::HIGH) const noexcept { // prepareProgram() is called for each RenderPrimitive in the scene, so it must be efficient. if (UTILS_UNLIKELY(!isCached(variant))) { - prepareProgramSlow(variant); + prepareProgramSlow(variant, priorityQueue); } } @@ -191,15 +193,16 @@ public: private: bool hasVariant(Variant variant) const noexcept; - void prepareProgramSlow(Variant variant) const noexcept; - void getSurfaceProgramSlow(Variant variant) const noexcept; - void getPostProcessProgramSlow(Variant variant) const noexcept; - backend::Program getProgramWithVariants(Variant variant, Variant vertexVariant, - Variant fragmentVariant) const noexcept; + void prepareProgramSlow(Variant variant, + CompilerPriorityQueue priorityQueue) const noexcept; + void getSurfaceProgramSlow(Variant variant, + CompilerPriorityQueue priorityQueue) const noexcept; + void getPostProcessProgramSlow(Variant variant, + CompilerPriorityQueue priorityQueue) const noexcept; + backend::Program getProgramWithVariants(Variant variant, + Variant vertexVariant, Variant fragmentVariant) const noexcept; - - void createAndCacheProgram(backend::Program&& p, - Variant variant) const noexcept; + void createAndCacheProgram(backend::Program&& p, Variant variant) const noexcept; // try to order by frequency of use mutable std::array, VARIANT_COUNT> mCachedPrograms; diff --git a/filament/src/details/Renderer.cpp b/filament/src/details/Renderer.cpp index 89f2b0b0f4..ec09b44bc4 100644 --- a/filament/src/details/Renderer.cpp +++ b/filament/src/details/Renderer.cpp @@ -227,7 +227,8 @@ bool FRenderer::beginFrame(FSwapChain* swapChain, uint64_t vsyncSteadyClockTimeN // NOTE: this makes synchronous calls to the driver driver.updateStreams(&driver); - // gives the backend a chance to execute periodic tasks + // Gives the backend a chance to execute periodic tasks. This must be called before + // the frame skipper. driver.tick(); /* diff --git a/libs/filabridge/include/filament/MaterialEnums.h b/libs/filabridge/include/filament/MaterialEnums.h index 7dfb48dd45..f87a8ef6bc 100644 --- a/libs/filabridge/include/filament/MaterialEnums.h +++ b/libs/filabridge/include/filament/MaterialEnums.h @@ -20,6 +20,7 @@ #define TNT_FILAMENT_MATERIAL_ENUM_H #include +#include #include #include @@ -232,7 +233,9 @@ enum class Property : uint8_t { // when adding new Properties, make sure to update MATERIAL_PROPERTIES_COUNT }; -enum class UserVariantFilterBit : uint32_t { +using UserVariantFilterMask = uint32_t; + +enum class UserVariantFilterBit : UserVariantFilterMask { DIRECTIONAL_LIGHTING = 0x01, DYNAMIC_LIGHTING = 0x02, SHADOW_RECEIVER = 0x04, @@ -240,10 +243,12 @@ enum class UserVariantFilterBit : uint32_t { FOG = 0x10, VSM = 0x20, SSR = 0x40, + ALL = 0x7F, }; -using UserVariantFilterMask = uint32_t; - } // namespace filament +template<> struct utils::EnableBitMaskOperators + : public std::true_type {}; + #endif diff --git a/libs/filaflat/include/filaflat/MaterialChunk.h b/libs/filaflat/include/filaflat/MaterialChunk.h index aa15fa5b2f..2b7ceda7fa 100644 --- a/libs/filaflat/include/filaflat/MaterialChunk.h +++ b/libs/filaflat/include/filaflat/MaterialChunk.h @@ -54,6 +54,8 @@ public: void visitShaders(utils::Invocable&& visitor) const; + bool hasShader(ShaderModel model, Variant variant, ShaderStage stage) const noexcept; + // These methods are for debugging purposes only (matdbg) // @{ static void decodeKey(uint32_t key, diff --git a/libs/filaflat/src/MaterialChunk.cpp b/libs/filaflat/src/MaterialChunk.cpp index 2d73970f5c..2cedfbf117 100644 --- a/libs/filaflat/src/MaterialChunk.cpp +++ b/libs/filaflat/src/MaterialChunk.cpp @@ -170,6 +170,14 @@ bool MaterialChunk::getSpirvShader(BlobDictionary const& dictionary, return true; } +bool MaterialChunk::hasShader(ShaderModel model, Variant variant, ShaderStage stage) const noexcept { + if (mBase == nullptr) { + return false; + } + auto pos = mOffsets.find(makeKey(model, variant, stage)); + return pos != mOffsets.end(); +} + bool MaterialChunk::getShader(ShaderContent& shaderContent, BlobDictionary const& dictionary, ShaderModel shaderModel, filament::Variant variant, ShaderStage stage) { switch (mMaterialTag) { diff --git a/libs/gltfio/src/ArchiveCache.cpp b/libs/gltfio/src/ArchiveCache.cpp index 152b0fc94c..07a08e428b 100644 --- a/libs/gltfio/src/ArchiveCache.cpp +++ b/libs/gltfio/src/ArchiveCache.cpp @@ -119,6 +119,16 @@ Material* ArchiveCache::getMaterial(const ArchiveRequirements& reqs) { .package(spec.package, spec.packageByteCount) .build(mEngine); } + + // compile everything at low priority + mMaterials[i]->compile(Material::CompilerPriorityQueue::LOW); + + // promote variants we care about to high priority + mMaterials[i]->compile(Material::CompilerPriorityQueue::HIGH, + UserVariantFilterBit::DIRECTIONAL_LIGHTING | + UserVariantFilterBit::DYNAMIC_LIGHTING | + UserVariantFilterBit::SHADOW_RECEIVER); + return mMaterials[i]; } } diff --git a/libs/utils/include/utils/BitmaskEnum.h b/libs/utils/include/utils/BitmaskEnum.h index 1efa394195..56354956c5 100644 --- a/libs/utils/include/utils/BitmaskEnum.h +++ b/libs/utils/include/utils/BitmaskEnum.h @@ -141,5 +141,4 @@ inline constexpr bool any(Enum lhs) noexcept { return !none(lhs); } - #endif // TNT_UTILS_BITMASKENUM_H