common: string_util: Use std::string_view for UTF16ToUTF8/UTF8ToUTF16W.

Merge pull request #9966 from bunnei/bounded-polyfill
common: bounded_threadsafe_queue: Use polyfill_thread.
2023-03-18 22:42:25 -07:00 · 2023-03-18 12:39:52 -04:00 · 2023-03-17 23:42:17 -07:00 · 2023-03-17 22:14:29 -07:00 · 2023-03-17 22:13:57 -07:00 · 2023-03-15 20:19:45 -04:00
8 changed files with 90 additions and 48 deletions
--- a/src/common/bounded_threadsafe_queue.h
+++ b/src/common/bounded_threadsafe_queue.h
@@ -9,10 +9,11 @@
 #include <memory>
 #include <mutex>
 #include <new>
-#include <stop_token>
 #include <type_traits>
 #include <utility>

+#include "common/polyfill_thread.h"
+
 namespace Common {

 #if defined(__cpp_lib_hardware_interference_size)
@@ -78,7 +79,7 @@ public:
        auto& slot = slots[idx(tail)];
        if (!slot.turn.test()) {
            std::unique_lock lock{cv_mutex};
-            cv.wait(lock, stop, [&slot] { return slot.turn.test(); });
+            Common::CondvarWait(cv, lock, stop, [&slot] { return slot.turn.test(); });
        }
        v = slot.move();
        slot.destroy();
--- a/src/common/string_util.cpp
+++ b/src/common/string_util.cpp
@@ -125,18 +125,18 @@ std::string ReplaceAll(std::string result, const std::string& src, const std::st
    return result;
 }

-std::string UTF16ToUTF8(const std::u16string& input) {
+std::string UTF16ToUTF8(std::u16string_view input) {
    std::wstring_convert<std::codecvt_utf8_utf16<char16_t>, char16_t> convert;
-    return convert.to_bytes(input);
+    return convert.to_bytes(input.data(), input.data() + input.size());
 }

-std::u16string UTF8ToUTF16(const std::string& input) {
+std::u16string UTF8ToUTF16(std::string_view input) {
    std::wstring_convert<std::codecvt_utf8_utf16<char16_t>, char16_t> convert;
-    return convert.from_bytes(input);
+    return convert.from_bytes(input.data(), input.data() + input.size());
 }

 #ifdef _WIN32
-static std::wstring CPToUTF16(u32 code_page, const std::string& input) {
+static std::wstring CPToUTF16(u32 code_page, std::string_view input) {
    const auto size =
        MultiByteToWideChar(code_page, 0, input.data(), static_cast<int>(input.size()), nullptr, 0);

@@ -154,7 +154,7 @@ static std::wstring CPToUTF16(u32 code_page, const std::string& input) {
    return output;
 }

-std::string UTF16ToUTF8(const std::wstring& input) {
+std::string UTF16ToUTF8(std::wstring_view input) {
    const auto size = WideCharToMultiByte(CP_UTF8, 0, input.data(), static_cast<int>(input.size()),
                                          nullptr, 0, nullptr, nullptr);
    if (size == 0) {
@@ -172,7 +172,7 @@ std::string UTF16ToUTF8(const std::wstring& input) {
    return output;
 }

-std::wstring UTF8ToUTF16W(const std::string& input) {
+std::wstring UTF8ToUTF16W(std::string_view input) {
    return CPToUTF16(CP_UTF8, input);
 }

--- a/src/common/string_util.h
+++ b/src/common/string_util.h
@@ -36,12 +36,12 @@ bool SplitPath(const std::string& full_path, std::string* _pPath, std::string* _
 [[nodiscard]] std::string ReplaceAll(std::string result, const std::string& src,
                                     const std::string& dest);

-[[nodiscard]] std::string UTF16ToUTF8(const std::u16string& input);
-[[nodiscard]] std::u16string UTF8ToUTF16(const std::string& input);
+[[nodiscard]] std::string UTF16ToUTF8(std::u16string_view input);
+[[nodiscard]] std::u16string UTF8ToUTF16(std::string_view input);

 #ifdef _WIN32
-[[nodiscard]] std::string UTF16ToUTF8(const std::wstring& input);
-[[nodiscard]] std::wstring UTF8ToUTF16W(const std::string& str);
+[[nodiscard]] std::string UTF16ToUTF8(std::wstring_view input);
+[[nodiscard]] std::wstring UTF8ToUTF16W(std::string_view str);

 #endif

--- a/src/video_core/gpu_thread.cpp
+++ b/src/video_core/gpu_thread.cpp
@@ -32,7 +32,8 @@ static void RunThread(std::stop_token stop_token, Core::System& system,
    VideoCore::RasterizerInterface* const rasterizer = renderer.ReadRasterizer();

    while (!stop_token.stop_requested()) {
-        CommandDataContainer next = state.queue.PopWait(stop_token);
+        CommandDataContainer next;
+        state.queue.Pop(next, stop_token);
        if (stop_token.stop_requested()) {
            break;
        }
--- a/src/video_core/gpu_thread.h
+++ b/src/video_core/gpu_thread.h
@@ -10,8 +10,8 @@
 #include <thread>
 #include <variant>

+#include "common/bounded_threadsafe_queue.h"
 #include "common/polyfill_thread.h"
-#include "common/threadsafe_queue.h"
 #include "video_core/framebuffer_config.h"

 namespace Tegra {
@@ -97,7 +97,7 @@ struct CommandDataContainer {

 /// Struct used to synchronize the GPU thread
 struct SynchState final {
-    using CommandQueue = Common::MPSCQueue<CommandDataContainer, true>;
+    using CommandQueue = Common::MPSCQueue<CommandDataContainer>;
    std::mutex write_lock;
    CommandQueue queue;
    u64 last_fence{};
--- a/src/video_core/renderer_vulkan/vk_scheduler.cpp
+++ b/src/video_core/renderer_vulkan/vk_scheduler.cpp
@@ -47,14 +47,15 @@ Scheduler::Scheduler(const Device& device_, StateTracker& state_tracker_)
 Scheduler::~Scheduler() = default;

 void Scheduler::Flush(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore) {
+    // When flushing, we only send data to the worker thread; no waiting is necessary.
    SubmitExecution(signal_semaphore, wait_semaphore);
    AllocateNewContext();
 }

 void Scheduler::Finish(VkSemaphore signal_semaphore, VkSemaphore wait_semaphore) {
+    // When finishing, we need to wait for the submission to have executed on the device.
    const u64 presubmit_tick = CurrentTick();
    SubmitExecution(signal_semaphore, wait_semaphore);
-    WaitWorker();
    Wait(presubmit_tick);
    AllocateNewContext();
 }
@@ -63,8 +64,13 @@ void Scheduler::WaitWorker() {
    MICROPROFILE_SCOPE(Vulkan_WaitForWorker);
    DispatchWork();

-    std::unique_lock lock{work_mutex};
-    wait_cv.wait(lock, [this] { return work_queue.empty(); });
+    // Ensure the queue is drained.
+    std::unique_lock ql{queue_mutex};
+    event_cv.wait(ql, [this] { return work_queue.empty(); });
+
+    // Now wait for execution to finish.
+    // This needs to be done in the same order as WorkerThread.
+    std::unique_lock el{execution_mutex};
 }

 void Scheduler::DispatchWork() {
@@ -72,10 +78,10 @@ void Scheduler::DispatchWork() {
        return;
    }
    {
-        std::scoped_lock lock{work_mutex};
+        std::scoped_lock ql{queue_mutex};
        work_queue.push(std::move(chunk));
    }
-    work_cv.notify_one();
+    event_cv.notify_all();
    AcquireNewChunk();
 }

@@ -137,30 +143,55 @@ bool Scheduler::UpdateRescaling(bool is_rescaling) {

 void Scheduler::WorkerThread(std::stop_token stop_token) {
    Common::SetCurrentThreadName("VulkanWorker");
-    do {
-        std::unique_ptr<CommandChunk> work;
-        bool has_submit{false};
-        {
-            std::unique_lock lock{work_mutex};
-            if (work_queue.empty()) {
-                wait_cv.notify_all();
-            }
-            Common::CondvarWait(work_cv, lock, stop_token, [&] { return !work_queue.empty(); });
-            if (stop_token.stop_requested()) {
-                continue;
-            }
-            work = std::move(work_queue.front());
-            work_queue.pop();

-            has_submit = work->HasSubmit();
+    const auto TryPopQueue{[this](auto& work) -> bool {
+        if (work_queue.empty()) {
+            return false;
+        }
+
+        work = std::move(work_queue.front());
+        work_queue.pop();
+        event_cv.notify_all();
+        return true;
+    }};
+
+    while (!stop_token.stop_requested()) {
+        std::unique_ptr<CommandChunk> work;
+
+        {
+            std::unique_lock lk{queue_mutex};
+
+            // Wait for work.
+            Common::CondvarWait(event_cv, lk, stop_token, [&] { return TryPopQueue(work); });
+
+            // If we've been asked to stop, we're done.
+            if (stop_token.stop_requested()) {
+                return;
+            }
+
+            // Exchange lock ownership so that we take the execution lock before
+            // the queue lock goes out of scope. This allows us to force execution
+            // to complete in the next step.
+            std::exchange(lk, std::unique_lock{execution_mutex});
+
+            // Perform the work, tracking whether the chunk was a submission
+            // before executing.
+            const bool has_submit = work->HasSubmit();
            work->ExecuteAll(current_cmdbuf);
+
+            // If the chunk was a submission, reallocate the command buffer.
+            if (has_submit) {
+                AllocateWorkerCommandBuffer();
+            }
        }
-        if (has_submit) {
-            AllocateWorkerCommandBuffer();
+
+        {
+            std::scoped_lock rl{reserve_mutex};
+
+            // Recycle the chunk back to the reserve.
+            chunk_reserve.emplace_back(std::move(work));
        }
-        std::scoped_lock reserve_lock{reserve_mutex};
-        chunk_reserve.push_back(std::move(work));
-    } while (!stop_token.stop_requested());
+    }
 }

 void Scheduler::AllocateWorkerCommandBuffer() {
@@ -289,13 +320,16 @@ void Scheduler::EndRenderPass() {
 }

 void Scheduler::AcquireNewChunk() {
-    std::scoped_lock lock{reserve_mutex};
+    std::scoped_lock rl{reserve_mutex};
+
    if (chunk_reserve.empty()) {
+        // If we don't have anything reserved, we need to make a new chunk.
        chunk = std::make_unique<CommandChunk>();
-        return;
+    } else {
+        // Otherwise, we can just take from the reserve.
+        chunk = std::make_unique<CommandChunk>();
+        chunk_reserve.pop_back();
    }
-    chunk = std::move(chunk_reserve.back());
-    chunk_reserve.pop_back();
 }

 } // namespace Vulkan
--- a/src/video_core/renderer_vulkan/vk_scheduler.h
+++ b/src/video_core/renderer_vulkan/vk_scheduler.h
@@ -232,10 +232,10 @@ private:

    std::queue<std::unique_ptr<CommandChunk>> work_queue;
    std::vector<std::unique_ptr<CommandChunk>> chunk_reserve;
+    std::mutex execution_mutex;
    std::mutex reserve_mutex;
-    std::mutex work_mutex;
-    std::condition_variable_any work_cv;
-    std::condition_variable wait_cv;
+    std::mutex queue_mutex;
+    std::condition_variable_any event_cv;
    std::jthread worker_thread;
 };

--- a/src/video_core/vulkan_common/vulkan_device.cpp
+++ b/src/video_core/vulkan_common/vulkan_device.cpp
@@ -401,6 +401,12 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
            loaded_extensions.erase(VK_EXT_EXTENDED_DYNAMIC_STATE_2_EXTENSION_NAME);
        }
    }
+    if (extensions.extended_dynamic_state3 && is_radv) {
+        LOG_WARNING(Render_Vulkan, "RADV has broken extendedDynamicState3ColorBlendEquation");
+        features.extended_dynamic_state3.extendedDynamicState3ColorBlendEnable = false;
+        features.extended_dynamic_state3.extendedDynamicState3ColorBlendEquation = false;
+        dynamic_state3_blending = false;
+    }
    if (extensions.vertex_input_dynamic_state && is_radv) {
        // TODO(ameerj): Blacklist only offending driver versions
        // TODO(ameerj): Confirm if RDNA1 is affected
Author	SHA1	Message	Date
bunnei	00d401d639	common: string_util: Use std::string_view for UTF16ToUTF8/UTF8ToUTF16W.	2023-03-18 22:42:25 -07:00
liamwhite	0e7e98e24e	Merge pull request #9966 from bunnei/bounded-polyfill common: bounded_threadsafe_queue: Use polyfill_thread.	2023-03-18 12:39:52 -04:00
bunnei	0eb3fa05e5	common: bounded_threadsafe_queue: Use polyfill_thread.	2023-03-17 23:42:17 -07:00
bunnei	889454f9bf	Merge pull request #9778 from behunin/my-box-chevy gpu_thread: Use bounded queue	2023-03-17 22:14:29 -07:00
bunnei	8bcaa8c2e4	Merge pull request #9953 from german77/amiibo_crc service: nfp: Actually write correct crc	2023-03-17 22:13:57 -07:00
liamwhite	6d76a54d37	Merge pull request #9955 from liamwhite/color-blend-equation vulkan: disable extendedDynamicState3ColorBlendEquation on radv	2023-03-15 20:19:45 -04:00
liamwhite	a04061e6ae	Merge pull request #9931 from liamwhite/sched vk_scheduler: split work queue waits and execution waits	2023-03-15 20:19:35 -04:00
Liam	da83afdeaf	vulkan: disable extendedDynamicState3ColorBlendEquation on radv	2023-03-15 15:55:07 -04:00
Liam	3f261f22c9	vk_scheduler: split work queue waits and execution waits	2023-03-12 17:19:44 -04:00
Behunin	44518b225c	gpu_thread: Use bounded queue	2023-03-03 18:20:56 -07:00