nvdec cleanup

author: ameerj 2020-11-23 13:25:01 -0500
committer: ameerj 2021-02-13 13:07:31 -0500
commit: ac265a72ce4176ceb3cd10a5548ab71519771640 (patch)
tree: 0acde029388d465a5801db9106dd8f4e026e57e8 /src/video_core/command_classes
parent: Merge pull request #5919 from ReinUsesLisp/stream-buffer-tragic (diff)
download: yuzu-ac265a72ce4176ceb3cd10a5548ab71519771640.tar.gz
yuzu-ac265a72ce4176ceb3cd10a5548ab71519771640.tar.xz
yuzu-ac265a72ce4176ceb3cd10a5548ab71519771640.zip
3 files changed, 24 insertions, 11 deletions
diff --git a/src/video_core/command_classes/codecs/codec.cpp b/src/video_core/command_classes/codecs/codec.cpp
index 39bc923a5..d02dc6260 100644
--- a/src/video_core/command_classes/codecs/codec.cpp
+++ b/src/video_core/command_classes/codecs/codec.cpp
@@ -44,8 +44,10 @@ Codec::~Codec() {
 }
 void Codec::SetTargetCodec(NvdecCommon::VideoCodec codec) {
-    LOG_INFO(Service_NVDRV, "NVDEC video codec initialized to {}", codec);
+    if (current_codec != codec) {
-    current_codec = codec;
+        LOG_INFO(Service_NVDRV, "NVDEC video codec initialized to {}", static_cast<u32>(codec));
+        current_codec = codec;
+    }
 }
 void Codec::StateWrite(u32 offset, u64 arguments) {
@@ -55,7 +57,6 @@ void Codec::StateWrite(u32 offset, u64 arguments) {
 void Codec::Decode() {
    bool is_first_frame = false;
    if (!initialized) {
        if (current_codec == NvdecCommon::VideoCodec::H264) {
            av_codec = avcodec_find_decoder(AV_CODEC_ID_H264);
diff --git a/src/video_core/command_classes/vic.cpp b/src/video_core/command_classes/vic.cpp
index 2b7569335..73680d057 100644
--- a/src/video_core/command_classes/vic.cpp
+++ b/src/video_core/command_classes/vic.cpp
@@ -18,7 +18,10 @@ extern "C" {
 namespace Tegra {
 Vic::Vic(GPU& gpu_, std::shared_ptr<Nvdec> nvdec_processor_)
-    : gpu(gpu_), nvdec_processor(std::move(nvdec_processor_)) {}
+    : gpu(gpu_),
+      nvdec_processor(std::move(nvdec_processor_)), converted_frame_buffer{nullptr, av_free}
+{}
 Vic::~Vic() = default;
 void Vic::VicStateWrite(u32 offset, u32 arguments) {
@@ -89,8 +92,10 @@ void Vic::Execute() {
        // Get Converted frame
        const std::size_t linear_size = frame->width * frame->height * 4;
-        using AVMallocPtr = std::unique_ptr<u8, decltype(&av_free)>;
+        // Only allocate frame_buffer once per stream, as the size is not expected to change
-        AVMallocPtr converted_frame_buffer{static_cast<u8*>(av_malloc(linear_size)), av_free};
+        if (!converted_frame_buffer) {
+            converted_frame_buffer = AVMallocPtr{static_cast<u8*>(av_malloc(linear_size)), av_free};
+        }
        const int converted_stride{frame->width * 4};
        u8* const converted_frame_buf_addr{converted_frame_buffer.get()};
@@ -104,12 +109,12 @@ void Vic::Execute() {
            const u32 block_height = static_cast<u32>(config.block_linear_height_log2);
            const auto size = Tegra::Texture::CalculateSize(true, 4, frame->width, frame->height, 1,
                                                            block_height, 0);
-            std::vector<u8> swizzled_data(size);
+            luma_buffer.resize(size);
            Tegra::Texture::SwizzleSubrect(frame->width, frame->height, frame->width * 4,
-                                           frame->width, 4, swizzled_data.data(),
+                                           frame->width, 4, luma_buffer.data(),
                                           converted_frame_buffer.get(), block_height, 0, 0);
-            gpu.MemoryManager().WriteBlock(output_surface_luma_address, swizzled_data.data(), size);
+            gpu.MemoryManager().WriteBlock(output_surface_luma_address, luma_buffer.data(), size);
        } else {
            // send pitch linear frame
            gpu.MemoryManager().WriteBlock(output_surface_luma_address, converted_frame_buf_addr,
@@ -132,8 +137,8 @@ void Vic::Execute() {
        const auto stride = frame->linesize[0];
        const auto half_stride = frame->linesize[1];
-        std::vector<u8> luma_buffer(aligned_width * surface_height);
+        luma_buffer.resize(aligned_width * surface_height);
-        std::vector<u8> chroma_buffer(aligned_width * half_height);
+        chroma_buffer.resize(aligned_width * half_height);
        // Populate luma buffer
        for (std::size_t y = 0; y < surface_height - 1; ++y) {
diff --git a/src/video_core/command_classes/vic.h b/src/video_core/command_classes/vic.h
index 8c4e284a1..6eaf72f21 100644
--- a/src/video_core/command_classes/vic.h
+++ b/src/video_core/command_classes/vic.h
@@ -97,6 +97,13 @@ private:
    GPU& gpu;
    std::shared_ptr<Tegra::Nvdec> nvdec_processor;
+    /// Avoid reallocation of the following buffers every frame, as their
+    /// size does not change during a stream
+    using AVMallocPtr = std::unique_ptr<u8, decltype(&av_free)>;
+    AVMallocPtr converted_frame_buffer;
+    std::vector<u8> luma_buffer;
+    std::vector<u8> chroma_buffer;
    GPUVAddr config_struct_address{};
    GPUVAddr output_surface_luma_address{};
    GPUVAddr output_surface_chroma_u_address{};
author	ameerj	2020-11-23 13:25:01 -0500
committer	ameerj	2021-02-13 13:07:31 -0500
commit	ac265a72ce4176ceb3cd10a5548ab71519771640 (patch)
tree	0acde029388d465a5801db9106dd8f4e026e57e8 /src/video_core/command_classes
parent	Merge pull request #5919 from ReinUsesLisp/stream-buffer-tragic (diff)
download	yuzu-ac265a72ce4176ceb3cd10a5548ab71519771640.tar.gz yuzu-ac265a72ce4176ceb3cd10a5548ab71519771640.tar.xz yuzu-ac265a72ce4176ceb3cd10a5548ab71519771640.zip