-
Notifications
You must be signed in to change notification settings - Fork 62
Improvement in average inference latency for models running on OVEP NPU #441
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
89094a8
2e4b205
63e8aee
cd88b0c
89127f0
d43219f
274e6af
6feae84
c1f3b3e
1e3dadd
61a2d4a
5800966
1af175e
0faaf9f
a1f92cd
6ee25da
a9f2357
8844556
8b32612
4830c89
48c5569
1791b90
f6f439b
39c0cba
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -145,6 +145,10 @@ ORT_API_STATUS_IMPL(OrtApis::CreateMemoryInfo, _In_ const char* name1, enum OrtA | |
| *out = new OrtMemoryInfo( | ||
| name1, type, OrtDevice(OrtDevice::GPU, OrtDevice::MemType::DEFAULT, static_cast<OrtDevice::DeviceId>(id1)), id1, | ||
| mem_type1); | ||
| } else if (strcmp(name1, onnxruntime::OpenVINO_RT_NPU) == 0) { | ||
| *out = new OrtMemoryInfo( | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Verify if the |
||
| name1, type, OrtDevice(OrtDevice::NPU, OrtDevice::MemType::DEFAULT, static_cast<OrtDevice::DeviceId>(id1)), id1, | ||
| mem_type1); | ||
| } else if (strcmp(name1, onnxruntime::CUDA_PINNED) == 0) { | ||
| *out = new OrtMemoryInfo( | ||
| onnxruntime::CUDA_PINNED, type, OrtDevice(OrtDevice::CPU, OrtDevice::MemType::CUDA_PINNED, static_cast<OrtDevice::DeviceId>(id1)), | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -48,14 +48,6 @@ BasicBackend::BasicBackend(std::unique_ptr<ONNX_NAMESPACE::ModelProto>& model_pr | |
| // Set the inference_num_threads property of the CPU | ||
| SetNumThreads(device_config); | ||
|
|
||
| #ifndef NDEBUG | ||
| if (IsDebugEnabled()) { | ||
| std::string file_name = subgraph_context.subgraph_name + "_static.onnx"; | ||
| std::fstream outfile(file_name, std::ios::out | std::ios::trunc | std::ios::binary); | ||
| model_proto.SerializeToOstream(outfile); | ||
| } | ||
| #endif | ||
|
|
||
| try { | ||
| std::string dev_prec = global_context.device_type + "_" + global_context_.precision_str; | ||
|
|
||
|
|
@@ -295,16 +287,92 @@ void BasicBackend::StartAsyncInference(Ort::KernelContext& context, OVInferReque | |
| ORT_THROW(msg); | ||
| } | ||
| } else { | ||
| OVTensorPtr graph_input_blob; | ||
| auto tensor = context.GetInput(subgraph_context_.input_names.at(input_name)); | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Please add comments to explain the code here. Is this compatible across all OV Versions. |
||
| auto allocator_name = tensor.GetTensorMemoryInfo().GetAllocatorName(); | ||
| ov_tensor_data_t ov_tensor_key; | ||
| ort_tensor_key_t ort_tensor_key{tensor.GetTensorRawData(), allocator_name}; | ||
| if (const auto& it = ort_ov_tensor_map.find(ort_tensor_key); it != ort_ov_tensor_map.end()) { | ||
| ov_tensor_key = it->second; | ||
| } else { | ||
| // Does this make sense for both types of allocators? | ||
| auto input = graph_input_info.at(input_idx); | ||
| ov_tensor_key.tensor_ptr = std::make_shared<ov::Tensor>(input.get_element_type(), input.get_shape(), | ||
| (void*)tensor.GetTensorRawData()); | ||
| if (allocator_name == OpenVINO_RT_NPU) { | ||
| ov_tensor_key.copy_needed = false; | ||
| } else { | ||
| ov_tensor_key.copy_needed = true; | ||
| } | ||
| ort_ov_tensor_map.emplace(ort_tensor_key, ov_tensor_key); | ||
|
|
||
| try { | ||
| infer_request->SetTensor(input_name, ov_tensor_key.tensor_ptr); | ||
| } catch (const char* msg) { | ||
| ORT_THROW(msg); | ||
| } | ||
| } | ||
|
|
||
| if (ov_tensor_key.copy_needed) { | ||
| const char* ort_tensor_data = tensor.GetTensorData<char>(); | ||
| size_t tensor_data_size = ov_tensor_key.tensor_ptr->get_byte_size(); | ||
| auto ort_batch_memory_offset = ort_tensor_data + tensor_data_size * batch_slice_idx; | ||
| std::memcpy(ov_tensor_key.tensor_ptr->data(), ort_batch_memory_offset, tensor_data_size); | ||
| } | ||
| } | ||
| input_idx++; | ||
| } | ||
|
|
||
| // Set the output blob as remote blob | ||
| auto graph_output_info = exe_network_.Get().outputs(); | ||
| auto output_idx = 0; | ||
| for (auto output_info_iter = graph_output_info.begin(); | ||
| output_info_iter != graph_output_info.end(); ++output_info_iter) { | ||
| auto output_names = output_info_iter->get_names(); | ||
| std::string onnx_output_name; | ||
| std::string output_name; | ||
| bool output_name_found = false; | ||
| // using the output name retrieved from ONNX original to match with the output names returned by OV tensors | ||
| for (auto it = subgraph_context_.output_names.begin(); it != subgraph_context_.output_names.end(); ++it) { | ||
| onnx_output_name = it->first; | ||
| if (output_names.find(onnx_output_name) != output_names.end()) { | ||
| // Assigning the output_name | ||
| output_name = it->first; | ||
| output_name_found = true; | ||
| break; | ||
| } | ||
| } | ||
| size_t batch_size = 1; | ||
| Ort::UnownedValue tensor = GetOutputTensor(context, | ||
| batch_size, | ||
| infer_request, | ||
| output_name, | ||
| subgraph_context_.output_names); | ||
| auto allocator_name = tensor.GetTensorMemoryInfo().GetAllocatorName(); | ||
|
|
||
| ov_tensor_data_t ov_tensor_data; | ||
| ort_tensor_key_t ort_tensor_key{tensor.GetTensorRawData(), allocator_name}; | ||
| if (const auto& it = ort_ov_tensor_map.find(ort_tensor_key); it != ort_ov_tensor_map.end()) { | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Please try not using auto to avoid coverity issues.
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Tried to avoid auto wherever possible in the new changes. For example this auto (line 354) has a data type of There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. I would recommend to avoid auto, when the return type is smaller for example cases like below - std::unique_ptr<ONNX_NAMESPACE::AttributeProto> sdk_version_attr = ONNX_NAMESPACE::AttributeProto::Create(); We can use auto only if the command is exceeding 120 characters length for the lintrunner and having complex type as mentioned in the chat above |
||
| ov_tensor_data = it->second; | ||
| } else { | ||
| auto output = graph_output_info.at(output_idx); | ||
| ov_tensor_data.tensor_ptr = std::make_shared<ov::Tensor>(output.get_element_type(), output.get_shape(), | ||
| (void*)tensor.GetTensorRawData()); | ||
| if(allocator_name == OpenVINO_RT_NPU) { | ||
| ov_tensor_data.copy_needed = false; | ||
| } else { | ||
| ov_tensor_data.copy_needed = true; | ||
| } | ||
| ort_ov_tensor_map.emplace(ort_tensor_key, ov_tensor_data); | ||
|
|
||
| try { | ||
| graph_input_blob = infer_request->GetTensor(input_name); | ||
| infer_request->SetTensor(output_name, ov_tensor_data.tensor_ptr); | ||
| } catch (const char* msg) { | ||
| ORT_THROW(msg); | ||
| } | ||
| FillInputBlob(std::move(graph_input_blob), batch_slice_idx, std::move(input_name), context, subgraph_context_); | ||
| } | ||
| input_idx++; | ||
| output_idx++; | ||
| } | ||
|
|
||
| // Start Async inference | ||
| infer_request->StartAsync(); | ||
| } catch (const char* msg) { | ||
|
|
@@ -430,7 +498,6 @@ void BasicBackend::CompleteAsyncInference(Ort::KernelContext& context, OVInferRe | |
| auto graph_output_info = exe_network_.Get().outputs(); | ||
| for (auto output_info_iter = graph_output_info.begin(); | ||
| output_info_iter != graph_output_info.end(); ++output_info_iter) { | ||
| OVTensorPtr graph_output_blob; | ||
| auto output_names = output_info_iter->get_names(); | ||
| std::string onnx_output_name; | ||
| std::string output_name; | ||
|
|
@@ -454,20 +521,24 @@ void BasicBackend::CompleteAsyncInference(Ort::KernelContext& context, OVInferRe | |
| " doesn't exist in the " | ||
| "list of OpenVINO output tensor names"); | ||
| } | ||
| try { | ||
| graph_output_blob = infer_request->GetTensor(output_name); | ||
| } catch (const char* msg) { | ||
| ORT_THROW(msg); | ||
| } | ||
|
|
||
| size_t batch_size = 1; | ||
| Ort::UnownedValue output_tensor = | ||
| GetOutputTensor(context, batch_size, infer_request, std::move(output_name), subgraph_context_.output_names); | ||
| auto mem_info = output_tensor.GetTensorMemoryInfo(); | ||
| if (mem_info.GetAllocatorName() == OpenVINO_GPU) { | ||
| return; | ||
| auto allocator_name = output_tensor.GetTensorMemoryInfo().GetAllocatorName(); | ||
| ov_tensor_data_t ov_tensor_data; | ||
| ort_tensor_key_t ort_tensor_key{output_tensor.GetTensorRawData(), allocator_name}; | ||
| if (const auto& it = ort_ov_tensor_map.find(ort_tensor_key); it != ort_ov_tensor_map.end()) { | ||
| ov_tensor_data = it->second; | ||
| } else { | ||
| size_t batch_slice = 0; | ||
| FillOutputBlob(std::move(graph_output_blob), output_tensor, batch_slice); | ||
| ORT_THROW(log_tag + "Expected all outputs to have associated OV::Tensor's"); | ||
| } | ||
|
|
||
| if (ov_tensor_data.copy_needed) { | ||
| auto ort_tensor_data = output_tensor.GetTensorMutableData<char>(); | ||
| size_t tensor_data_size = ov_tensor_data.tensor_ptr->get_byte_size(); | ||
| auto ort_batch_memory_offset = ort_tensor_data /*+ tensor_data_size * batch_size*/; | ||
| std::memcpy(ort_batch_memory_offset, ov_tensor_data.tensor_ptr->data(), tensor_data_size); | ||
| } | ||
| } | ||
|
|
||
|
|
||
| Original file line number | Diff line number | Diff line change | ||||
|---|---|---|---|---|---|---|
|
|
@@ -11,6 +11,7 @@ | |||||
| #include <string> | ||||||
| #include <condition_variable> | ||||||
| #include <mutex> | ||||||
| #include <map> | ||||||
|
|
||||||
| #include "core/session/onnxruntime_cxx_api.h" | ||||||
| #include "core/providers/openvino/contexts.h" | ||||||
|
|
@@ -20,6 +21,11 @@ | |||||
| namespace onnxruntime { | ||||||
| namespace openvino_ep { | ||||||
|
|
||||||
| struct ov_tensor_data_t { | ||||||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Class & struct naming case convention is not matching the overall ORT convention, you can use name |
||||||
| OVTensorPtr tensor_ptr; | ||||||
| bool copy_needed; | ||||||
| }; | ||||||
|
|
||||||
| class InferRequestsQueue; | ||||||
| class BasicBackend : public IBackend { | ||||||
| public: | ||||||
|
|
@@ -60,6 +66,9 @@ class BasicBackend : public IBackend { | |||||
| #if defined IO_BUFFER_ENABLED | ||||||
| OVRemoteContextPtr remote_context_; | ||||||
| #endif | ||||||
|
|
||||||
| using ort_tensor_key_t = std::pair<const void *, const std::string>; | ||||||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Naming a typedef or a namespace can be conflicting with variable name as suggested above,
Suggested change
|
||||||
| std::map<ort_tensor_key_t, ov_tensor_data_t> ort_ov_tensor_map; | ||||||
| }; | ||||||
|
|
||||||
| class InferRequestsQueue { | ||||||
|
|
||||||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -10,6 +10,9 @@ | |
| #include "core/providers/openvino/onnx_ctx_model_helper.h" | ||
| #include "core/providers/openvino/ov_versions/capability.h" | ||
| #include "openvino/core/version.hpp" | ||
| #ifdef USE_DEVICE_MEMORY | ||
| #include "core/providers/openvino/ov_allocator.h" | ||
| #endif | ||
|
|
||
| #define MEMCPY_S(dest, src, destsz, srcsz) memcpy(dest, src, std::min(destsz, srcsz)) | ||
|
|
||
|
|
@@ -180,4 +183,18 @@ common::Status OpenVINOExecutionProvider::Compile( | |
| return Status::OK(); | ||
| } | ||
|
|
||
| #ifdef USE_DEVICE_MEMORY | ||
| std::vector<AllocatorPtr> OpenVINOExecutionProvider::CreatePreferredAllocators() { | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Why this function name is not NPU Specific ? |
||
| AllocatorCreationInfo npu_allocator_info { | ||
| [this](OrtDevice::DeviceId device_id) { | ||
| return std::make_unique<OVRTAllocator>(global_context_->ie_core.Get(), OrtDevice::NPU, device_id, OpenVINO_RT_NPU); | ||
| }, | ||
| 0, | ||
| }; | ||
|
|
||
| // fill in allocator | ||
| return std::vector<AllocatorPtr>{CreateAllocator(npu_allocator_info)}; | ||
| } | ||
| #endif | ||
|
|
||
| } // namespace onnxruntime | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -189,7 +189,9 @@ class OpenVINOExecutionProvider : public IExecutionProvider { | |
| const void* GetExecutionHandle() const noexcept override { | ||
| return nullptr; | ||
| } | ||
|
|
||
| #ifdef USE_DEVICE_MEMORY | ||
| std::vector<AllocatorPtr> CreatePreferredAllocators() override; | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. If this is only for NPU, the function name should suggest that
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. yes, this is for NPU. |
||
| #endif | ||
| private: | ||
| std::unique_ptr<openvino_ep::GlobalContext> global_context_; | ||
| openvino_ep::EPCtxHandler ep_ctx_handle_{}; | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,55 @@ | ||
| // Copyright (C) Intel Corporation | ||
Check warningCode scanning / lintrunner CLANGFORMAT/format
See https://clang.llvm.org/docs/ClangFormat.html.
Run `lintrunner -a` to apply this patch.
|
||
| // Licensed under the MIT License | ||
| #ifdef USE_DEVICE_MEMORY | ||
| #include "core/providers/openvino/ov_allocator.h" | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Will add npu plugin include files work acroos openvino versions ? |
||
| #include "core/providers/openvino/ov_interface.h" | ||
| #include "openvino/runtime/intel_npu/level_zero/level_zero.hpp" | ||
| #include "openvino/runtime/intel_npu/properties.hpp" | ||
|
|
||
| namespace onnxruntime { | ||
|
|
||
| using namespace openvino_ep; | ||
|
|
||
| constexpr size_t default_alignment = 4096; | ||
|
|
||
| static inline size_t align_up(size_t size, size_t pow2_alignment) { | ||
| return (size + pow2_alignment - 1) & ~(pow2_alignment - 1); | ||
| } | ||
|
|
||
| OVRTAllocator::OVRTAllocator(ov::Core& core, OrtDevice::DeviceType device_type, OrtDevice::DeviceId device_id, const char* name) : IAllocator(OrtMemoryInfo(name, OrtAllocatorType::OrtDeviceAllocator, OrtDevice(device_type, OrtDevice::MemType::DEFAULT, device_id), device_id, OrtMemTypeCPUInput)), core_(core) { | ||
| if (device_type == OrtDevice::NPU) { | ||
| remote_ctx_ = core_.get_default_context("NPU").as<ov::intel_npu::level_zero::ZeroContext>(); | ||
| } else { | ||
| ORT_THROW("Invalid device type"); | ||
| } | ||
| } | ||
|
|
||
| void* OVRTAllocator::Alloc(size_t size) { | ||
| try { | ||
| size_t alloc_size = align_up(size + sizeof(ov::Tensor*) + default_alignment, default_alignment); | ||
| ov::Tensor* tensor = new ov::Tensor(remote_ctx_.create_host_tensor(ov::element::Type_t::u8, | ||
| { alloc_size })); | ||
| uintptr_t data_ptr = reinterpret_cast<uintptr_t>(tensor->data()); | ||
|
|
||
| ov::Tensor** ptr = reinterpret_cast<ov::Tensor**>(align_up(data_ptr + sizeof(ov::Tensor*), default_alignment)); | ||
| ptr[-1] = tensor; | ||
|
|
||
| return reinterpret_cast<void*>(ptr); | ||
|
|
||
| } catch (const ov::Exception& e) { | ||
| ORT_THROW(std::string("Alloc failed: ") + e.what()); | ||
| } | ||
| return nullptr; | ||
| } | ||
|
|
||
| void OVRTAllocator::Free(void* p) { | ||
| try { | ||
| ov::Tensor** ptr = reinterpret_cast<ov::Tensor**>(p); | ||
| delete ptr[-1]; | ||
| } catch (const ov::Exception& e) { | ||
| ORT_THROW(std::string("Free failed: ") + e.what()); | ||
| } | ||
| } | ||
|
|
||
| } // namespace onnxruntime | ||
| #endif | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,25 @@ | ||
| // Copyright (C) Intel Corporation | ||
Check warningCode scanning / lintrunner CLANGFORMAT/format
See https://clang.llvm.org/docs/ClangFormat.html.
Run `lintrunner -a` to apply this patch.
|
||
| // Licensed under the MIT License | ||
| #ifdef USE_DEVICE_MEMORY | ||
| #pragma once | ||
|
|
||
| #include "core/common/inlined_containers.h" | ||
| #include "core/framework/allocator.h" | ||
| #include "openvino/runtime/remote_context.hpp" | ||
|
|
||
|
|
||
| namespace onnxruntime { | ||
|
|
||
| class OVRTAllocator : public IAllocator { | ||
| public: | ||
| OVRTAllocator(ov::Core &core, OrtDevice::DeviceType device_type, OrtDevice::DeviceId device_id, const char* name); | ||
| void* Alloc(size_t size) override; | ||
| void Free(void* p) override; | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Can be called inside destructor like |
||
|
|
||
| private: | ||
| ov::Core &core_; | ||
| ov::RemoteContext remote_ctx_; | ||
| }; | ||
|
|
||
| } // namespace onnxruntime | ||
| #endif | ||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
How is OpenVINO_RT different from OpenVINO_CPU ?