Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
89094a8
Prototype shared memory allocator on Windows using OV-EP
javier-intel Jul 18, 2024
2e4b205
Partially working allocator.
ericcraw Aug 23, 2024
63e8aee
Hard code onnx perf to use RT NPU allocator for inputs
ericcraw Aug 24, 2024
cd88b0c
Fix allocation lookups coming from different level zero contexts
ericcraw Aug 26, 2024
89127f0
Page align OV allocation
ericcraw Aug 26, 2024
d43219f
Allocate input as WC
ericcraw Aug 26, 2024
274e6af
Only set tensors when they have changed.
ericcraw Aug 26, 2024
6feae84
Revert "Allocate input as WC"
ericcraw Aug 26, 2024
c1f3b3e
Hard code onnx perf to use RT NPU for outputs
ericcraw Aug 27, 2024
1e3dadd
Revert "Hard code onnx perf to use RT NPU for outputs"
ericcraw Aug 27, 2024
61a2d4a
Hard code onnx perf to use RT NPU for outputs fixed
ericcraw Aug 27, 2024
5800966
Fix onnx_perf_test app crash on tensor destroy
ericcraw Aug 27, 2024
1af175e
Merge branch 'ort_allocator_hacking_temp' into npu_allocator_
saurabhkale17 Aug 30, 2024
0faaf9f
refactor: remove redundant ort_shape_to_ovshape lambda function
saurabhkale17 Aug 28, 2024
a1f92cd
alocate buffer in NPU visible region from perf test application
saurabhkale17 Aug 29, 2024
6ee25da
remove redundant code
saurabhkale17 Aug 29, 2024
a9f2357
add command line parameter in perf test for using remote tensors
saurabhkale17 Aug 29, 2024
8844556
remove redundant code
saurabhkale17 Aug 30, 2024
8b32612
remove redundant statements
saurabhkale17 Aug 30, 2024
4830c89
fix crash during inference
saurabhkale17 Sep 2, 2024
48c5569
remove redundant code
saurabhkale17 Sep 3, 2024
1791b90
enable backward compatibility of remote tensor feature
saurabhkale17 Sep 4, 2024
f6f439b
Revert "enable backward compatibility of remote tensor feature"
saurabhkale17 Sep 5, 2024
39c0cba
enable backward compatibility of remote tensor feature in OVEP
saurabhkale17 Sep 5, 2024
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions cmake/onnxruntime_providers_openvino.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,10 @@
message(FATAL_ERROR "OpenVINO 2024.0 and newer are supported. Please, use latest OpenVINO release")
endif()

if(OpenVINO_VERSION VERSION_GREATER_EQUAL 2024.4)
add_definitions(-DUSE_DEVICE_MEMORY=1)
endif()

if (WIN32)
unset(CMAKE_MAP_IMPORTED_CONFIG_RELWITHDEBINFO)
endif()
Expand Down
2 changes: 2 additions & 0 deletions include/onnxruntime/core/framework/allocator.h
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,8 @@ constexpr const char* HIP = "Hip";
constexpr const char* HIP_PINNED = "HipPinned";
constexpr const char* OpenVINO_CPU = "OpenVINO_CPU";
constexpr const char* OpenVINO_GPU = "OpenVINO_GPU";
constexpr const char* OpenVINO_RT = "OpenVINO_RT";

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

How is OpenVINO_RT different from OpenVINO_CPU ?

constexpr const char* OpenVINO_RT_NPU = "OpenVINO_RT_NPU";
constexpr const char* WEBGPU_BUFFER = "WebGPU_Buffer";

constexpr size_t kAllocAlignment = 256;
Expand Down
4 changes: 4 additions & 0 deletions onnxruntime/core/framework/allocator.cc
Original file line number Diff line number Diff line change
Expand Up @@ -145,6 +145,10 @@ ORT_API_STATUS_IMPL(OrtApis::CreateMemoryInfo, _In_ const char* name1, enum OrtA
*out = new OrtMemoryInfo(
name1, type, OrtDevice(OrtDevice::GPU, OrtDevice::MemType::DEFAULT, static_cast<OrtDevice::DeviceId>(id1)), id1,
mem_type1);
} else if (strcmp(name1, onnxruntime::OpenVINO_RT_NPU) == 0) {
*out = new OrtMemoryInfo(

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Verify if the out variable memory is deleted later on to avoid memory leaks, once can use smart pointer to avoid explicit delete call

name1, type, OrtDevice(OrtDevice::NPU, OrtDevice::MemType::DEFAULT, static_cast<OrtDevice::DeviceId>(id1)), id1,
mem_type1);
} else if (strcmp(name1, onnxruntime::CUDA_PINNED) == 0) {
*out = new OrtMemoryInfo(
onnxruntime::CUDA_PINNED, type, OrtDevice(OrtDevice::CPU, OrtDevice::MemType::CUDA_PINNED, static_cast<OrtDevice::DeviceId>(id1)),
Expand Down
117 changes: 94 additions & 23 deletions onnxruntime/core/providers/openvino/backends/basic_backend.cc
Original file line number Diff line number Diff line change
Expand Up @@ -48,14 +48,6 @@ BasicBackend::BasicBackend(std::unique_ptr<ONNX_NAMESPACE::ModelProto>& model_pr
// Set the inference_num_threads property of the CPU
SetNumThreads(device_config);

#ifndef NDEBUG
if (IsDebugEnabled()) {
std::string file_name = subgraph_context.subgraph_name + "_static.onnx";
std::fstream outfile(file_name, std::ios::out | std::ios::trunc | std::ios::binary);
model_proto.SerializeToOstream(outfile);
}
#endif

try {
std::string dev_prec = global_context.device_type + "_" + global_context_.precision_str;

Expand Down Expand Up @@ -295,16 +287,92 @@ void BasicBackend::StartAsyncInference(Ort::KernelContext& context, OVInferReque
ORT_THROW(msg);
}
} else {
OVTensorPtr graph_input_blob;
auto tensor = context.GetInput(subgraph_context_.input_names.at(input_name));

@sfatimar sfatimar Sep 4, 2024

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Please add comments to explain the code here. Is this compatible across all OV Versions.

auto allocator_name = tensor.GetTensorMemoryInfo().GetAllocatorName();
ov_tensor_data_t ov_tensor_key;
ort_tensor_key_t ort_tensor_key{tensor.GetTensorRawData(), allocator_name};
if (const auto& it = ort_ov_tensor_map.find(ort_tensor_key); it != ort_ov_tensor_map.end()) {
ov_tensor_key = it->second;
} else {
// Does this make sense for both types of allocators?
auto input = graph_input_info.at(input_idx);
ov_tensor_key.tensor_ptr = std::make_shared<ov::Tensor>(input.get_element_type(), input.get_shape(),
(void*)tensor.GetTensorRawData());
if (allocator_name == OpenVINO_RT_NPU) {
ov_tensor_key.copy_needed = false;
} else {
ov_tensor_key.copy_needed = true;
}
ort_ov_tensor_map.emplace(ort_tensor_key, ov_tensor_key);

try {
infer_request->SetTensor(input_name, ov_tensor_key.tensor_ptr);
} catch (const char* msg) {
ORT_THROW(msg);
}
}

if (ov_tensor_key.copy_needed) {
const char* ort_tensor_data = tensor.GetTensorData<char>();
size_t tensor_data_size = ov_tensor_key.tensor_ptr->get_byte_size();
auto ort_batch_memory_offset = ort_tensor_data + tensor_data_size * batch_slice_idx;
std::memcpy(ov_tensor_key.tensor_ptr->data(), ort_batch_memory_offset, tensor_data_size);
}
}
input_idx++;
}

// Set the output blob as remote blob
auto graph_output_info = exe_network_.Get().outputs();
auto output_idx = 0;
for (auto output_info_iter = graph_output_info.begin();
output_info_iter != graph_output_info.end(); ++output_info_iter) {
auto output_names = output_info_iter->get_names();
std::string onnx_output_name;
std::string output_name;
bool output_name_found = false;
// using the output name retrieved from ONNX original to match with the output names returned by OV tensors
for (auto it = subgraph_context_.output_names.begin(); it != subgraph_context_.output_names.end(); ++it) {
onnx_output_name = it->first;
if (output_names.find(onnx_output_name) != output_names.end()) {
// Assigning the output_name
output_name = it->first;
output_name_found = true;
break;
}
}
size_t batch_size = 1;
Ort::UnownedValue tensor = GetOutputTensor(context,
batch_size,
infer_request,
output_name,
subgraph_context_.output_names);
auto allocator_name = tensor.GetTensorMemoryInfo().GetAllocatorName();

ov_tensor_data_t ov_tensor_data;
ort_tensor_key_t ort_tensor_key{tensor.GetTensorRawData(), allocator_name};
if (const auto& it = ort_ov_tensor_map.find(ort_tensor_key); it != ort_ov_tensor_map.end()) {

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Please try not using auto to avoid coverity issues.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Tried to avoid auto wherever possible in the new changes.
But for the situations where the data type is very complex it will be better with auto.

For example this auto (line 354) has a data type of
class std::_Tree_iterator<class std::_Tree_val<struct std::_Tree_simple_types<struct std::pair<struct std::pair<void const * __ptr64,class std::basic_string<char,struct std::char_traits,class std::allocator > const > const ,struct onnxruntime::openvino_ep::ov_tensor_data_t> > > >

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I would recommend to avoid auto, when the return type is smaller for example cases like below -

std::unique_ptr<ONNX_NAMESPACE::AttributeProto> sdk_version_attr = ONNX_NAMESPACE::AttributeProto::Create();

We can use auto only if the command is exceeding 120 characters length for the lintrunner and having complex type as mentioned in the chat above

ov_tensor_data = it->second;
} else {
auto output = graph_output_info.at(output_idx);
ov_tensor_data.tensor_ptr = std::make_shared<ov::Tensor>(output.get_element_type(), output.get_shape(),
(void*)tensor.GetTensorRawData());
if(allocator_name == OpenVINO_RT_NPU) {
ov_tensor_data.copy_needed = false;
} else {
ov_tensor_data.copy_needed = true;
}
ort_ov_tensor_map.emplace(ort_tensor_key, ov_tensor_data);

try {
graph_input_blob = infer_request->GetTensor(input_name);
infer_request->SetTensor(output_name, ov_tensor_data.tensor_ptr);
} catch (const char* msg) {
ORT_THROW(msg);
}
FillInputBlob(std::move(graph_input_blob), batch_slice_idx, std::move(input_name), context, subgraph_context_);
}
input_idx++;
output_idx++;
}

// Start Async inference
infer_request->StartAsync();
} catch (const char* msg) {
Expand Down Expand Up @@ -430,7 +498,6 @@ void BasicBackend::CompleteAsyncInference(Ort::KernelContext& context, OVInferRe
auto graph_output_info = exe_network_.Get().outputs();
for (auto output_info_iter = graph_output_info.begin();
output_info_iter != graph_output_info.end(); ++output_info_iter) {
OVTensorPtr graph_output_blob;
auto output_names = output_info_iter->get_names();
std::string onnx_output_name;
std::string output_name;
Expand All @@ -454,20 +521,24 @@ void BasicBackend::CompleteAsyncInference(Ort::KernelContext& context, OVInferRe
" doesn't exist in the "
"list of OpenVINO output tensor names");
}
try {
graph_output_blob = infer_request->GetTensor(output_name);
} catch (const char* msg) {
ORT_THROW(msg);
}

size_t batch_size = 1;
Ort::UnownedValue output_tensor =
GetOutputTensor(context, batch_size, infer_request, std::move(output_name), subgraph_context_.output_names);
auto mem_info = output_tensor.GetTensorMemoryInfo();
if (mem_info.GetAllocatorName() == OpenVINO_GPU) {
return;
auto allocator_name = output_tensor.GetTensorMemoryInfo().GetAllocatorName();
ov_tensor_data_t ov_tensor_data;
ort_tensor_key_t ort_tensor_key{output_tensor.GetTensorRawData(), allocator_name};
if (const auto& it = ort_ov_tensor_map.find(ort_tensor_key); it != ort_ov_tensor_map.end()) {
ov_tensor_data = it->second;
} else {
size_t batch_slice = 0;
FillOutputBlob(std::move(graph_output_blob), output_tensor, batch_slice);
ORT_THROW(log_tag + "Expected all outputs to have associated OV::Tensor's");
}

if (ov_tensor_data.copy_needed) {
auto ort_tensor_data = output_tensor.GetTensorMutableData<char>();
size_t tensor_data_size = ov_tensor_data.tensor_ptr->get_byte_size();
auto ort_batch_memory_offset = ort_tensor_data /*+ tensor_data_size * batch_size*/;
std::memcpy(ort_batch_memory_offset, ov_tensor_data.tensor_ptr->data(), tensor_data_size);
}
}

Expand Down
9 changes: 9 additions & 0 deletions onnxruntime/core/providers/openvino/backends/basic_backend.h
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
#include <string>
#include <condition_variable>
#include <mutex>
#include <map>

#include "core/session/onnxruntime_cxx_api.h"
#include "core/providers/openvino/contexts.h"
Expand All @@ -20,6 +21,11 @@
namespace onnxruntime {
namespace openvino_ep {

struct ov_tensor_data_t {

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Class & struct naming case convention is not matching the overall ORT convention, you can use name struct OVTensorData, above looks like a variable name

OVTensorPtr tensor_ptr;
bool copy_needed;
};

class InferRequestsQueue;
class BasicBackend : public IBackend {
public:
Expand Down Expand Up @@ -60,6 +66,9 @@ class BasicBackend : public IBackend {
#if defined IO_BUFFER_ENABLED
OVRemoteContextPtr remote_context_;
#endif

using ort_tensor_key_t = std::pair<const void *, const std::string>;

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Naming a typedef or a namespace can be conflicting with variable name as suggested above,

Suggested change
using ort_tensor_key_t = std::pair<const void *, const std::string>;
using ORTTensorKey = std::pair<const void *, const std::string>;

std::map<ort_tensor_key_t, ov_tensor_data_t> ort_ov_tensor_map;
};

class InferRequestsQueue {
Expand Down
17 changes: 17 additions & 0 deletions onnxruntime/core/providers/openvino/openvino_execution_provider.cc
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,9 @@
#include "core/providers/openvino/onnx_ctx_model_helper.h"
#include "core/providers/openvino/ov_versions/capability.h"
#include "openvino/core/version.hpp"
#ifdef USE_DEVICE_MEMORY
#include "core/providers/openvino/ov_allocator.h"
#endif

#define MEMCPY_S(dest, src, destsz, srcsz) memcpy(dest, src, std::min(destsz, srcsz))

Expand Down Expand Up @@ -180,4 +183,18 @@ common::Status OpenVINOExecutionProvider::Compile(
return Status::OK();
}

#ifdef USE_DEVICE_MEMORY
std::vector<AllocatorPtr> OpenVINOExecutionProvider::CreatePreferredAllocators() {

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Why this function name is not NPU Specific ?

AllocatorCreationInfo npu_allocator_info {
[this](OrtDevice::DeviceId device_id) {
return std::make_unique<OVRTAllocator>(global_context_->ie_core.Get(), OrtDevice::NPU, device_id, OpenVINO_RT_NPU);
},
0,
};

// fill in allocator
return std::vector<AllocatorPtr>{CreateAllocator(npu_allocator_info)};
}
#endif

} // namespace onnxruntime
Original file line number Diff line number Diff line change
Expand Up @@ -189,7 +189,9 @@ class OpenVINOExecutionProvider : public IExecutionProvider {
const void* GetExecutionHandle() const noexcept override {
return nullptr;
}

#ifdef USE_DEVICE_MEMORY
std::vector<AllocatorPtr> CreatePreferredAllocators() override;

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

If this is only for NPU, the function name should suggest that

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

yes, this is for NPU.
This CreatePreferredAllocators is the function of OpenVINOExecutionProvider class.
OpenVINOExecutionProvider class is inherited from IExecutionProvider class which has this function.
And this is a function which is used by all the execution providers.

#endif
private:
std::unique_ptr<openvino_ep::GlobalContext> global_context_;
openvino_ep::EPCtxHandler ep_ctx_handle_{};
Expand Down
55 changes: 55 additions & 0 deletions onnxruntime/core/providers/openvino/ov_allocator.cc
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
// Copyright (C) Intel Corporation
Comment thread Fixed
Comment thread Fixed

Check warning

Code scanning / lintrunner

CLANGFORMAT/format

See https://clang.llvm.org/docs/ClangFormat.html. Run `lintrunner -a` to apply this patch.
// Licensed under the MIT License
#ifdef USE_DEVICE_MEMORY
#include "core/providers/openvino/ov_allocator.h"

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Will add npu plugin include files work acroos openvino versions ?

#include "core/providers/openvino/ov_interface.h"
#include "openvino/runtime/intel_npu/level_zero/level_zero.hpp"
#include "openvino/runtime/intel_npu/properties.hpp"

namespace onnxruntime {

using namespace openvino_ep;

constexpr size_t default_alignment = 4096;

static inline size_t align_up(size_t size, size_t pow2_alignment) {
return (size + pow2_alignment - 1) & ~(pow2_alignment - 1);
}

OVRTAllocator::OVRTAllocator(ov::Core& core, OrtDevice::DeviceType device_type, OrtDevice::DeviceId device_id, const char* name) : IAllocator(OrtMemoryInfo(name, OrtAllocatorType::OrtDeviceAllocator, OrtDevice(device_type, OrtDevice::MemType::DEFAULT, device_id), device_id, OrtMemTypeCPUInput)), core_(core) {
if (device_type == OrtDevice::NPU) {
remote_ctx_ = core_.get_default_context("NPU").as<ov::intel_npu::level_zero::ZeroContext>();
} else {
ORT_THROW("Invalid device type");
}
}

void* OVRTAllocator::Alloc(size_t size) {
try {
size_t alloc_size = align_up(size + sizeof(ov::Tensor*) + default_alignment, default_alignment);
ov::Tensor* tensor = new ov::Tensor(remote_ctx_.create_host_tensor(ov::element::Type_t::u8,
{ alloc_size }));
uintptr_t data_ptr = reinterpret_cast<uintptr_t>(tensor->data());

ov::Tensor** ptr = reinterpret_cast<ov::Tensor**>(align_up(data_ptr + sizeof(ov::Tensor*), default_alignment));
ptr[-1] = tensor;

return reinterpret_cast<void*>(ptr);

} catch (const ov::Exception& e) {
ORT_THROW(std::string("Alloc failed: ") + e.what());
}
return nullptr;
}

void OVRTAllocator::Free(void* p) {
try {
ov::Tensor** ptr = reinterpret_cast<ov::Tensor**>(p);
delete ptr[-1];
} catch (const ov::Exception& e) {
ORT_THROW(std::string("Free failed: ") + e.what());
}
}

} // namespace onnxruntime
#endif
25 changes: 25 additions & 0 deletions onnxruntime/core/providers/openvino/ov_allocator.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
// Copyright (C) Intel Corporation
Comment thread Fixed
Comment thread Fixed

Check warning

Code scanning / lintrunner

CLANGFORMAT/format

See https://clang.llvm.org/docs/ClangFormat.html. Run `lintrunner -a` to apply this patch.
// Licensed under the MIT License
#ifdef USE_DEVICE_MEMORY
#pragma once

#include "core/common/inlined_containers.h"
#include "core/framework/allocator.h"
#include "openvino/runtime/remote_context.hpp"


namespace onnxruntime {

class OVRTAllocator : public IAllocator {
public:
OVRTAllocator(ov::Core &core, OrtDevice::DeviceType device_type, OrtDevice::DeviceId device_id, const char* name);
void* Alloc(size_t size) override;
void Free(void* p) override;

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Can be called inside destructor like ~OVRTAllocator() { Free(ptr); } This will manage the destruction once you step out of its local scope while using it automatically


private:
ov::Core &core_;
ov::RemoteContext remote_ctx_;
};

} // namespace onnxruntime
#endif
Loading