From fad590a05908406d50bd55e8f653d61c9ebdc524 Mon Sep 17 00:00:00 2001 From: Chi Lo <54722500+chilo-ms@users.noreply.github.com> Date: Thu, 17 Feb 2022 10:30:03 -0800 Subject: [PATCH] Enhance TRT EP unit tests (#10493) * Re-write tensorrt ep cache test * refactor the code * refactor * move stdc++fs flag to CMakeLists.txt --- cmake/CMakeLists.txt | 10 + cmake/onnxruntime_unittests.cmake | 1 + .../tensorrt/tensorrt_execution_provider.cc | 152 +---------- .../tensorrt_execution_provider_utils.h | 197 ++++++++++++++ .../providers/tensorrt/tensorrt_basic_test.cc | 250 ++++++++++++++---- 5 files changed, 410 insertions(+), 200 deletions(-) create mode 100644 onnxruntime/core/providers/tensorrt/tensorrt_execution_provider_utils.h diff --git a/cmake/CMakeLists.txt b/cmake/CMakeLists.txt index c6353a01dd..5d04280368 100644 --- a/cmake/CMakeLists.txt +++ b/cmake/CMakeLists.txt @@ -1702,6 +1702,16 @@ if (onnxruntime_USE_CUDA) endif() endif() +if (onnxruntime_USE_TENSORRT) + # needs to link with stdc++fs in Linux + if (UNIX) + if (NOT APPLE) + set(FS_STDLIB stdc++fs) + endif() + endif() + list(APPEND onnxruntime_EXTERNAL_LIBRARIES ${FS_STDLIB}) +endif() + if (onnxruntime_USE_MIGRAPHX) if (WIN32) message(FATAL_ERROR "MIGraphX does not support build in Windows!") diff --git a/cmake/onnxruntime_unittests.cmake b/cmake/onnxruntime_unittests.cmake index e181acc765..03bd6234c9 100644 --- a/cmake/onnxruntime_unittests.cmake +++ b/cmake/onnxruntime_unittests.cmake @@ -519,6 +519,7 @@ set(onnxruntime_test_providers_libs if(onnxruntime_USE_TENSORRT) list(APPEND onnxruntime_test_framework_src_patterns ${TEST_SRC_DIR}/providers/tensorrt/*) + list(APPEND onnxruntime_test_framework_src_patterns "${ONNXRUNTIME_ROOT}/core/providers/tensorrt/tensorrt_execution_provider_utils.h") list(APPEND onnxruntime_test_framework_libs onnxruntime_providers_tensorrt) list(APPEND onnxruntime_test_providers_dependencies onnxruntime_providers_tensorrt onnxruntime_providers_shared) endif() diff --git a/onnxruntime/core/providers/tensorrt/tensorrt_execution_provider.cc b/onnxruntime/core/providers/tensorrt/tensorrt_execution_provider.cc index 7deec5ca6d..b9579f69db 100644 --- a/onnxruntime/core/providers/tensorrt/tensorrt_execution_provider.cc +++ b/onnxruntime/core/providers/tensorrt/tensorrt_execution_provider.cc @@ -8,19 +8,17 @@ #include "core/session/onnxruntime_cxx_api.h" #include "core/common/safeint.h" #include "tensorrt_execution_provider.h" +#include "tensorrt_execution_provider_utils.h" #include "core/providers/cuda/shared_inc/cuda_call.h" #include "core/providers/cuda/math/unary_elementwise_ops_impl.h" #include "core/providers/cuda/gpu_data_transfer.h" #include "cuda_runtime_api.h" #include "gsl/gsl" -#include #include #include #include #include #include -#include "flatbuffers/idl.h" -#include "ort_trt_int8_cal_table.fbs.h" #ifdef _WIN32 #include @@ -41,88 +39,7 @@ using namespace ONNX_NAMESPACE; using namespace ::onnxruntime::logging; -namespace fs = std::experimental::filesystem; namespace { -std::string GetCachePath(const std::string& root, const std::string& name) { - if (root.empty()) { - return name; - } else { - fs::path path = root; - path.append(name); - return path.string(); - } -} - -float ConvertSinglePrecisionIEEE754ToFloat(unsigned long input) { - int s = (input >> 31) & 0x01; - int e = ((input & 0x7f800000) >> 23) - 127; - int p = -1; - double m = 0.0; - for (int i = 0; i < 23; ++i) { - m += ((input >> (23 - i - 1)) & 0x01) * pow(2.0, p--); - } - return static_cast((s ? -1 : 1) * pow(2.0, e) * (m + 1.0)); -} - -/* -* Seralize engine profile -* The profile contains min/max shape ranges of dynamic shape dimensions of each input tensor -* For example, assume tensor_a has two dynamic shape dimensions: dim_0 and dim_2, and tensor_b -* has one dynamic shape dimension: dim_1. The data in profile will be, -* key: tensor_a, value: dim_0 min_shape max_shape dim_2 min_shape max_shape -* key: tensor_b, value: dim_1 min_shape max_shape -*/ -void SerializeProfile(const std::string& file_name, std::unordered_map>>& shape_ranges) { - // Serialize profile - flexbuffers::Builder builder; - auto profile_start = builder.StartMap(); - for (auto outer_it = shape_ranges.begin(); outer_it != shape_ranges.end(); ++outer_it) { - builder.TypedVector(outer_it->first.c_str(), [&] { - for (auto inner_it = outer_it->second.begin(); inner_it != outer_it->second.end(); ++inner_it) { - builder.Int(inner_it->first); - builder.Int(inner_it->second.first); - builder.Int(inner_it->second.second); - } - }); - } - builder.EndMap(profile_start); - builder.Finish(); - - // Save flexbuffer - std::ofstream file(file_name, std::ios::binary | std::ios::out); - auto buf = builder.GetBuffer(); - size_t size = builder.GetSize(); - file.write(reinterpret_cast(&buf[0]), size); - file.close(); -} - -// Deserialize engine profile -std::unordered_map>> DeserializeProfile(std::ifstream& infile) { - // Load flexbuffer - infile.seekg(0, std::ios::end); - size_t length = infile.tellg(); - infile.seekg(0, std::ios::beg); - std::unique_ptr data{new char[length]}; - infile.read((char*)data.get(), length); - infile.close(); - - // Deserialize profile - std::unordered_map>> shape_ranges; - auto tensors_range_entries = flexbuffers::GetRoot((const uint8_t*)data.get(), length).AsMap(); - auto keys = tensors_range_entries.Keys(); - auto values = tensors_range_entries.Values(); - for (size_t i = 0, end = keys.size(); i < end; ++i) { - auto dim_range_vectors = values[i].AsTypedVector(); - std::unordered_map> inner_map; - for (size_t j = 0, end = dim_range_vectors.size() / 3; j < end; ++j) { - size_t idx = 3 * j; - inner_map[dim_range_vectors[idx].AsInt64()] = std::make_pair(dim_range_vectors[idx + 1].AsInt64(), dim_range_vectors[idx + 2].AsInt64()); - } - shape_ranges[keys[i].AsString().c_str()] = inner_map; - } - return shape_ranges; -} - // Check if cycle exists in the graph after partitioning bool FindCycleHelper(size_t i, const std::list* adjacency_map, bool visited[], bool* st, std::vector& cycles) { if (!visited[i]) { @@ -142,73 +59,6 @@ bool FindCycleHelper(size_t i, const std::list* adjacency_map, bool visi return false; } -/* -* Read calibration table for INT8 quantization -* Two kind of calibration tables are supported, -* 1. ORT generated calibration table -* The table is pre-serialized by flatbuffers. -* Each entry in the table is a key-value pair, -* key: tensor name, value: maximum absolute value in floating point -* For example, -* data_0 2.008338 -* ... -* 2. Native TensorRT generated calibration table -* Data format is defined by TensorRT as, -* tensor name : scale in 32-bit single precision IEEE754 format -* For example, -* TRT-7103-EntropyCalibration2 -* data_0: 4000889d -* ... -*/ -bool ReadDynamicRange(const std::string file_name, const bool is_trt_calibration_table, std::unordered_map& dynamic_range_map) { - std::ifstream infile(file_name, std::ios::binary | std::ios::in); - if (!infile) { - return false; - } - - if (is_trt_calibration_table) { - // Native TensorRT generated calibration table - std::string line; - char delim = ':'; - if (std::getline(infile, line)) { - std::istringstream first_line(line); - std::string version; - std::getline(first_line, version, delim); - std::size_t found = version.find("TRT-"); - if (found != std::string::npos) { - while (std::getline(infile, line)) { - std::istringstream in_line(line); - std::string str; - std::getline(in_line, str, delim); - std::string tensor_name = str; - std::getline(in_line, str, delim); - unsigned long scale_int = std::strtoul(str.c_str(), nullptr, 16); - float scale_float = ConvertSinglePrecisionIEEE754ToFloat(scale_int); - float dynamic_range = scale_float * 127.0f; - dynamic_range_map[tensor_name] = dynamic_range; - } - } else { - throw std::runtime_error("This is not a TensorRT generated calibration table " + file_name); - } - } - } else { - // ORT generated calibration table - infile.seekg(0, std::ios::end); - size_t length = infile.tellg(); - infile.seekg(0, std::ios::beg); - std::unique_ptr data{new char[length]}; - infile.read((char*)data.get(), length); - infile.close(); - auto flat_table = flatbuffers::GetRoot((const uint8_t*)data.get()); - auto flat_dict = flat_table->dict(); - for (size_t i = 0, end = flat_dict->size(); i < end; ++i) { - flatbuffers::uoffset_t idx = static_cast(i); - dynamic_range_map[flat_dict->Get(idx)->key()->str()] = std::stof(flat_dict->Get(idx)->value()->str()); - } - } - return true; -} - bool SetDynamicRange(nvinfer1::INetworkDefinition& network, std::unordered_map& dynamic_range_map) { // Set dynamic range for input tensors for (int i = 0; i < network.getNbInputs(); ++i) { diff --git a/onnxruntime/core/providers/tensorrt/tensorrt_execution_provider_utils.h b/onnxruntime/core/providers/tensorrt/tensorrt_execution_provider_utils.h new file mode 100644 index 0000000000..98d2055acc --- /dev/null +++ b/onnxruntime/core/providers/tensorrt/tensorrt_execution_provider_utils.h @@ -0,0 +1,197 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. +// Licensed under the MIT License. + +#include +#include +#include +#include +#include +#include "flatbuffers/idl.h" +#include "ort_trt_int8_cal_table.fbs.h" + +namespace fs = std::experimental::filesystem; + +namespace onnxruntime { + +float ConvertSinglePrecisionIEEE754ToFloat(unsigned long input) { + int s = (input >> 31) & 0x01; + int e = ((input & 0x7f800000) >> 23) - 127; + int p = -1; + double m = 0.0; + for (int i = 0; i < 23; ++i) { + m += ((input >> (23 - i - 1)) & 0x01) * pow(2.0, p--); + } + return static_cast((s ? -1 : 1) * pow(2.0, e) * (m + 1.0)); +} + +/* +* Read calibration table for INT8 quantization +* Two kind of calibration tables are supported, +* 1. ORT generated calibration table +* The table is pre-serialized by flatbuffers. +* Each entry in the table is a key-value pair, +* key: tensor name, value: maximum absolute value in floating point +* For example, +* data_0 2.008338 +* ... +* 2. Native TensorRT generated calibration table +* Data format is defined by TensorRT as, +* tensor name : scale in 32-bit single precision IEEE754 format +* For example, +* TRT-7103-EntropyCalibration2 +* data_0: 4000889d +* ... +*/ +bool ReadDynamicRange(const std::string file_name, const bool is_trt_calibration_table, std::unordered_map& dynamic_range_map) { + std::ifstream infile(file_name, std::ios::binary | std::ios::in); + if (!infile) { + return false; + } + + if (is_trt_calibration_table) { + // Native TensorRT generated calibration table + std::string line; + char delim = ':'; + if (std::getline(infile, line)) { + std::istringstream first_line(line); + std::string version; + std::getline(first_line, version, delim); + std::size_t found = version.find("TRT-"); + if (found != std::string::npos) { + while (std::getline(infile, line)) { + std::istringstream in_line(line); + std::string str; + std::getline(in_line, str, delim); + std::string tensor_name = str; + std::getline(in_line, str, delim); + unsigned long scale_int = std::strtoul(str.c_str(), nullptr, 16); + float scale_float = ConvertSinglePrecisionIEEE754ToFloat(scale_int); + float dynamic_range = scale_float * 127.0f; + dynamic_range_map[tensor_name] = dynamic_range; + } + } else { + throw std::runtime_error("This is not a TensorRT generated calibration table " + file_name); + } + } + } else { + // ORT generated calibration table + infile.seekg(0, std::ios::end); + size_t length = infile.tellg(); + infile.seekg(0, std::ios::beg); + std::unique_ptr data{new char[length]}; + infile.read((char*)data.get(), length); + infile.close(); + auto flat_table = flatbuffers::GetRoot((const uint8_t*)data.get()); + auto flat_dict = flat_table->dict(); + for (size_t i = 0, end = flat_dict->size(); i < end; ++i) { + flatbuffers::uoffset_t idx = static_cast(i); + dynamic_range_map[flat_dict->Get(idx)->key()->str()] = std::stof(flat_dict->Get(idx)->value()->str()); + } + } + return true; +} + +/* +* Seralize engine profile +* The profile contains min/max shape ranges of dynamic shape dimensions of each input tensor +* For example, assume tensor_a has two dynamic shape dimensions: dim_0 and dim_2, and tensor_b +* has one dynamic shape dimension: dim_1. The data in profile will be, +* key: tensor_a, value: dim_0 min_shape max_shape dim_2 min_shape max_shape +* key: tensor_b, value: dim_1 min_shape max_shape +*/ +void SerializeProfile(const std::string& file_name, std::unordered_map>>& shape_ranges) { + // Serialize profile + flexbuffers::Builder builder; + auto profile_start = builder.StartMap(); + for (auto outer_it = shape_ranges.begin(); outer_it != shape_ranges.end(); ++outer_it) { + builder.TypedVector(outer_it->first.c_str(), [&] { + for (auto inner_it = outer_it->second.begin(); inner_it != outer_it->second.end(); ++inner_it) { + builder.Int(inner_it->first); + builder.Int(inner_it->second.first); + builder.Int(inner_it->second.second); + } + }); + } + builder.EndMap(profile_start); + builder.Finish(); + + // Save flexbuffer + std::ofstream file(file_name, std::ios::binary | std::ios::out); + auto buf = builder.GetBuffer(); + size_t size = builder.GetSize(); + file.write(reinterpret_cast(&buf[0]), size); + file.close(); +} + +// Deserialize engine profile +std::unordered_map>> DeserializeProfile(std::ifstream& infile) { + // Load flexbuffer + infile.seekg(0, std::ios::end); + size_t length = infile.tellg(); + infile.seekg(0, std::ios::beg); + std::unique_ptr data{new char[length]}; + infile.read((char*)data.get(), length); + infile.close(); + + // Deserialize profile + std::unordered_map>> shape_ranges; + auto tensors_range_entries = flexbuffers::GetRoot((const uint8_t*)data.get(), length).AsMap(); + auto keys = tensors_range_entries.Keys(); + auto values = tensors_range_entries.Values(); + for (size_t i = 0, end = keys.size(); i < end; ++i) { + auto dim_range_vectors = values[i].AsTypedVector(); + std::unordered_map> inner_map; + for (size_t j = 0, end = dim_range_vectors.size() / 3; j < end; ++j) { + size_t idx = 3 * j; + inner_map[dim_range_vectors[idx].AsInt64()] = std::make_pair(dim_range_vectors[idx + 1].AsInt64(), dim_range_vectors[idx + 2].AsInt64()); + } + shape_ranges[keys[i].AsString().c_str()] = inner_map; + } + return shape_ranges; +} + +/* + * Get cache by name + * + */ +std::string GetCachePath(const std::string& root, const std::string& name) { + if (root.empty()) { + return name; + } else { + fs::path path = root; + path.append(name); + return path.string(); + } +} + +/* + * Get cache by type + * + * \param root root path of the cache + * \param file_extension It could be ".engine", ".profile" or ".timing" +*/ +std::vector GetCachesByType(const std::string& root, std::string file_extension) { + std::vector cache_files; + for (const auto & entry : fs::directory_iterator(root)) { + if (fs::path(file_extension) == fs::path(entry).extension()) { + cache_files.push_back(fs::path(entry)); + } + } + return cache_files; +} + +bool IsCacheExistedByType(const std::string& root, std::string file_extension) { + auto cache_files = GetCachesByType(root, file_extension); + if (cache_files.size() == 0) { + return false; + } + return true; +} + +void RemoveCachesByType(const std::string& root, std::string file_extension) { + auto cache_files = GetCachesByType(root, file_extension); + for (const auto & entry : cache_files) { + fs::remove(entry); + } +} +} diff --git a/onnxruntime/test/providers/tensorrt/tensorrt_basic_test.cc b/onnxruntime/test/providers/tensorrt/tensorrt_basic_test.cc index c24401a1b8..597cb6c290 100644 --- a/onnxruntime/test/providers/tensorrt/tensorrt_basic_test.cc +++ b/onnxruntime/test/providers/tensorrt/tensorrt_basic_test.cc @@ -7,6 +7,9 @@ #include "gtest/gtest.h" #include "test/util/include/default_providers.h" #include "test/util/include/scoped_env_vars.h" +#include "core/providers/tensorrt/tensorrt_provider_options.h" +#include "core/providers/tensorrt/tensorrt_execution_provider_utils.h" +#include using namespace std; using namespace ONNX_NAMESPACE; @@ -15,6 +18,8 @@ using namespace ::onnxruntime::logging; namespace onnxruntime { namespace test { +class TensorrtExecutionProviderCacheTest: public testing::TestWithParam {}; + template void VerifyOutputs(const std::vector& fetches, const std::vector& expected_dims, const std::vector& expected_values) { @@ -26,11 +31,28 @@ void VerifyOutputs(const std::vector& fetches, const std::vector dims) { + onnxruntime::Model model(graph_name, false, DefaultLoggingManager().DefaultLogger()); auto& graph = model.MainGraph(); std::vector inputs; std::vector outputs; @@ -38,9 +60,10 @@ TEST(TensorrtExecutionProviderTest, EngineCachingTest) { // FLOAT tensor ONNX_NAMESPACE::TypeProto float_tensor; float_tensor.mutable_tensor_type()->set_elem_type(ONNX_NAMESPACE::TensorProto_DataType_FLOAT); - float_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); - float_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_param("sym1"); - float_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_param("sym2"); + + for (auto dim: dims) { + float_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(dim); + } auto& input_arg_1 = graph.GetOrCreateNodeArg("X", &float_tensor); auto& input_arg_2 = graph.GetOrCreateNodeArg("Y", &float_tensor); @@ -61,11 +84,31 @@ TEST(TensorrtExecutionProviderTest, EngineCachingTest) { auto status = graph.Resolve(); ASSERT_TRUE(status.IsOK()); - std::string model_file_name = "trt_execution_provider_enginecaching_test.onnx"; - status = onnxruntime::Model::Save(model, model_file_name); + status = onnxruntime::Model::Save(model, model_name); +} + +TEST_P(TensorrtExecutionProviderCacheTest, Run) { + // GetParam() returns the parameter of following format: + // ##cache type##_##input shape type## + std::string param = GetParam(); + size_t pos = param.find("_"); + std::string input_type = param.substr(pos + 1); + ASSERT_NE(pos, std::string::npos); + std::string cache_type = ToUTF8String(param.substr(0, pos)); + + std::string model_name = "trt_execution_provider_" + cache_type + "caching_test_" + input_type + ".onnx"; + std::vector dims; + if (input_type.compare("dynamic") == 0) { + dims = {1, -1, -1}; //dynamic shape input + } + else { + dims = {1, 3, 2}; + } + + CreateBaseModel(model_name, cache_type + "cachingtest", dims); SessionOptions so; - so.session_logid = "TensorrtExecutionProviderTest.EngineCachingTest"; + so.session_logid = "TensorrtExecutionProvider" + cache_type + "cacheTest"; RunOptions run_options; run_options.run_tag = so.session_logid; InferenceSession session_object{so, GetEnvironment()}; @@ -73,12 +116,6 @@ TEST(TensorrtExecutionProviderTest, EngineCachingTest) { auto cuda_provider = DefaultCudaExecutionProvider(); cuda_provider->RegisterAllocator(allocator_manager); auto cpu_allocator = cuda_provider->GetAllocator(0, OrtMemTypeCPU); - // First run with input shape {1, 3, 2} - // TRT engine and profile will be created and cached - // Data in profile, - // X: 1, 3, 3, 2, 2, 2 - // Y: 1, 3, 3, 2, 2, 2 - // Z: 1, 3, 3, 2, 2, 2 std::vector dims_mul_x = {1, 3, 2}; std::vector values_mul_x = {1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 6.0f}; OrtValue ml_value_x; @@ -101,45 +138,160 @@ TEST(TensorrtExecutionProviderTest, EngineCachingTest) { std::vector expected_dims_mul_m = {1, 3, 2}; std::vector expected_values_mul_m = {3.0f, 6.0f, 9.0f, 12.0f, 15.0f, 18.0f}; - std::unique_ptr execution_provider = DefaultTensorrtExecutionProvider(); - EXPECT_TRUE(session_object.RegisterExecutionProvider(std::move(execution_provider)).IsOK()); - status = session_object.Load(model_file_name); - ASSERT_TRUE(status.IsOK()); - status = session_object.Initialize(); - ASSERT_TRUE(status.IsOK()); + OrtTensorRTProviderOptionsV2 params{ + 0, + 0, + nullptr, + 1000, + 1, + 1 << 30, + 0, + 0, + nullptr, + 0, + 0, + 0, + 0, + 0, + nullptr, + 0, + nullptr, + 0}; - // Now run - status = session_object.Run(run_options, feeds, output_names, &fetches); - ASSERT_TRUE(status.IsOK()); - VerifyOutputs(fetches, expected_dims_mul_m, expected_values_mul_m); + if (cache_type.compare("engine") == 0) { - // Second run with input shape {1, 1, 6} - // TRT engine and profile will be updated - // Data in profile, - // X: 1, 1, 3, 2, 2, 6 - // Y: 1, 1, 3, 2, 2, 6 - // Z: 1, 1, 3, 2, 2, 6 - dims_mul_x = {1, 1, 6}; - CreateMLValue(cpu_allocator, dims_mul_x, values_mul_x, &ml_value_x); - CreateMLValue(cpu_allocator, dims_mul_x, values_mul_x, &ml_value_y); - CreateMLValue(cpu_allocator, dims_mul_x, values_mul_x, &ml_value_z); - feeds.clear(); - feeds.insert(std::make_pair("X", ml_value_x)); - feeds.insert(std::make_pair("Y", ml_value_y)); - feeds.insert(std::make_pair("Z", ml_value_z)); + /* Following code block tests the functionality of engine and optimization profile of ORT TRT, including: + * - engine cache serialization/de-serialization + * - profile cache serialization/de-serialization + * - engine/profile cache should be updated when the input shape changes + * - min/max shape ranges of dynamic shape dimensions saved in profile cache + * - read corrupted profile cache #TODO + * + */ - // prepare outputs - fetches.clear(); + params.trt_engine_cache_enable = 1; + std::unique_ptr execution_provider = TensorrtExecutionProviderWithOptions(¶ms); + EXPECT_TRUE(session_object.RegisterExecutionProvider(std::move(execution_provider)).IsOK()); + auto status = session_object.Load(model_name); + ASSERT_TRUE(status.IsOK()); + status = session_object.Initialize(); + ASSERT_TRUE(status.IsOK()); - // prepare expected inputs and outputs - expected_dims_mul_m = {1, 1, 6}; + // run inference + // TRT engine will be created and cached + // TRT profile will be created and cached only for dynamic input shape + // Data in profile, + // X: 1, 3, 3, 2, 2, 2 + // Y: 1, 3, 3, 2, 2, 2 + // Z: 1, 3, 3, 2, 2, 2 + status = session_object.Run(run_options, feeds, output_names, &fetches); + ASSERT_TRUE(status.IsOK()); + VerifyOutputs(fetches, expected_dims_mul_m, expected_values_mul_m); + ASSERT_TRUE(IsCacheExistedByType("./", ".engine")); - // Now run - status = session_object.Run(run_options, feeds, output_names, &fetches); - ASSERT_TRUE(status.IsOK()); - VerifyOutputs(fetches, expected_dims_mul_m, expected_values_mul_m); + std::vector profile_files; + + // profile cache only being generated for dynamic input shape + if (input_type.compare("static") == 0) { + ASSERT_TRUE(!IsCacheExistedByType("./", ".profile")); + } else { + ASSERT_TRUE(IsCacheExistedByType("./", ".profile")); + + profile_files = GetCachesByType("./", ".profile"); + ASSERT_EQ(profile_files.size(), 1); + std::ifstream profile_file(profile_files[0], std::ios::binary | std::ios::in); + auto shape_ranges = DeserializeProfile(profile_file); + + // check min/max shape ranges of dynamic shape dimensions + for(auto it = shape_ranges.cbegin(); it != shape_ranges.cend(); ++it) { + auto ranges = it->second; + for (auto it2 = ranges.cbegin(); it2 != ranges.cend(); ++it2) { + if (it2->first == 1) { + ASSERT_EQ(it2->second.first, 3); + ASSERT_EQ(it2->second.second, 3); + } else if (it2->first == 2) { + ASSERT_EQ(it2->second.first, 2); + ASSERT_EQ(it2->second.second, 2); + } + } + } + } + + // another inference run with input shape {1, 1, 6} + // TRT engine and profile will be updated + // Data in profile, + // X: 1, 1, 3, 2, 2, 6 + // Y: 1, 1, 3, 2, 2, 6 + // Z: 1, 1, 3, 2, 2, 6 + dims_mul_x = {1, 1, 6}; + CreateMLValue(cpu_allocator, dims_mul_x, values_mul_x, &ml_value_x); + CreateMLValue(cpu_allocator, dims_mul_x, values_mul_x, &ml_value_y); + CreateMLValue(cpu_allocator, dims_mul_x, values_mul_x, &ml_value_z); + feeds.clear(); + feeds.insert(std::make_pair("X", ml_value_x)); + feeds.insert(std::make_pair("Y", ml_value_y)); + feeds.insert(std::make_pair("Z", ml_value_z)); + + // prepare outputs + fetches.clear(); + + // prepare expected inputs and outputs + expected_dims_mul_m = {1, 1, 6}; + + status = session_object.Run(run_options, feeds, output_names, &fetches); + + if (input_type.compare("static") == 0) { + // Can't run inference since input shape changes but the engine is built with static input + ASSERT_FALSE(status.IsOK()); + } else { + ASSERT_TRUE(status.IsOK()); + VerifyOutputs(fetches, expected_dims_mul_m, expected_values_mul_m); + + profile_files = GetCachesByType("./", ".profile"); + ASSERT_EQ(profile_files.size(), 1); + std::ifstream profile_file2(profile_files[0], std::ios::binary | std::ios::in); + auto shape_ranges2 = DeserializeProfile(profile_file2); + + // check min/max shape ranges of dynamic shape dimensions + for(auto it = shape_ranges2.cbegin(); it != shape_ranges2.cend(); ++it) { + auto ranges = it->second; + for (auto it2 = ranges.cbegin(); it2 != ranges.cend(); ++it2) { + if (it2->first == 1) { + ASSERT_EQ(it2->second.first, 1); + ASSERT_EQ(it2->second.second, 3); + } else if (it2->first == 2) { + ASSERT_EQ(it2->second.first, 2); + ASSERT_EQ(it2->second.second, 6); + } + } + } + } + } else if (cache_type.compare("timing") == 0) { + // add test code here + } + + // clean up caches + RemoveCachesByType("./", ".engine"); + RemoveCachesByType("./", ".profile"); } +/* + * The TensorrtExecutionProviderCacheTest aims to test the functionality of all the engine/profile/timing caches of ORT TRT. + * It uses value-parameterized test and the parameter in the test is a composite parameter which has following format: + * ##cache type##_##input shape type## + * - cache type (could be engine cache or timing cache. Note: profile cache will be tested along with engine cache) + * - input shape type (could be dynamic input shape or static input shape) + * + * We have following test parameters: + * - engine_static: engine cache enabled with non-dynamic input shape + * - engine_dynamic: engine cache enabled with dynamic input shape + * - timing_static: will be added + * - timing_dynamic: will be added + */ +INSTANTIATE_TEST_SUITE_P(TensorrtExecutionProviderCacheTests, TensorrtExecutionProviderCacheTest, testing::Values("engine_static", + "engine_dynamic"), + [](const ::testing::TestParamInfo& info) {return info.param;}); + TEST(TensorrtExecutionProviderTest, FunctionTest) { onnxruntime::Model model("functiontest", false, DefaultLoggingManager().DefaultLogger()); auto& graph = model.MainGraph();