Edgchen1/endian utils (#2181)

This commit is contained in:
edgchen1 2019-10-21 22:28:35 -07:00 committed by Changming Sun
parent 3507197599
commit 856c6cae0a
7 changed files with 260 additions and 104 deletions

View file

@ -3,14 +3,16 @@
#pragma once
#include <cstdint>
#include <cstring>
#include <string>
#include <stdint.h>
#include <type_traits>
#include <map>
#include <unordered_map>
#include "core/common/common.h"
#include "core/common/exceptions.h"
#include "core/framework/endian.h"
struct OrtValue;
@ -60,43 +62,29 @@ inline bool operator<(const MLFloat16& left, const MLFloat16& right) {
return left.val < right.val;
}
struct ort_endian {
union q {
uint16_t v_;
uint8_t b_[2];
constexpr explicit q(uint16_t v) noexcept : v_(v) {}
};
static constexpr bool is_little() {
return q(0x200).b_[0] == 0x0;
}
static constexpr bool is_big() {
return q(0x200).b_[0] == 0x2;
}
};
//BFloat16
struct BFloat16 {
uint16_t val{0};
explicit BFloat16() = default;
explicit BFloat16(uint16_t v) : val(v) {}
explicit BFloat16(float v) {
uint16_t* dst = reinterpret_cast<uint16_t*>(&v);
if (ort_endian::is_little()) {
val = dst[1];
if (endian::native == endian::little) {
std::memcpy(&val, reinterpret_cast<char*>(&v) + sizeof(uint16_t), sizeof(uint16_t));
} else {
val = dst[0];
std::memcpy(&val, &v, sizeof(uint16_t));
}
}
float ToFloat() const {
float result;
uint16_t* dst = reinterpret_cast<uint16_t*>(&result);
if (ort_endian::is_little()) {
dst[1] = val;
dst[0] = 0;
char* const first = reinterpret_cast<char*>(&result);
char* const second = first + sizeof(uint16_t);
if (endian::native == endian::little) {
std::memset(first, 0, sizeof(uint16_t));
std::memcpy(second, &val, sizeof(uint16_t));
} else {
dst[0] = val;
dst[1] = 0;
std::memcpy(first, &val, sizeof(uint16_t));
std::memset(second, 0, sizeof(uint16_t));
}
return result;
}

View file

@ -0,0 +1,27 @@
// Copyright (c) Microsoft Corporation. All rights reserved.
// Licensed under the MIT License.
#pragma once
namespace onnxruntime {
// the semantics of this enum should match std::endian from C++20
enum class endian {
#if defined(_WIN32)
little = 0,
big = 1,
native = little,
#elif defined(__GNUC__) || defined(__clang__)
little = __ORDER_LITTLE_ENDIAN__,
big = __ORDER_BIG_ENDIAN__,
native = __BYTE_ORDER__,
#else
#error onnxruntime::endian is not implemented in this environment.
#endif
};
static_assert(
endian::native == endian::little || endian::native == endian::big,
"Only little-endian or big-endian native byte orders are supported.");
} // namespace onnxruntime

View file

@ -0,0 +1,63 @@
// Copyright (c) Microsoft Corporation. All rights reserved.
// Licensed under the MIT License.
#include "core/framework/endian_utils.h"
#include <cassert>
#include <cstring>
#include "core/framework/endian.h"
namespace onnxruntime {
namespace utils {
namespace {
// analogous to std::reverse_copy
template <typename BidirIt, typename OutputIt>
OutputIt ReverseCopy(BidirIt first, BidirIt last, OutputIt d_first) {
while (last != first) {
--last;
*d_first = *last;
++d_first;
}
return d_first;
}
} // namespace
void SwapByteOrderCopy(
size_t element_size_in_bytes,
gsl::span<const char> source_bytes, gsl::span<char> destination_bytes) {
assert(element_size_in_bytes > 0);
assert(source_bytes.size_bytes() % element_size_in_bytes == 0);
assert(source_bytes.size_bytes() == destination_bytes.size_bytes());
for (size_t element_offset = 0, element_offset_end = source_bytes.size_bytes();
element_offset < element_offset_end;
element_offset += element_size_in_bytes) {
const auto source_element_bytes =
source_bytes.subspan(element_offset, element_size_in_bytes);
const auto dest_element_bytes =
destination_bytes.subspan(element_offset, element_size_in_bytes);
ReverseCopy(
source_element_bytes.data(),
source_element_bytes.data() + source_element_bytes.size_bytes(),
dest_element_bytes.data());
}
}
namespace detail {
void CopyLittleEndian(size_t element_size_in_bytes, gsl::span<const char> source_bytes, gsl::span<char> destination_bytes) {
if (endian::native == endian::little) {
std::memcpy(destination_bytes.data(), source_bytes.data(), source_bytes.size_bytes());
} else {
SwapByteOrderCopy(element_size_in_bytes, source_bytes, destination_bytes);
}
}
} // namespace detail
} // namespace utils
} // namespace onnxruntime

View file

@ -0,0 +1,76 @@
// Copyright (c) Microsoft Corporation. All rights reserved.
// Licensed under the MIT License.
#pragma once
#include <type_traits>
#include "gsl/gsl"
#include "core/common/status.h"
#include "core/common/common.h"
namespace onnxruntime {
namespace utils {
/**
* Swaps the byte order of elements in a buffer.
* This is a low-level funtion - please be sure to pass in valid arguments.
* In particular, source_bytes and destination_bytes should have the same size,
* which should be a multiple of element_size_in_bytes. element_size_in_bytes
* should also be greater than zero.
*
* @param element_size_in_bytes The size of an individual element, in bytes.
* @param source_bytes The source byte span.
* @param destination_bytes The destination byte span.
*/
void SwapByteOrderCopy(
size_t element_size_in_bytes, gsl::span<const char> source_bytes, gsl::span<char> destination_bytes);
namespace detail {
/**
* Copies between two buffers where one is little-endian and the other has
* native endian-ness.
*/
void CopyLittleEndian(
size_t element_size_in_bytes, gsl::span<const char> source_bytes, gsl::span<char> destination_bytes);
} // namespace detail
/**
* Reads from a little-endian source.
*/
template <typename T>
common::Status ReadLittleEndian(gsl::span<const char> source_bytes, gsl::span<T> destination) {
// std::is_trivially_copyable is not implemented in older versions of GCC
#if !defined(__GNUC__) || __GNUC__ >= 5
static_assert(std::is_trivially_copyable<T>::value, "T must be trivially copyable");
#endif
ORT_RETURN_IF_NOT(source_bytes.size_bytes() == destination.size_bytes(),
"source and destination buffer size mismatch");
const auto destination_bytes = gsl::make_span(
reinterpret_cast<char*>(destination.data()), destination.size_bytes());
detail::CopyLittleEndian(sizeof(T), source_bytes, destination_bytes);
return common::Status::OK();
}
/**
* Writes to a little-endian destination.
*/
template <typename T>
common::Status WriteLittleEndian(gsl::span<const T> source, gsl::span<char> destination_bytes) {
// std::is_trivially_copyable is not implemented in older versions of GCC
#if !defined(__GNUC__) || __GNUC__ >= 5
static_assert(std::is_trivially_copyable<T>::value, "T must be trivially copyable");
#endif
ORT_RETURN_IF_NOT(source.size_bytes() == destination_bytes.size_bytes(),
"source and destination buffer size mismatch");
const auto source_bytes = gsl::make_span(
reinterpret_cast<const char*>(source.data()), source.size_bytes());
detail::CopyLittleEndian(sizeof(T), source_bytes, destination_bytes);
return common::Status::OK();
}
} // namespace utils
} // namespace onnxruntime

View file

@ -10,6 +10,7 @@
#include "core/common/logging/logging.h"
#include "core/graph/onnx_protobuf.h"
#include "core/framework/endian_utils.h"
#include "core/framework/op_kernel.h"
#include "core/framework/tensor.h"
#include "core/framework/ort_value_pattern_planner.h"
@ -24,17 +25,6 @@ using namespace ::onnxruntime::common;
namespace {
#ifdef __GNUC__
constexpr inline bool IsLittleEndianOrder() noexcept { return __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__; }
#else
// On Windows and Mac, this function should always return true
GSL_SUPPRESS(type .1) // allow use of reinterpret_cast for this special case
inline bool IsLittleEndianOrder() noexcept {
static int n = 1;
return (*reinterpret_cast<char*>(&n) == 1);
}
#endif
std::vector<int64_t> GetTensorShapeFromTensorProto(const ONNX_NAMESPACE::TensorProto& tensor_proto) {
const auto& dims = tensor_proto.dims();
std::vector<int64_t> tensor_shape_vec(static_cast<size_t>(dims.size()));
@ -49,35 +39,19 @@ std::vector<int64_t> GetTensorShapeFromTensorProto(const ONNX_NAMESPACE::TensorP
template <typename T>
static Status UnpackTensorWithRawData(const void* raw_data, size_t raw_data_length, size_t expected_size,
/*out*/ T* p_data) {
// allow this low level routine to be somewhat unsafe. assuming it's thoroughly tested and valid
GSL_SUPPRESS(type) // type.1 reinterpret-cast; type.4 C-style casts; type.5 'T result;' is uninitialized;
GSL_SUPPRESS(bounds .1) // pointer arithmetic
GSL_SUPPRESS(f .23) // buff and temp_bytes never tested for nullness and could be gsl::not_null
{
size_t expected_size_in_bytes;
if (!onnxruntime::IAllocator::CalcMemSizeForArray(expected_size, sizeof(T), &expected_size_in_bytes)) {
return Status(onnxruntime::common::ONNXRUNTIME, onnxruntime::common::INVALID_ARGUMENT, "size overflow");
}
if (raw_data_length != expected_size_in_bytes)
return ORT_MAKE_STATUS(ONNXRUNTIME, INVALID_ARGUMENT,
"UnpackTensor: the pre-allocated size does not match the raw data size, expected ",
expected_size_in_bytes, ", got ", raw_data_length);
if (IsLittleEndianOrder()) {
memcpy(p_data, raw_data, raw_data_length);
} else {
const size_t type_size = sizeof(T);
const char* buff = reinterpret_cast<const char*>(raw_data);
for (size_t i = 0; i < raw_data_length; i += type_size, buff += type_size) {
T result;
const char* temp_bytes = reinterpret_cast<char*>(&result);
for (size_t j = 0; j < type_size; ++j) {
memcpy((void*)&temp_bytes[j], (void*)&buff[type_size - 1 - i], 1);
}
p_data[i] = result;
}
}
return Status::OK();
size_t expected_size_in_bytes;
if (!onnxruntime::IAllocator::CalcMemSizeForArray(expected_size, sizeof(T), &expected_size_in_bytes)) {
return Status(onnxruntime::common::ONNXRUNTIME, onnxruntime::common::INVALID_ARGUMENT, "size overflow");
}
if (raw_data_length != expected_size_in_bytes)
return ORT_MAKE_STATUS(ONNXRUNTIME, INVALID_ARGUMENT,
"UnpackTensor: the pre-allocated size does not match the raw data size, expected ",
expected_size_in_bytes, ", got ", raw_data_length);
const char* const raw_data_bytes = reinterpret_cast<const char*>(raw_data);
ORT_RETURN_IF_ERROR(onnxruntime::utils::ReadLittleEndian(
gsl::make_span(raw_data_bytes, raw_data_length), gsl::make_span(p_data, expected_size)));
return Status::OK();
}
} // namespace
@ -280,7 +254,7 @@ TensorShape GetTensorShapeFromTensorShapeProto(const ONNX_NAMESPACE::TensorShape
std::vector<int64_t> tensor_shape_vec(static_cast<size_t>(dims.size()));
for (int i = 0; i < dims.size(); ++i) {
tensor_shape_vec[i] = HasDimValue(dims[i]) ? dims[i].dim_value()
: -1; /* symbolic dimensions are represented as -1 in onnxruntime*/
: -1; /* symbolic dimensions are represented as -1 in onnxruntime*/
}
return TensorShape(std::move(tensor_shape_vec));
}
@ -401,7 +375,7 @@ Status TensorProtoToMLValue(const Env& env, const ORTCHAR_T* tensor_proto_path,
raw_data = tensor_proto.raw_data().data();
raw_data_len = tensor_proto.raw_data().size();
}
if (IsLittleEndianOrder() && raw_data != nullptr && deleter_for_file_data.d.f != nullptr) {
if (endian::native == endian::little && raw_data != nullptr && deleter_for_file_data.d.f != nullptr) {
tensor_data = const_cast<void*>(raw_data);
MoveOrtCallback(deleter_for_file_data.d, deleter);
} else {
@ -541,7 +515,7 @@ TensorProto::DataType GetTensorProtoType(const Tensor& tensor) {
ONNX_NAMESPACE::TensorProto TensorToTensorProto(const Tensor& tensor, const std::string& tensor_proto_name,
const ONNX_NAMESPACE::TypeProto& tensor_proto_type) {
// Given we are using the raw_data field in the protobuf, this will work only for little-endian format.
ORT_ENFORCE(IsLittleEndianOrder());
ORT_ENFORCE(endian::native == endian::little);
// Set name, dimensions, type, and data of the TensorProto.
ONNX_NAMESPACE::TensorProto tensor_proto;

View file

@ -0,0 +1,52 @@
// Copyright (c) Microsoft Corporation. All rights reserved.
// Licensed under the MIT License.
#include "core/framework/endian.h"
#include "core/framework/endian_utils.h"
#include <vector>
#include "gtest/gtest.h"
namespace onnxruntime {
namespace utils {
namespace test {
TEST(EndianTest, EndiannessDetection) {
const uint16_t test_value = 0x1234;
const char* test_value_first_byte = reinterpret_cast<const char*>(&test_value);
if (endian::native == endian::little) {
EXPECT_EQ(*test_value_first_byte, 0x34);
} else if (endian::native == endian::big) {
EXPECT_EQ(*test_value_first_byte, 0x12);
}
}
TEST(EndianTest, SwapByteOrderCopy) {
const auto src = std::vector<char>{
'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l'};
auto result = std::vector<char>(src.size());
{
SwapByteOrderCopy(3, gsl::make_span(src), gsl::make_span(result));
const auto expected = std::vector<char>{
'c', 'b', 'a',
'f', 'e', 'd',
'i', 'h', 'g',
'l', 'k', 'j'};
EXPECT_EQ(result, expected);
}
{
SwapByteOrderCopy(4, gsl::make_span(src), gsl::make_span(result));
const auto expected = std::vector<char>{
'd', 'c', 'b', 'a',
'h', 'g', 'f', 'e',
'l', 'k', 'j', 'i'};
EXPECT_EQ(result, expected);
}
}
} // namespace test
} // namespace utils
} // namespace onnxruntime

View file

@ -8,6 +8,7 @@
#include <limits>
#include <gsl/gsl>
#include "core/framework/data_types.h"
#include "core/framework/endian.h"
#include "core/framework/allocator.h"
#include "core/session/onnxruntime_cxx_api.h"
#include "core/graph/onnx_protobuf.h"
@ -22,16 +23,6 @@ struct OrtStatus {
namespace onnxruntime {
namespace test {
#ifdef __GNUC__
constexpr inline bool IsLittleEndianOrder() noexcept { return __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__; }
#else
// On Windows and Mac, this function should always return true
GSL_SUPPRESS(type .1) // allow use of reinterpret_cast for this special case
inline bool IsLittleEndianOrder() noexcept {
static int n = 1;
return (*reinterpret_cast<char*>(&n) == 1);
}
#endif
//From core common
inline void MakeStringInternal(std::ostringstream& /*ss*/) noexcept {
@ -78,34 +69,19 @@ std::vector<int64_t> GetTensorShapeFromTensorProto(const onnx::TensorProto& tens
template <typename T>
static void UnpackTensorWithRawData(const void* raw_data, size_t raw_data_length, size_t expected_size,
/*out*/ T* p_data) {
// allow this low level routine to be somewhat unsafe. assuming it's thoroughly tested and valid
GSL_SUPPRESS(type) // type.1 reinterpret-cast; type.4 C-style casts; type.5 'T result;' is uninitialized;
GSL_SUPPRESS(bounds .1) // pointer arithmetic
GSL_SUPPRESS(f .23) // buff and temp_bytes never tested for nullness and could be gsl::not_null
{
size_t expected_size_in_bytes;
if (!onnxruntime::IAllocator::CalcMemSizeForArray(expected_size, sizeof(T), &expected_size_in_bytes)) {
throw Ort::Exception("size overflow", OrtErrorCode::ORT_FAIL);
}
if (raw_data_length != expected_size_in_bytes)
throw Ort::Exception(MakeString("UnpackTensor: the pre-allocated size does not match the raw data size, expected ",
expected_size_in_bytes, ", got ", raw_data_length),
OrtErrorCode::ORT_FAIL);
if (IsLittleEndianOrder()) {
memcpy(p_data, raw_data, raw_data_length);
} else {
const size_t type_size = sizeof(T);
const char* buff = reinterpret_cast<const char*>(raw_data);
for (size_t i = 0; i < raw_data_length; i += type_size, buff += type_size) {
T result;
const char* temp_bytes = reinterpret_cast<char*>(&result);
for (size_t j = 0; j < type_size; ++j) {
memcpy((void*)&temp_bytes[j], (void*)&buff[type_size - 1 - i], 1);
}
p_data[i] = result;
}
}
size_t expected_size_in_bytes;
if (!onnxruntime::IAllocator::CalcMemSizeForArray(expected_size, sizeof(T), &expected_size_in_bytes)) {
throw Ort::Exception("size overflow", OrtErrorCode::ORT_FAIL);
}
if (raw_data_length != expected_size_in_bytes)
throw Ort::Exception(MakeString("UnpackTensor: the pre-allocated size does not match the raw data size, expected ",
expected_size_in_bytes, ", got ", raw_data_length),
OrtErrorCode::ORT_FAIL);
if (endian::native != endian::little) {
throw Ort::Exception("UnpackTensorWithRawData only handles little-endian native byte order for now.",
OrtErrorCode::ORT_NOT_IMPLEMENTED);
}
memcpy(p_data, raw_data, raw_data_length);
}
// This macro doesn't work for Float16/bool/string tensors