onnxruntime/winml/lib/Api.Image/CpuDetensorizer.h
raoanag 424107a82a
Merge main to WindowsAI (#18122)
### Description
Merge main to WindowsAI



### Motivation and Context
<!-- - Why is this change required? What problem does it solve?
- If it fixes an open issue, please link to the issue here. -->

---------

Signed-off-by: Nash <george.nash@intel.com>
Signed-off-by: Yiming Hu <yiming.hu@amd.com>
Signed-off-by: Liqun Fu <liqfu@microsoft.com>
Co-authored-by: Kaz Nishimura <kazssym@linuxfront.com>
Co-authored-by: Tianlei Wu <tlwu@microsoft.com>
Co-authored-by: Nat Kershaw (MSFT) <nakersha@microsoft.com>
Co-authored-by: Yulong Wang <7679871+fs-eire@users.noreply.github.com>
Co-authored-by: Changming Sun <chasun@microsoft.com>
Co-authored-by: zesongw <zesong.wang@intel.com>
Co-authored-by: Yi Zhang <zhanyi@microsoft.com>
Co-authored-by: Dmitri Smirnov <yuslepukhin@users.noreply.github.com>
Co-authored-by: Yifan Li <109183385+yf711@users.noreply.github.com>
Co-authored-by: simonjub <78098752+simonjub@users.noreply.github.com>
Co-authored-by: PeixuanZuo <94887879+PeixuanZuo@users.noreply.github.com>
Co-authored-by: Adrian Lizarraga <adlizarraga@microsoft.com>
Co-authored-by: Edward Chen <18449977+edgchen1@users.noreply.github.com>
Co-authored-by: Arthur Islamov <arthur@islamov.ai>
Co-authored-by: Jambay Kinley <jambaykinley@microsoft.com>
Co-authored-by: Justin Chu <justinchuby@users.noreply.github.com>
Co-authored-by: Wei-Sheng Chin <wschin@outlook.com>
Co-authored-by: Bowen Bao <bowbao@microsoft.com>
Co-authored-by: Hariharan Seshadri <shariharan91@gmail.com>
Co-authored-by: Numfor Tiapo <numsmt2@gmail.com>
Co-authored-by: Vincent Wang <wangwchpku@outlook.com>
Co-authored-by: Pranav Sharma <prs@microsoft.com>
Co-authored-by: George Nash <george.nash@intel.com>
Co-authored-by: Abhishek Jindal <abjindal@microsoft.com>
Co-authored-by: pengwa <pengwa@microsoft.com>
Co-authored-by: Yiming Hu <woinck@users.noreply.github.com>
Co-authored-by: Jiajia Qin <jiajia.qin@intel.com>
Co-authored-by: Lukas Berbuer <36054362+lukasberbuer@users.noreply.github.com>
Co-authored-by: Wanming Lin <wanming.lin@intel.com>
Co-authored-by: Xavier Dupré <xadupre@users.noreply.github.com>
Co-authored-by: aimilefth <60664743+aimilefth@users.noreply.github.com>
Co-authored-by: Baiju Meswani <bmeswani@microsoft.com>
Co-authored-by: Adam Pocock <adam.pocock@oracle.com>
Co-authored-by: Chi Lo <54722500+chilo-ms@users.noreply.github.com>
Co-authored-by: RandySheriffH <48490400+RandySheriffH@users.noreply.github.com>
Co-authored-by: Randy Shuai <rashuai@microsoft.com>
Co-authored-by: Vadym Stupakov <vadim.stupakov@gmail.com>
Co-authored-by: Jian Chen <cjian@microsoft.com>
Co-authored-by: Brian Lambert <98757707+brian-pieces@users.noreply.github.com>
Co-authored-by: Nicolò Lucchesi <nicolo.lucchesi@gmail.com>
Co-authored-by: liqun Fu <liqfu@microsoft.com>
Co-authored-by: trajep <trajepl@gmail.com>
Co-authored-by: Scott McKay <skottmckay@gmail.com>
Co-authored-by: Mustafa Ateş Uzun <mustafauzun0@gmail.com>
Co-authored-by: MistEO <mistereo@hotmail.com>
Co-authored-by: satyajandhyala <satya.k.jandhyala@gmail.com>
Co-authored-by: shaahji <96227573+shaahji@users.noreply.github.com>
Co-authored-by: Rachel Guo <35738743+YUNQIUGUO@users.noreply.github.com>
Co-authored-by: rachguo <rachguo@rachguos-Mini.attlocal.net>
Co-authored-by: Caroline Zhu <wolfivyaura@gmail.com>
Co-authored-by: Caroline Zhu <carolinezhu@microsoft.com>
Co-authored-by: Guenther Schmuelling <guschmue@microsoft.com>
Co-authored-by: xhcao <xinghua.cao@intel.com>
Co-authored-by: Ella Charlaix <80481427+echarlaix@users.noreply.github.com>
Co-authored-by: Xu Xing <xing.xu@intel.com>
Co-authored-by: Hector Li <hecli@microsoft.com>
Co-authored-by: Ye Wang <52801275+wangyems@users.noreply.github.com>
Co-authored-by: Your Name <you@example.com>
Co-authored-by: Benedikt Hilmes <benedikt.hilmes@rwth-aachen.de>
Co-authored-by: rachguo <rachguo@rachguos-Mac-mini.local>
Co-authored-by: George Wu <jywu@microsoft.com>
Co-authored-by: JiCheng <wejoncy@163.com>
Co-authored-by: Sheil Kumar <smk2007@gmail.com>
Co-authored-by: Sheil Kumar <sheilk@microsoft.com>
Co-authored-by: cloudhan <guangyunhan@microsoft.com>
Co-authored-by: kyoshisuki <143475866+kyoshisuki@users.noreply.github.com>
Co-authored-by: aciddelgado <139922440+aciddelgado@users.noreply.github.com>
Co-authored-by: tlwu@microsoft.com <tlwu@a100.crj0ad2y1kku1j4yxl4sj10o4e.gx.internal.cloudapp.net>
Co-authored-by: Maximilian Müller <44298237+gedoensmax@users.noreply.github.com>
Co-authored-by: Tang, Cheng <souptc@gmail.com>
Co-authored-by: Cheng Tang <chenta@microsoft.com@orttrainingdev9.d32nl1ml4oruzj4qz3bqlggovf.px.internal.cloudapp.net>
Co-authored-by: Cheng Tang <chenta@microsoft.com>
Co-authored-by: Jeff Daily <jeff.daily@amd.com>
Co-authored-by: cloudhan <cloudhan@outlook.com>
Co-authored-by: Yufeng Li <liyufeng1987@gmail.com>
Co-authored-by: Zhang Lei <zhang.huanning@hotmail.com>
Co-authored-by: Dwayne Robinson <fdwr@hotmail.com>
Co-authored-by: Zhipeng Han <zhipeng.han@outlook.com>
Co-authored-by: Thiago Crepaldi <thiago.crepaldi@microsoft.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
Co-authored-by: Patrice Vignola <vignola.patrice@gmail.com>
Co-authored-by: kunal-vaishnavi <115581922+kunal-vaishnavi@users.noreply.github.com>
Co-authored-by: snadampal <87143774+snadampal@users.noreply.github.com>
Co-authored-by: Sumit Agarwal <sumitagarwal330@gmail.com>
Co-authored-by: Ashwini Khade <askhade@microsoft.com>
Co-authored-by: Yang Gu <yang.gu@intel.com>
Co-authored-by: Cheng Tang <chenta@a100.crj0ad2y1kku1j4yxl4sj10o4e.gx.internal.cloudapp.net>
Co-authored-by: mindest <30493312+mindest@users.noreply.github.com>
Co-authored-by: Scott McKay <Scott.McKay@microsoft.com>
Co-authored-by: Xavier Dupre <xadupre@microsoft.com@orttrainingdev9.d32nl1ml4oruzj4qz3bqlggovf.px.internal.cloudapp.net>
Co-authored-by: guyang3532 <62738430+guyang3532@users.noreply.github.com>
Co-authored-by: Carson M <carson@pyke.io>
Co-authored-by: sophies927 <107952697+sophies927@users.noreply.github.com>
2023-10-27 17:08:01 -07:00

265 lines
9.6 KiB
C++

// Copyright (c) Microsoft Corporation. All rights reserved.
// Licensed under the MIT License.
#pragma once
#include "inc/ImageConversionTypes.h"
#include "inc/NominalRangeConverter.h"
namespace _winml {
class CpuDetensorizer {
public:
template <typename T>
static HRESULT Detensorize(
_In_ ImageTensorChannelType formatFrom,
_In_ ImageTensorChannelType formatTo,
_In_ winml::LearningModelPixelRange pixelRange,
_In_ T* pCPUTensor,
_In_ uint32_t bufferWidth,
_In_ uint32_t tensorHeight,
_In_ uint32_t tensorWidth,
_Inout_ BYTE* pData
) {
#pragma warning(push)
#pragma warning(disable : 26014 \
) // warning about possible out of bounds accesing pData, but input is checked for BGRA8 format, so uiCapacity should be in multiples of 4 \
// output is BGRA8: so blue at i, green is at i + 1, red is at i + 2
uint32_t bytesPerPixel = formatTo == kImageTensorChannelTypeGRAY8 ? 1 : 4;
// bufferWidth may have padding because of optimization, but bytesPerRow includes only the real tensor data. We need to jump
// over bufferWidth's extra padding
uint32_t bytesPerRow = tensorWidth * bytesPerPixel;
uint32_t end = bufferWidth * tensorHeight;
size_t tensorPlaneSize = tensorWidth * tensorHeight;
auto nominalRangeConverter = NominalRangeConverter(pixelRange);
if (formatFrom == formatTo && (formatFrom == kImageTensorChannelTypeBGR8 || formatFrom == kImageTensorChannelTypeRGB8)) {
for (uint32_t i = 0; i < tensorHeight; i++) {
BYTE* pPixel = pData;
InterleaveRowFloatToByte(
pCPUTensor + i * tensorWidth,
pCPUTensor + tensorPlaneSize + i * tensorWidth,
pCPUTensor + tensorPlaneSize * 2 + i * tensorWidth,
tensorWidth,
pPixel,
bytesPerPixel,
nominalRangeConverter
);
pData += bufferWidth;
}
} else if ((formatFrom == kImageTensorChannelTypeRGB8 && formatTo == kImageTensorChannelTypeBGR8) || (formatFrom == kImageTensorChannelTypeBGR8 && formatTo == kImageTensorChannelTypeRGB8)) {
for (uint32_t i = 0; i < tensorHeight; i++) {
BYTE* pPixel = pData;
InterleaveRowFloatToByte(
pCPUTensor + tensorPlaneSize * 2 + i * tensorWidth,
pCPUTensor + tensorPlaneSize + i * tensorWidth,
pCPUTensor + i * tensorWidth,
tensorWidth,
pPixel,
bytesPerPixel,
nominalRangeConverter
);
pData += bufferWidth;
}
} else if (formatFrom == kImageTensorChannelTypeGRAY8 && (formatTo == kImageTensorChannelTypeBGR8 || formatTo == kImageTensorChannelTypeRGB8)) {
// just replicate the gray data across each channel
for (uint32_t i = 0; i < end; i += bufferWidth) {
for (uint32_t j = i; j < i + bytesPerRow; j += 4) {
BYTE bGray = DetensorizeValue<T>(pCPUTensor, nominalRangeConverter);
pData[j] = bGray;
pData[j + 1] = bGray;
pData[j + 2] = bGray;
pData[j + 3] = 255;
pCPUTensor++;
}
}
} else if (formatFrom == kImageTensorChannelTypeGRAY8 && formatTo == kImageTensorChannelTypeGRAY8) {
for (uint32_t i = 0; i < end; i += bufferWidth) {
for (uint32_t j = i; j < i + bytesPerRow; j += 1) {
BYTE bGray = DetensorizeValue<T>(pCPUTensor, nominalRangeConverter);
pData[j] = bGray;
pCPUTensor++;
}
}
} else if (formatFrom == kImageTensorChannelTypeBGR8 && formatTo == kImageTensorChannelTypeGRAY8) {
for (uint32_t i = 0; i < end; i += bufferWidth) {
for (uint32_t j = i; j < i + bytesPerRow; j += 1) {
BYTE red, green, blue;
blue = DetensorizeValue(pCPUTensor, nominalRangeConverter);
green = DetensorizeValue(pCPUTensor + tensorPlaneSize, nominalRangeConverter);
red = DetensorizeValue(pCPUTensor + tensorPlaneSize * 2, nominalRangeConverter);
pData[j] = static_cast<BYTE>(0.2126f * red + 0.7152f * green + 0.0722f * blue);
pCPUTensor++;
}
}
} else if (formatFrom == kImageTensorChannelTypeRGB8 && formatTo == kImageTensorChannelTypeGRAY8) {
for (uint32_t i = 0; i < end; i += bufferWidth) {
for (uint32_t j = i; j < i + bytesPerRow; j += 1) {
BYTE red, green, blue;
red = DetensorizeValue(pCPUTensor, nominalRangeConverter);
green = DetensorizeValue(pCPUTensor + tensorPlaneSize, nominalRangeConverter);
blue = DetensorizeValue(pCPUTensor + tensorPlaneSize * 2, nominalRangeConverter);
pData[j] = static_cast<BYTE>(0.2126f * red + 0.7152f * green + 0.0722f * blue);
pCPUTensor++;
}
}
}
#pragma warning(pop)
else {
return E_INVALIDARG;
}
return S_OK;
}
private:
template <typename T>
static float ReadTensor(const T* pCPUTensor, const NominalRangeConverter& nominalRangeConverter) {
return nominalRangeConverter.Denormalize(*pCPUTensor);
}
// clang-format off
template <>
#if _MSVC_LANG < 202002L
static
#endif
float ReadTensor<DirectX::PackedVector::HALF>(
const DirectX::PackedVector::HALF* pCPUTensor, const NominalRangeConverter& nominalRangeConverter
) {
return nominalRangeConverter.Denormalize(DirectX::PackedVector::XMConvertHalfToFloat(*pCPUTensor));
}
template <typename T>
static BYTE DetensorizeValue(const T* pCPUTensor, const NominalRangeConverter& nominalRangeConverter) {
return static_cast<BYTE>(std::max(0.0f, std::min(255.0f, ReadTensor(pCPUTensor, nominalRangeConverter) + 0.5f)));
}
template <typename T>
static void InterleaveRowFloatToByte(
const T* xChannel,
const T* yChannel,
const T* zChannel,
uint32_t tensorWidth,
BYTE* pData,
uint32_t bytesPerPixel,
const NominalRangeConverter& nominalRangeConverter
) {
BYTE* pPixel = pData;
uint32_t tensorWidthRemaining = tensorWidth;
while (tensorWidthRemaining > 0) {
pPixel[0] = DetensorizeValue(xChannel, nominalRangeConverter);
pPixel[1] = DetensorizeValue(yChannel, nominalRangeConverter);
pPixel[2] = DetensorizeValue(zChannel, nominalRangeConverter);
pPixel[3] = 255;
pPixel += 4;
xChannel++;
yChannel++;
zChannel++;
tensorWidthRemaining--;
}
}
// clang-format off
#if defined(_M_AMD64) || defined(_M_IX86)
template <>
#if _MSVC_LANG < 202002L
static
#endif
void InterleaveRowFloatToByte(
const float* xChannel,
const float* yChannel,
const float* zChannel,
uint32_t tensorWidth,
BYTE* pData,
uint32_t bytesPerPixel,
const NominalRangeConverter& nominalRangeConverter
) {
BYTE* pPixel = pData;
uint32_t tensorWidthRemaining = tensorWidth;
__m128 maxv = _mm_set1_ps(255.0f);
__m128 zero = _mm_setzero_ps();
// Prep an alpha register with 8 bit - 255 alpha values
__m128i alpha = _mm_setzero_si128();
alpha = _mm_cmpeq_epi32(alpha, alpha);
alpha = _mm_srli_epi16(alpha, 8);
while (tensorWidthRemaining >= 8) {
// Load, saturate, and convert to ints, 8 - 32 bit floats from X channel
__m128i vXIntsLo = _mm_cvtps_epi32(_mm_min_ps(nominalRangeConverter.Denormalize(_mm_loadu_ps(xChannel)), maxv));
__m128i vXIntsHi =
_mm_cvtps_epi32(_mm_min_ps(nominalRangeConverter.Denormalize(_mm_loadu_ps(xChannel + 4)), maxv));
// Pack 32 bit ints into 16 bit ints
__m128i vXWords = _mm_packs_epi32(vXIntsLo, vXIntsHi);
// Load, saturate, and convert to ints, 8 - 32 bit floats from Y channel
__m128i vYIntsLo = _mm_cvtps_epi32(_mm_min_ps(nominalRangeConverter.Denormalize(_mm_loadu_ps(yChannel)), maxv));
__m128i vYIntsHi =
_mm_cvtps_epi32(_mm_min_ps(nominalRangeConverter.Denormalize(_mm_loadu_ps(yChannel + 4)), maxv));
// Pack 32 bit ints into 16 bit ints
__m128i vYWords = _mm_packs_epi32(vYIntsLo, vYIntsHi);
// Load, saturate, and convert to ints, 8 - 32 bit floats from Z channel
__m128i vZIntsLo = _mm_cvtps_epi32(_mm_min_ps(nominalRangeConverter.Denormalize(_mm_loadu_ps(zChannel)), maxv));
__m128i vZIntsHi =
_mm_cvtps_epi32(_mm_min_ps(nominalRangeConverter.Denormalize(_mm_loadu_ps(zChannel + 4)), maxv));
// Pack 32 bit ints into 16 bit ints
__m128i vZWords = _mm_packs_epi32(vZIntsLo, vZIntsHi);
// Pack 16 bit ints into 8 bit uints
__m128i vXZBytes = _mm_packus_epi16(vXWords, vZWords);
__m128i vYABytes = _mm_packus_epi16(vYWords, alpha);
// Interleave bytes into XY order
__m128i vXYBytesInterleaved = _mm_unpacklo_epi8(vXZBytes, vYABytes);
// Interleave bytes into ZA order
__m128i vZABytesInterleaved = _mm_unpackhi_epi8(vXZBytes, vYABytes);
// Interleave 16 bits to get XYZA XYZA ordering
__m128i vPixelBytesLo = _mm_unpacklo_epi16(vXYBytesInterleaved, vZABytesInterleaved);
__m128i vPixelBytesHi = _mm_unpackhi_epi16(vXYBytesInterleaved, vZABytesInterleaved);
// Write out bytes now in proper order
_mm_storeu_si128((__m128i*)pPixel, vPixelBytesLo);
_mm_storeu_si128((__m128i*)(pPixel + 16), vPixelBytesHi);
xChannel += 8;
yChannel += 8;
zChannel += 8;
pPixel += 8 * static_cast<uint64_t>(bytesPerPixel);
tensorWidthRemaining -= 8;
}
// Anything remaining deal with it one at a time
while (tensorWidthRemaining > 0) {
pPixel[0] = DetensorizeValue(xChannel, nominalRangeConverter);
pPixel[1] = DetensorizeValue(yChannel, nominalRangeConverter);
pPixel[2] = DetensorizeValue(zChannel, nominalRangeConverter);
pPixel[3] = 255;
pPixel += static_cast<uint64_t>(bytesPerPixel);
xChannel++;
yChannel++;
zChannel++;
tensorWidthRemaining--;
}
}
#endif
};
} // namespace _winml