mirror of
https://github.com/saymrwulf/onnxruntime.git
synced 2026-07-22 19:23:30 +00:00
### Description Merge main to WindowsAI ### Motivation and Context <!-- - Why is this change required? What problem does it solve? - If it fixes an open issue, please link to the issue here. --> --------- Signed-off-by: Nash <george.nash@intel.com> Signed-off-by: Yiming Hu <yiming.hu@amd.com> Signed-off-by: Liqun Fu <liqfu@microsoft.com> Co-authored-by: Kaz Nishimura <kazssym@linuxfront.com> Co-authored-by: Tianlei Wu <tlwu@microsoft.com> Co-authored-by: Nat Kershaw (MSFT) <nakersha@microsoft.com> Co-authored-by: Yulong Wang <7679871+fs-eire@users.noreply.github.com> Co-authored-by: Changming Sun <chasun@microsoft.com> Co-authored-by: zesongw <zesong.wang@intel.com> Co-authored-by: Yi Zhang <zhanyi@microsoft.com> Co-authored-by: Dmitri Smirnov <yuslepukhin@users.noreply.github.com> Co-authored-by: Yifan Li <109183385+yf711@users.noreply.github.com> Co-authored-by: simonjub <78098752+simonjub@users.noreply.github.com> Co-authored-by: PeixuanZuo <94887879+PeixuanZuo@users.noreply.github.com> Co-authored-by: Adrian Lizarraga <adlizarraga@microsoft.com> Co-authored-by: Edward Chen <18449977+edgchen1@users.noreply.github.com> Co-authored-by: Arthur Islamov <arthur@islamov.ai> Co-authored-by: Jambay Kinley <jambaykinley@microsoft.com> Co-authored-by: Justin Chu <justinchuby@users.noreply.github.com> Co-authored-by: Wei-Sheng Chin <wschin@outlook.com> Co-authored-by: Bowen Bao <bowbao@microsoft.com> Co-authored-by: Hariharan Seshadri <shariharan91@gmail.com> Co-authored-by: Numfor Tiapo <numsmt2@gmail.com> Co-authored-by: Vincent Wang <wangwchpku@outlook.com> Co-authored-by: Pranav Sharma <prs@microsoft.com> Co-authored-by: George Nash <george.nash@intel.com> Co-authored-by: Abhishek Jindal <abjindal@microsoft.com> Co-authored-by: pengwa <pengwa@microsoft.com> Co-authored-by: Yiming Hu <woinck@users.noreply.github.com> Co-authored-by: Jiajia Qin <jiajia.qin@intel.com> Co-authored-by: Lukas Berbuer <36054362+lukasberbuer@users.noreply.github.com> Co-authored-by: Wanming Lin <wanming.lin@intel.com> Co-authored-by: Xavier Dupré <xadupre@users.noreply.github.com> Co-authored-by: aimilefth <60664743+aimilefth@users.noreply.github.com> Co-authored-by: Baiju Meswani <bmeswani@microsoft.com> Co-authored-by: Adam Pocock <adam.pocock@oracle.com> Co-authored-by: Chi Lo <54722500+chilo-ms@users.noreply.github.com> Co-authored-by: RandySheriffH <48490400+RandySheriffH@users.noreply.github.com> Co-authored-by: Randy Shuai <rashuai@microsoft.com> Co-authored-by: Vadym Stupakov <vadim.stupakov@gmail.com> Co-authored-by: Jian Chen <cjian@microsoft.com> Co-authored-by: Brian Lambert <98757707+brian-pieces@users.noreply.github.com> Co-authored-by: Nicolò Lucchesi <nicolo.lucchesi@gmail.com> Co-authored-by: liqun Fu <liqfu@microsoft.com> Co-authored-by: trajep <trajepl@gmail.com> Co-authored-by: Scott McKay <skottmckay@gmail.com> Co-authored-by: Mustafa Ateş Uzun <mustafauzun0@gmail.com> Co-authored-by: MistEO <mistereo@hotmail.com> Co-authored-by: satyajandhyala <satya.k.jandhyala@gmail.com> Co-authored-by: shaahji <96227573+shaahji@users.noreply.github.com> Co-authored-by: Rachel Guo <35738743+YUNQIUGUO@users.noreply.github.com> Co-authored-by: rachguo <rachguo@rachguos-Mini.attlocal.net> Co-authored-by: Caroline Zhu <wolfivyaura@gmail.com> Co-authored-by: Caroline Zhu <carolinezhu@microsoft.com> Co-authored-by: Guenther Schmuelling <guschmue@microsoft.com> Co-authored-by: xhcao <xinghua.cao@intel.com> Co-authored-by: Ella Charlaix <80481427+echarlaix@users.noreply.github.com> Co-authored-by: Xu Xing <xing.xu@intel.com> Co-authored-by: Hector Li <hecli@microsoft.com> Co-authored-by: Ye Wang <52801275+wangyems@users.noreply.github.com> Co-authored-by: Your Name <you@example.com> Co-authored-by: Benedikt Hilmes <benedikt.hilmes@rwth-aachen.de> Co-authored-by: rachguo <rachguo@rachguos-Mac-mini.local> Co-authored-by: George Wu <jywu@microsoft.com> Co-authored-by: JiCheng <wejoncy@163.com> Co-authored-by: Sheil Kumar <smk2007@gmail.com> Co-authored-by: Sheil Kumar <sheilk@microsoft.com> Co-authored-by: cloudhan <guangyunhan@microsoft.com> Co-authored-by: kyoshisuki <143475866+kyoshisuki@users.noreply.github.com> Co-authored-by: aciddelgado <139922440+aciddelgado@users.noreply.github.com> Co-authored-by: tlwu@microsoft.com <tlwu@a100.crj0ad2y1kku1j4yxl4sj10o4e.gx.internal.cloudapp.net> Co-authored-by: Maximilian Müller <44298237+gedoensmax@users.noreply.github.com> Co-authored-by: Tang, Cheng <souptc@gmail.com> Co-authored-by: Cheng Tang <chenta@microsoft.com@orttrainingdev9.d32nl1ml4oruzj4qz3bqlggovf.px.internal.cloudapp.net> Co-authored-by: Cheng Tang <chenta@microsoft.com> Co-authored-by: Jeff Daily <jeff.daily@amd.com> Co-authored-by: cloudhan <cloudhan@outlook.com> Co-authored-by: Yufeng Li <liyufeng1987@gmail.com> Co-authored-by: Zhang Lei <zhang.huanning@hotmail.com> Co-authored-by: Dwayne Robinson <fdwr@hotmail.com> Co-authored-by: Zhipeng Han <zhipeng.han@outlook.com> Co-authored-by: Thiago Crepaldi <thiago.crepaldi@microsoft.com> Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> Co-authored-by: Patrice Vignola <vignola.patrice@gmail.com> Co-authored-by: kunal-vaishnavi <115581922+kunal-vaishnavi@users.noreply.github.com> Co-authored-by: snadampal <87143774+snadampal@users.noreply.github.com> Co-authored-by: Sumit Agarwal <sumitagarwal330@gmail.com> Co-authored-by: Ashwini Khade <askhade@microsoft.com> Co-authored-by: Yang Gu <yang.gu@intel.com> Co-authored-by: Cheng Tang <chenta@a100.crj0ad2y1kku1j4yxl4sj10o4e.gx.internal.cloudapp.net> Co-authored-by: mindest <30493312+mindest@users.noreply.github.com> Co-authored-by: Scott McKay <Scott.McKay@microsoft.com> Co-authored-by: Xavier Dupre <xadupre@microsoft.com@orttrainingdev9.d32nl1ml4oruzj4qz3bqlggovf.px.internal.cloudapp.net> Co-authored-by: guyang3532 <62738430+guyang3532@users.noreply.github.com> Co-authored-by: Carson M <carson@pyke.io> Co-authored-by: sophies927 <107952697+sophies927@users.noreply.github.com>
296 lines
12 KiB
C++
296 lines
12 KiB
C++
// Copyright (c) Microsoft Corporation. All rights reserved.
|
|
// Licensed under the MIT License.
|
|
|
|
#pragma once
|
|
|
|
#include "inc/ImageConversionTypes.h"
|
|
#include "inc/NominalRangeConverter.h"
|
|
|
|
namespace _winml {
|
|
|
|
class CpuTensorizer {
|
|
public:
|
|
template <typename T>
|
|
static HRESULT TensorizeData(
|
|
_In_ ImageTensorChannelType formatFrom,
|
|
_In_ ImageTensorChannelType formatTo,
|
|
_In_ winml::LearningModelPixelRange pixelRange,
|
|
_In_ BYTE* pBuffer,
|
|
_In_ UINT32 bufferWidth,
|
|
_In_ const wgi::BitmapBounds& inputBounds,
|
|
_Inout_ T* pCPUTensor
|
|
) {
|
|
#pragma warning(push)
|
|
#pragma warning(disable : 26014 \
|
|
) // warning about possible out of bounds accesing pData, but input is checked for BGRA8 format, so uiCapacity should be in multiples of 4 \
|
|
// input is BGRA8: so blue at i, green is at i + 1, red is at i + 2
|
|
|
|
uint32_t bytesPerPixel = formatFrom == kImageTensorChannelTypeGRAY8 ? 1 : 4;
|
|
|
|
// bufferWidth may have padding because of optimization, but bytesPerRow includes only the real tensor data. We need to jump
|
|
// over bufferWidth's extra padding
|
|
uint32_t bytesPerRow = inputBounds.Width * bytesPerPixel;
|
|
uint32_t start = (inputBounds.Y * bufferWidth) + (inputBounds.X * bytesPerPixel);
|
|
uint32_t end = start + bufferWidth * inputBounds.Height;
|
|
uint32_t pixelInd = 0;
|
|
|
|
uint32_t xElements = inputBounds.Width - inputBounds.X;
|
|
uint32_t yElements = inputBounds.Height - inputBounds.Y;
|
|
|
|
auto nominalRangeConverter = NominalRangeConverter(pixelRange);
|
|
|
|
if (formatFrom == kImageTensorChannelTypeBGR8 && formatTo == kImageTensorChannelTypeBGR8 || formatFrom == kImageTensorChannelTypeRGB8 && formatTo == kImageTensorChannelTypeRGB8) {
|
|
// Convert BGR8 -> BGR8 or RGB8 -> RGB8
|
|
for (uint64_t y = 0; y < yElements; y++) {
|
|
DeinterleaveRowByteToFloat(
|
|
pBuffer + y * bufferWidth + start,
|
|
pCPUTensor + y * inputBounds.Width,
|
|
pCPUTensor + (inputBounds.Height * inputBounds.Width) + y * inputBounds.Width,
|
|
pCPUTensor + (inputBounds.Height * inputBounds.Width) * 2 + y * inputBounds.Width,
|
|
xElements,
|
|
bytesPerPixel,
|
|
nominalRangeConverter
|
|
);
|
|
}
|
|
} else if (formatFrom == kImageTensorChannelTypeBGR8 && formatTo == kImageTensorChannelTypeRGB8 || formatFrom == kImageTensorChannelTypeRGB8 && formatTo == kImageTensorChannelTypeBGR8) {
|
|
// Convert RGB8 -> BGR8 or BGR8 -> RGB8
|
|
for (uint32_t y = 0; y < yElements; y++) {
|
|
DeinterleaveRowByteToFloat(
|
|
pBuffer + y * bufferWidth + start,
|
|
pCPUTensor + (inputBounds.Height * inputBounds.Width) * 2 + y * inputBounds.Width,
|
|
pCPUTensor + (inputBounds.Height * inputBounds.Width) + y * inputBounds.Width,
|
|
pCPUTensor + y * inputBounds.Width,
|
|
xElements,
|
|
bytesPerPixel,
|
|
nominalRangeConverter
|
|
);
|
|
}
|
|
} else if (formatTo == kImageTensorChannelTypeGRAY8 && (formatFrom == kImageTensorChannelTypeBGR8 || formatFrom == kImageTensorChannelTypeRGB8)) {
|
|
// Convert BGR8 -> GRAY8 or RGB8 -> GRAY8
|
|
uint32_t blueIncrement = formatFrom == kImageTensorChannelTypeBGR8 ? 0 : 2;
|
|
uint32_t redIncrement = formatFrom == kImageTensorChannelTypeBGR8 ? 2 : 0;
|
|
|
|
for (UINT32 i = start; i < end; i += bufferWidth) {
|
|
for (UINT32 j = i; j < i + bytesPerRow; j += bytesPerPixel) {
|
|
float red = float(pBuffer[j + redIncrement]);
|
|
float green = float(pBuffer[j + 1]);
|
|
float blue = float(pBuffer[j + blueIncrement]);
|
|
float gray = 0.2126f * red + 0.7152f * green + 0.0722f * blue;
|
|
pCPUTensor[pixelInd] = ConvertByteToFloat<T>(static_cast<BYTE>(gray), nominalRangeConverter);
|
|
pixelInd++;
|
|
}
|
|
}
|
|
} else if (formatFrom == kImageTensorChannelTypeGRAY8 && (formatTo == kImageTensorChannelTypeBGR8 || formatTo == kImageTensorChannelTypeRGB8)) {
|
|
// Convert GRAY8 -> BGR8 or GRAY8 -> RGB8
|
|
for (UINT32 i = start; i < end; i += bufferWidth) {
|
|
for (UINT32 j = i; j < i + bytesPerRow; j += bytesPerPixel) {
|
|
pCPUTensor[pixelInd] = ConvertByteToFloat<T>(pBuffer[j], nominalRangeConverter);
|
|
pCPUTensor[(inputBounds.Height * inputBounds.Width) + pixelInd] =
|
|
ConvertByteToFloat<T>(pBuffer[j], nominalRangeConverter);
|
|
pCPUTensor[(inputBounds.Height * inputBounds.Width * 2) + pixelInd] =
|
|
ConvertByteToFloat<T>(pBuffer[j], nominalRangeConverter);
|
|
pixelInd++;
|
|
}
|
|
}
|
|
} else if (formatFrom == kImageTensorChannelTypeGRAY8 && formatTo == kImageTensorChannelTypeGRAY8) {
|
|
// Convert GRAY8 -> GRAY8
|
|
for (UINT32 i = start; i < end; i += bufferWidth) {
|
|
for (UINT32 j = i; j < i + bytesPerRow; j += bytesPerPixel) {
|
|
pCPUTensor[pixelInd] = ConvertByteToFloat<T>(pBuffer[j], nominalRangeConverter);
|
|
pixelInd++;
|
|
}
|
|
}
|
|
}
|
|
#pragma warning(pop)
|
|
else {
|
|
return E_INVALIDARG;
|
|
}
|
|
return S_OK;
|
|
}
|
|
|
|
private:
|
|
template <typename T>
|
|
static T ConvertByteToFloat(const BYTE& input, const NominalRangeConverter& nominalRangeConverter);
|
|
|
|
// clang-format off
|
|
template <>
|
|
#if _MSVC_LANG < 202002L
|
|
static
|
|
#endif
|
|
float ConvertByteToFloat(const BYTE& input, const NominalRangeConverter& nominalRangeConverter) {
|
|
return nominalRangeConverter.Normalize(static_cast<float>(input));
|
|
}
|
|
|
|
// clang-format off
|
|
template <>
|
|
#if _MSVC_LANG < 202002L
|
|
static
|
|
#endif
|
|
DirectX::PackedVector::HALF ConvertByteToFloat(
|
|
const BYTE& input,
|
|
const NominalRangeConverter& nominalRangeConverter
|
|
) {
|
|
return nominalRangeConverter.Normalize(DirectX::PackedVector::XMConvertFloatToHalf(input));
|
|
}
|
|
|
|
template <typename T>
|
|
static void DeinterleaveRowByteToFloat(
|
|
_In_ BYTE* pBuffer,
|
|
_Inout_ T* xChannel,
|
|
_Inout_ T* yChannel,
|
|
_Inout_ T* zChannel,
|
|
uint32_t pixelElements,
|
|
uint32_t bytesPerPixel,
|
|
const NominalRangeConverter& nominalRangeConverter
|
|
) {
|
|
UINT32 j;
|
|
|
|
for (j = 0; j < (pixelElements & 0xFFFFFFFC); j += 4) {
|
|
xChannel[j] = ConvertByteToFloat<T>(pBuffer[0], nominalRangeConverter);
|
|
yChannel[j] = ConvertByteToFloat<T>(pBuffer[1], nominalRangeConverter);
|
|
zChannel[j] = ConvertByteToFloat<T>(pBuffer[2], nominalRangeConverter);
|
|
xChannel[j + 1] = ConvertByteToFloat<T>(pBuffer[4], nominalRangeConverter);
|
|
yChannel[j + 1] = ConvertByteToFloat<T>(pBuffer[5], nominalRangeConverter);
|
|
zChannel[j + 1] = ConvertByteToFloat<T>(pBuffer[6], nominalRangeConverter);
|
|
xChannel[j + 2] = ConvertByteToFloat<T>(pBuffer[8], nominalRangeConverter);
|
|
yChannel[j + 2] = ConvertByteToFloat<T>(pBuffer[9], nominalRangeConverter);
|
|
zChannel[j + 2] = ConvertByteToFloat<T>(pBuffer[10], nominalRangeConverter);
|
|
xChannel[j + 3] = ConvertByteToFloat<T>(pBuffer[12], nominalRangeConverter);
|
|
yChannel[j + 3] = ConvertByteToFloat<T>(pBuffer[13], nominalRangeConverter);
|
|
zChannel[j + 3] = ConvertByteToFloat<T>(pBuffer[14], nominalRangeConverter);
|
|
pBuffer += bytesPerPixel * 4;
|
|
}
|
|
|
|
for (; j < pixelElements; j++) {
|
|
xChannel[j] = ConvertByteToFloat<T>(pBuffer[0], nominalRangeConverter);
|
|
yChannel[j] = ConvertByteToFloat<T>(pBuffer[1], nominalRangeConverter);
|
|
zChannel[j] = ConvertByteToFloat<T>(pBuffer[2], nominalRangeConverter);
|
|
pBuffer += bytesPerPixel;
|
|
}
|
|
}
|
|
|
|
// clang-format off
|
|
#if defined(_M_AMD64) || defined(_M_IX86)
|
|
template <>
|
|
#if _MSVC_LANG < 202002L
|
|
static
|
|
#endif
|
|
void DeinterleaveRowByteToFloat(
|
|
_In_ BYTE* pBuffer,
|
|
_Inout_ float* xChannel,
|
|
_Inout_ float* yChannel,
|
|
_Inout_ float* zChannel,
|
|
uint32_t pixelElements,
|
|
uint32_t bytesPerPixel,
|
|
const NominalRangeConverter& nominalRangeConverter
|
|
) {
|
|
assert(bytesPerPixel == 4);
|
|
|
|
__m128i ZeroVector = _mm_setzero_si128();
|
|
while (pixelElements >= 8) {
|
|
// Load 8 Pixels into 2 Registers
|
|
// vBytes0 = X0 Y0 Z0 A0 X1 Y1...
|
|
// vBytes0 = X4 Y4 Z4 A4 X2 Y2...
|
|
__m128i vBytes0 = _mm_loadu_si128((__m128i*)pBuffer);
|
|
__m128i vBytes1 = _mm_loadu_si128((__m128i*)(pBuffer + 16));
|
|
|
|
// Shuffle to get
|
|
// vi0 = X0 X4 Y0 Y4...A1 A5 (A is Alpha which is ignored)
|
|
// vi1 = X2 X6 Y2 Y6...A2 A6
|
|
__m128i vi0 = _mm_unpacklo_epi8(vBytes0, vBytes1);
|
|
__m128i vi1 = _mm_unpackhi_epi8(vBytes0, vBytes1);
|
|
|
|
// Shuffle again to get
|
|
// vi0 = X0 X2 X4 X6...A4 A6 (All even byes)
|
|
// vi1 = X1 X3 X5 X7...A3 A7 (All odd bytes)
|
|
__m128i vi2 = _mm_unpacklo_epi8(vi0, vi1);
|
|
__m128i vi3 = _mm_unpackhi_epi8(vi0, vi1);
|
|
|
|
// Shuffle last time to get desired order
|
|
// vi0 = X0 X1 X2 X3...Y6 Y7 (All even byes)
|
|
// vi1 = Z0 Z1 Z2 Z3...A6 A7 (All odd bytes)
|
|
__m128i vi4 = _mm_unpacklo_epi8(vi2, vi3);
|
|
__m128i vi5 = _mm_unpackhi_epi8(vi2, vi3);
|
|
|
|
// unpack with zeros to get 16 bit ints
|
|
// vXWords = X0 X1...X6 X7
|
|
__m128i vXWords = _mm_unpacklo_epi8(vi4, ZeroVector);
|
|
|
|
// unpack again with zeros to get 32 bit ints
|
|
__m128i vXIntsLo = _mm_unpacklo_epi16(vXWords, ZeroVector);
|
|
__m128i vXIntsHi = _mm_unpackhi_epi16(vXWords, ZeroVector);
|
|
|
|
// store 256 bits of X channel Floats
|
|
_mm_storeu_ps(xChannel, nominalRangeConverter.Normalize(_mm_cvtepi32_ps(vXIntsLo)));
|
|
_mm_storeu_ps(xChannel + 4, nominalRangeConverter.Normalize(_mm_cvtepi32_ps(vXIntsHi)));
|
|
xChannel += 8;
|
|
|
|
// unpack again for Y
|
|
__m128i vYWords = _mm_unpackhi_epi8(vi4, ZeroVector);
|
|
|
|
__m128i vYIntsLo = _mm_unpacklo_epi16(vYWords, ZeroVector);
|
|
__m128i vYIntsHi = _mm_unpackhi_epi16(vYWords, ZeroVector);
|
|
|
|
_mm_storeu_ps(yChannel, nominalRangeConverter.Normalize(_mm_cvtepi32_ps(vYIntsLo)));
|
|
_mm_storeu_ps(yChannel + 4, nominalRangeConverter.Normalize(_mm_cvtepi32_ps(vYIntsHi)));
|
|
yChannel += 8;
|
|
|
|
// unpack again for Z
|
|
__m128i vZWords = _mm_unpacklo_epi8(vi5, ZeroVector);
|
|
|
|
__m128i vZIntsLo = _mm_unpacklo_epi16(vZWords, ZeroVector);
|
|
__m128i vZIntsHi = _mm_unpackhi_epi16(vZWords, ZeroVector);
|
|
|
|
_mm_storeu_ps(zChannel, nominalRangeConverter.Normalize(_mm_cvtepi32_ps(vZIntsLo)));
|
|
_mm_storeu_ps(zChannel + 4, nominalRangeConverter.Normalize(_mm_cvtepi32_ps(vZIntsHi)));
|
|
zChannel += 8;
|
|
|
|
pBuffer += 32;
|
|
pixelElements -= 8;
|
|
}
|
|
if (pixelElements >= 4) {
|
|
// load 4 pixels = 16 values
|
|
__m128i vBytes = _mm_loadu_si128((__m128i*)pBuffer);
|
|
|
|
// unpack to 16 bits
|
|
__m128i vWords0 = _mm_unpacklo_epi8(vBytes, ZeroVector);
|
|
__m128i vWords1 = _mm_unpackhi_epi8(vBytes, ZeroVector);
|
|
|
|
// unpack to 32 bits
|
|
__m128i vInts0 = _mm_unpacklo_epi16(vWords0, ZeroVector);
|
|
__m128i vInts1 = _mm_unpackhi_epi16(vWords0, ZeroVector);
|
|
__m128i vInts2 = _mm_unpacklo_epi16(vWords1, ZeroVector);
|
|
__m128i vInts3 = _mm_unpackhi_epi16(vWords1, ZeroVector);
|
|
|
|
// Normalize to floats
|
|
__m128 vFloats0 = _mm_cvtepi32_ps(vInts0);
|
|
__m128 vFloats1 = _mm_cvtepi32_ps(vInts1);
|
|
__m128 vFloats2 = _mm_cvtepi32_ps(vInts2);
|
|
__m128 vFloats3 = _mm_cvtepi32_ps(vInts3);
|
|
|
|
// We want have row but need cols so transpose 4x4 matrix
|
|
_MM_TRANSPOSE4_PS(vFloats0, vFloats1, vFloats2, vFloats3);
|
|
|
|
// Drop alpha channel transposed to vFloats3 write out rest
|
|
_mm_storeu_ps(xChannel, nominalRangeConverter.Normalize(vFloats0));
|
|
_mm_storeu_ps(yChannel, nominalRangeConverter.Normalize(vFloats1));
|
|
_mm_storeu_ps(zChannel, nominalRangeConverter.Normalize(vFloats2));
|
|
|
|
xChannel += 4;
|
|
yChannel += 4;
|
|
zChannel += 4;
|
|
pBuffer += 4 * 4;
|
|
pixelElements -= 4;
|
|
}
|
|
|
|
// Any remainder just do one at a time
|
|
for (uint32_t j = 0; j < pixelElements; j++) {
|
|
xChannel[j] = nominalRangeConverter.Normalize(static_cast<float>(pBuffer[0]));
|
|
yChannel[j] = nominalRangeConverter.Normalize(static_cast<float>(pBuffer[1]));
|
|
zChannel[j] = nominalRangeConverter.Normalize(static_cast<float>(pBuffer[2]));
|
|
pBuffer += bytesPerPixel;
|
|
}
|
|
}
|
|
#endif
|
|
};
|
|
} // namespace _winml
|