Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
ColorSpace.cu593 linesDownload Raw Back to Utils
1/*
2 * This copyright notice applies to this header file only:
3 *
4 * Copyright (c) 2010-2024 NVIDIA Corporation
5 *
6 * Permission is hereby granted, free of charge, to any person
7 * obtaining a copy of this software and associated documentation
8 * files (the "Software"), to deal in the Software without
9 * restriction, including without limitation the rights to use,
10 * copy, modify, merge, publish, distribute, sublicense, and/or sell
11 * copies of the software, and to permit persons to whom the
12 * software is furnished to do so, subject to the following
13 * conditions:
14 *
15 * The above copyright notice and this permission notice shall be
16 * included in all copies or substantial portions of the Software.
17 *
18 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
19 * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES
20 * OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
21 * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
22 * HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
23 * WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
24 * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
25 * OTHER DEALINGS IN THE SOFTWARE.
26 */
27
28#include "ColorSpace.h"
29
30__constant__ float matYuv2Rgb[3][3];
31__constant__ float matRgb2Yuv[3][3];
32__constant__ int   yuvOffsets[2];
33
34void inline GetConstants(int iMatrix, bool video_full_range, float &wr, float &wb, float& yscale, float& cscale, bool yuv2rgb) {
35    yscale = 1.0f;
36    cscale = 1.0f;
37
38    if (!video_full_range)
39    {
40        yscale = yuv2rgb ? 255.0f / 219.0f : 219.0f / 255.0f;
41        cscale = yuv2rgb ? 255.0f / 224.0f : 224.0f / 255.0f;
42    }
43
44    switch (iMatrix)
45    {
46    case ColorSpaceStandard_BT709:
47    default:
48        wr = 0.2126f; wb = 0.0722f;
49        break;
50
51    case ColorSpaceStandard_FCC:
52        wr = 0.30f; wb = 0.11f;
53        break;
54
55    case ColorSpaceStandard_BT470:
56    case ColorSpaceStandard_BT601:
57        wr = 0.2990f; wb = 0.1140f;
58        break;
59
60    case ColorSpaceStandard_SMPTE240M:
61        wr = 0.212f; wb = 0.087f;
62        break;
63
64    case ColorSpaceStandard_BT2020:
65    case ColorSpaceStandard_BT2020C:
66        wr = 0.2627f; wb = 0.0593f;
67        break;
68    }
69}
70
71void SetMatYuv2Rgb(int iMatrix, int bytesPerPixel, bool video_full_range = 0) {
72    float wr, wb;
73    float yscale, cscale;
74    int offsets[2];
75
76    GetConstants(iMatrix, video_full_range, wr, wb, yscale, cscale, 1);
77    float mat[3][3] = {
78        1.0f, 0.0f, (1.0f - wr) / 0.5f,
79        1.0f, -wb * (1.0f - wb) / 0.5f / (1 - wb - wr), -wr * (1 - wr) / 0.5f / (1 - wb - wr),
80        1.0f, (1.0f - wb) / 0.5f, 0.0f,
81    };
82    for (int i = 0; i < 3; i++) {
83        for (int j = 0; j < 3; j++) {
84            mat[i][j] = (float)(1.0 * (j == 0? yscale : cscale) * mat[i][j]);
85        }
86    }
87    cudaMemcpyToSymbol(matYuv2Rgb, mat, sizeof(mat));
88
89    offsets[0] = video_full_range == 0 ? 1 << (bytesPerPixel * 8 - 4) : 0;    // low
90    offsets[1] = 1 << (bytesPerPixel * 8 - 1);    // mid
91    cudaMemcpyToSymbol(yuvOffsets, offsets, sizeof(offsets));
92}
93
94void SetMatRgb2Yuv(int iMatrix, int bytesPerPixel, bool video_full_range = 0) {
95    float wr, wb;
96    float yscale, cscale;
97    int offsets[2];
98
99    GetConstants(iMatrix, video_full_range, wr, wb, yscale, cscale, 0);
100    float mat[3][3] = {
101        wr, 1.0f - wb - wr, wb,
102        -0.5f * wr / (1.0f - wb), -0.5f * (1 - wb - wr) / (1.0f - wb), 0.5f,
103        0.5f, -0.5f * (1.0f - wb - wr) / (1.0f - wr), -0.5f * wb / (1.0f - wr),
104    };
105    for (int i = 0; i < 3; i++) {
106        for (int j = 0; j < 3; j++) {
107            mat[i][j] = (float)(1.0 * (i == 0 ? yscale : cscale) * mat[i][j]);
108        }
109    }
110    cudaMemcpyToSymbol(matRgb2Yuv, mat, sizeof(mat));
111
112    offsets[0] = video_full_range == 0 ? 1 << (bytesPerPixel * 8 - 4) : 0;    // low
113    offsets[1] = 1 << (bytesPerPixel * 8 - 1);    // mid
114    cudaMemcpyToSymbol(yuvOffsets, offsets, sizeof(offsets));
115}
116
117template<class T>
118__device__ static T Clamp(T x, T lower, T upper) {
119    return x < lower ? lower : (x > upper ? upper : x);
120}
121
122template<class Rgb, class YuvUnit>
123__device__ inline Rgb YuvToRgbForPixel(YuvUnit y, YuvUnit u, YuvUnit v) {
124    float fy = (int)y - yuvOffsets[0], fu = (int)u - yuvOffsets[1], fv = (int)v - yuvOffsets[1];
125    const float maxf = (1 << sizeof(YuvUnit) * 8) - 1.0f;
126    YuvUnit 
127        r = (YuvUnit)Clamp(matYuv2Rgb[0][0] * fy + matYuv2Rgb[0][1] * fu + matYuv2Rgb[0][2] * fv, 0.0f, maxf),
128        g = (YuvUnit)Clamp(matYuv2Rgb[1][0] * fy + matYuv2Rgb[1][1] * fu + matYuv2Rgb[1][2] * fv, 0.0f, maxf),
129        b = (YuvUnit)Clamp(matYuv2Rgb[2][0] * fy + matYuv2Rgb[2][1] * fu + matYuv2Rgb[2][2] * fv, 0.0f, maxf);
130    
131    Rgb rgb{};
132    const int nShift = abs((int)sizeof(YuvUnit) - (int)sizeof(rgb.c.r)) * 8;
133    if (sizeof(YuvUnit) >= sizeof(rgb.c.r)) {
134        rgb.c.r = r >> nShift;
135        rgb.c.g = g >> nShift;
136        rgb.c.b = b >> nShift;
137    } else {
138        rgb.c.r = r << nShift;
139        rgb.c.g = g << nShift;
140        rgb.c.b = b << nShift;
141    }
142    return rgb;
143}
144
145template<class YuvUnitx2, class Rgb, class RgbIntx2>
146__global__ static void YuvToRgbKernel(uint8_t *pYuv, int nYuvPitch, uint8_t *pRgb, int nRgbPitch, int nWidth, int nHeight) {
147    int x = (threadIdx.x + blockIdx.x * blockDim.x) * 2;
148    int y = (threadIdx.y + blockIdx.y * blockDim.y) * 2;
149    if (x + 1 >= nWidth || y + 1 >= nHeight) {
150        return;
151    }
152
153    uint8_t *pSrc = pYuv + x * sizeof(YuvUnitx2) / 2 + y * nYuvPitch;
154    uint8_t *pDst = pRgb + x * sizeof(Rgb) + y * nRgbPitch;
155
156    YuvUnitx2 l0 = *(YuvUnitx2 *)pSrc;
157    YuvUnitx2 l1 = *(YuvUnitx2 *)(pSrc + nYuvPitch);
158    YuvUnitx2 ch = *(YuvUnitx2 *)(pSrc + (nHeight - y / 2) * nYuvPitch);
159
160    *(RgbIntx2 *)pDst = RgbIntx2 {
161        YuvToRgbForPixel<Rgb>(l0.x, ch.x, ch.y).d,
162        YuvToRgbForPixel<Rgb>(l0.y, ch.x, ch.y).d,
163    };
164    *(RgbIntx2 *)(pDst + nRgbPitch) = RgbIntx2 {
165        YuvToRgbForPixel<Rgb>(l1.x, ch.x, ch.y).d, 
166        YuvToRgbForPixel<Rgb>(l1.y, ch.x, ch.y).d,
167    };
168}
169
170template<class YuvUnitx2, class Rgb, class RgbIntx2>
171__global__ static void Yuv422ToRgbKernel(uint8_t *pYuv, int nYuvPitch, uint8_t *pRgb, int nRgbPitch, int nWidth, int nHeight) {
172    int x = (threadIdx.x + blockIdx.x * blockDim.x) * 2;
173    int y = (threadIdx.y + blockIdx.y * blockDim.y);
174    if (x + 1 >= nWidth || y >= nHeight) {
175        return;
176    }
177
178    uint8_t *pSrc = pYuv + x * sizeof(YuvUnitx2) / 2 + y * nYuvPitch;
179    uint8_t *pDst = pRgb + x * sizeof(Rgb) + y * nRgbPitch;
180
181    YuvUnitx2 l = *(YuvUnitx2 *)pSrc;
182    YuvUnitx2 ch = *(YuvUnitx2 *)(pSrc + (nHeight * nYuvPitch));
183
184    *(RgbIntx2 *)pDst = RgbIntx2 {
185        YuvToRgbForPixel<Rgb>(l.x, ch.x, ch.y).d,
186        YuvToRgbForPixel<Rgb>(l.y, ch.x, ch.y).d,
187    };
188}
189
190template<class YuvUnitx2, class Rgb, class RgbIntx2>
191__global__ static void Yuv444ToRgbKernel(uint8_t *pYuv, int nYuvPitch, uint8_t *pRgb, int nRgbPitch, int nWidth, int nHeight) {
192    int x = (threadIdx.x + blockIdx.x * blockDim.x) * 2;
193    int y = (threadIdx.y + blockIdx.y * blockDim.y);
194    if (x + 1 >= nWidth || y  >= nHeight) {
195        return;
196    }
197
198    uint8_t *pSrc = pYuv + x * sizeof(YuvUnitx2) / 2 + y * nYuvPitch;
199    uint8_t *pDst = pRgb + x * sizeof(Rgb) + y * nRgbPitch;
200
201    YuvUnitx2 l0 = *(YuvUnitx2 *)pSrc;
202    YuvUnitx2 ch1 = *(YuvUnitx2 *)(pSrc + (nHeight * nYuvPitch));
203    YuvUnitx2 ch2 = *(YuvUnitx2 *)(pSrc + (2 * nHeight * nYuvPitch));
204
205    *(RgbIntx2 *)pDst = RgbIntx2{
206        YuvToRgbForPixel<Rgb>(l0.x, ch1.x, ch2.x).d,
207        YuvToRgbForPixel<Rgb>(l0.y, ch1.y, ch2.y).d,
208    };
209}
210
211template<class YuvUnitx2, class Rgb, class RgbUnitx2>
212__global__ static void YuvToRgbPlanarKernel(uint8_t *pYuv, int nYuvPitch, uint8_t *pRgbp, int nRgbpPitch, int nWidth, int nHeight) {
213    int x = (threadIdx.x + blockIdx.x * blockDim.x) * 2;
214    int y = (threadIdx.y + blockIdx.y * blockDim.y) * 2;
215    if (x + 1 >= nWidth || y + 1 >= nHeight) {
216        return;
217    }
218
219    uint8_t *pSrc = pYuv + x * sizeof(YuvUnitx2) / 2 + y * nYuvPitch;
220
221    YuvUnitx2 l0 = *(YuvUnitx2 *)pSrc;
222    YuvUnitx2 l1 = *(YuvUnitx2 *)(pSrc + nYuvPitch);
223    YuvUnitx2 ch = *(YuvUnitx2 *)(pSrc + (nHeight - y / 2) * nYuvPitch);
224
225    Rgb rgb0 = YuvToRgbForPixel<Rgb>(l0.x, ch.x, ch.y),
226        rgb1 = YuvToRgbForPixel<Rgb>(l0.y, ch.x, ch.y),
227        rgb2 = YuvToRgbForPixel<Rgb>(l1.x, ch.x, ch.y),
228        rgb3 = YuvToRgbForPixel<Rgb>(l1.y, ch.x, ch.y);
229
230    uint8_t *pDst = pRgbp + x * sizeof(RgbUnitx2) / 2 + y * nRgbpPitch;
231    *(RgbUnitx2 *)pDst = RgbUnitx2 {rgb0.v.x, rgb1.v.x};
232    *(RgbUnitx2 *)(pDst + nRgbpPitch) = RgbUnitx2 {rgb2.v.x, rgb3.v.x};
233    pDst += nRgbpPitch * nHeight;
234    *(RgbUnitx2 *)pDst = RgbUnitx2 {rgb0.v.y, rgb1.v.y};
235    *(RgbUnitx2 *)(pDst + nRgbpPitch) = RgbUnitx2 {rgb2.v.y, rgb3.v.y};
236    pDst += nRgbpPitch * nHeight;
237    *(RgbUnitx2 *)pDst = RgbUnitx2 {rgb0.v.z, rgb1.v.z};
238    *(RgbUnitx2 *)(pDst + nRgbpPitch) = RgbUnitx2 {rgb2.v.z, rgb3.v.z};
239}
240
241template<class YuvUnitx2, class Rgb, class RgbUnitx2>
242__global__ static void Yuv422ToRgbPlanarKernel(uint8_t *pYuv, int nYuvPitch, uint8_t *pRgbp, int nRgbpPitch, int nWidth, int nHeight) {
243    int x = (threadIdx.x + blockIdx.x * blockDim.x) * 2;
244    int y = (threadIdx.y + blockIdx.y * blockDim.y);
245    if (x + 1 >= nWidth || y >= nHeight) {
246        return;
247    }
248
249    uint8_t *pSrc = pYuv + x * sizeof(YuvUnitx2) / 2 + y * nYuvPitch;
250
251    YuvUnitx2 l = *(YuvUnitx2 *)pSrc;
252    YuvUnitx2 ch = *(YuvUnitx2 *)(pSrc + nHeight * nYuvPitch);
253
254    Rgb rgb0 = YuvToRgbForPixel<Rgb>(l.x, ch.x, ch.y),
255        rgb1 = YuvToRgbForPixel<Rgb>(l.y, ch.x, ch.y);
256
257    uint8_t *pDst = pRgbp + x * sizeof(RgbUnitx2) / 2 + y * nRgbpPitch;
258    *(RgbUnitx2 *)pDst = RgbUnitx2 {rgb0.v.x, rgb1.v.x};
259
260    pDst += nRgbpPitch * nHeight;
261    *(RgbUnitx2 *)pDst = RgbUnitx2 {rgb0.v.y, rgb1.v.y};
262
263    pDst += nRgbpPitch * nHeight;
264    *(RgbUnitx2 *)pDst = RgbUnitx2 {rgb0.v.z, rgb1.v.z};
265}
266
267template<class YuvUnitx2, class Rgb, class RgbUnitx2>
268__global__ static void Yuv444ToRgbPlanarKernel(uint8_t *pYuv, int nYuvPitch, uint8_t *pRgbp, int nRgbpPitch, int nWidth, int nHeight) {
269    int x = (threadIdx.x + blockIdx.x * blockDim.x) * 2;
270    int y = (threadIdx.y + blockIdx.y * blockDim.y);
271    if (x + 1 >= nWidth || y >= nHeight) {
272        return;
273    }
274
275    uint8_t *pSrc = pYuv + x * sizeof(YuvUnitx2) / 2 + y * nYuvPitch;
276
277    YuvUnitx2 l0 = *(YuvUnitx2 *)pSrc;
278    YuvUnitx2 ch1 = *(YuvUnitx2 *)(pSrc + (nHeight * nYuvPitch));
279    YuvUnitx2 ch2 = *(YuvUnitx2 *)(pSrc + (2 * nHeight * nYuvPitch));
280
281    Rgb rgb0 = YuvToRgbForPixel<Rgb>(l0.x, ch1.x, ch2.x),
282        rgb1 = YuvToRgbForPixel<Rgb>(l0.y, ch1.y, ch2.y);
283
284
285    uint8_t *pDst = pRgbp + x * sizeof(RgbUnitx2) / 2 + y * nRgbpPitch;
286    *(RgbUnitx2 *)pDst = RgbUnitx2{ rgb0.v.x, rgb1.v.x };
287
288    pDst += nRgbpPitch * nHeight;
289    *(RgbUnitx2 *)pDst = RgbUnitx2{ rgb0.v.y, rgb1.v.y };
290
291    pDst += nRgbpPitch * nHeight;
292    *(RgbUnitx2 *)pDst = RgbUnitx2{ rgb0.v.z, rgb1.v.z };
293}
294
295template <class COLOR24>
296void Nv12ToColor24(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
297    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
298
299    YuvToRgbKernel<uchar2, COLOR24, uchar6>
300        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
301        (dpNv12, nNv12Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
302}
303
304template <class COLOR32>
305void Nv12ToColor32(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
306    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
307
308    YuvToRgbKernel<uchar2, COLOR32, uint2>
309        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
310        (dpNv12, nNv12Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
311}
312
313template <class COLOR64>
314void Nv12ToColor64(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
315    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
316    YuvToRgbKernel<uchar2, COLOR64, ulonglong2>
317        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
318        (dpNv12, nNv12Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
319}
320
321template <class COLOR32>
322void Nv16ToColor32(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
323    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
324    Yuv422ToRgbKernel<uchar2, COLOR32, uint2>
325        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
326        (dpNv16, nNv16Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
327}
328
329template <class COLOR24>
330void Nv16ToColor24(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
331    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
332    Yuv422ToRgbKernel<uchar2, COLOR24, uchar6>
333        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
334        (dpNv16, nNv16Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
335}
336
337template <class COLOR64>
338void Nv16ToColor64(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
339    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
340    Yuv422ToRgbKernel<uchar2, COLOR64, ulonglong2>
341        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
342        (dpNv16, nNv16Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
343}
344
345template <class COLOR24>
346void YUV444ToColor24(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
347    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
348    Yuv444ToRgbKernel<uchar2, COLOR24, uchar6>
349        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >>>
350        (dpYUV444, nPitch, dpBgra, nBgraPitch, nWidth, nHeight);
351}
352
353template <class COLOR32>
354void YUV444ToColor32(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
355    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
356    Yuv444ToRgbKernel<uchar2, COLOR32, uint2>
357        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >>>
358        (dpYUV444, nPitch, dpBgra, nBgraPitch, nWidth, nHeight);
359}
360
361template <class COLOR64>
362void YUV444ToColor64(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
363    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
364    Yuv444ToRgbKernel<uchar2, COLOR64, ulonglong2>
365        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >>>
366        (dpYUV444, nPitch, dpBgra, nBgraPitch, nWidth, nHeight);
367}
368
369template <class COLOR32>
370void P016ToColor32(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
371    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
372    YuvToRgbKernel<ushort2, COLOR32, uint2>
373        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
374        (dpP016, nP016Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
375}
376
377template <class COLOR24>
378void P016ToColor24(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
379    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
380    YuvToRgbKernel<ushort2, COLOR24, uchar6>
381        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
382        (dpP016, nP016Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
383}
384
385template <class COLOR64>
386void P016ToColor64(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
387    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
388    YuvToRgbKernel<ushort2, COLOR64, ulonglong2>
389        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
390        (dpP016, nP016Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
391}
392
393template <class COLOR32>
394void P216ToColor32(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
395    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
396    Yuv422ToRgbKernel<ushort2, COLOR32, uint2>
397        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
398        (dpP216, nP216Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
399}
400
401template <class COLOR24>
402void P216ToColor24(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
403    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
404    Yuv422ToRgbKernel<ushort2, COLOR24, uchar6>
405        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
406        (dpP216, nP216Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
407}
408
409template <class COLOR64>
410void P216ToColor64(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
411    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
412    Yuv422ToRgbKernel<ushort2, COLOR64, ulonglong2>
413        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
414        (dpP216, nP216Pitch, dpBgra, nBgraPitch, nWidth, nHeight);
415}
416
417template <class COLOR32>
418void YUV444P16ToColor32(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
419    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
420    Yuv444ToRgbKernel<ushort2, COLOR32, uint2>
421        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >>>
422        (dpYUV444, nPitch, dpBgra, nBgraPitch, nWidth, nHeight);
423}
424
425template <class COLOR24>
426void YUV444P16ToColor24(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
427    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
428    Yuv444ToRgbKernel<ushort2, COLOR24, uchar6>
429        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >>>
430        (dpYUV444, nPitch, dpBgra, nBgraPitch, nWidth, nHeight);
431}
432
433template <class COLOR64>
434void YUV444P16ToColor64(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
435    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
436    Yuv444ToRgbKernel<ushort2, COLOR64, ulonglong2>
437        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >>>
438        (dpYUV444, nPitch, dpBgra, nBgraPitch, nWidth, nHeight);
439}
440
441template <class COLOR32>
442void Nv12ToColorPlanar(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
443    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
444    YuvToRgbPlanarKernel<uchar2, COLOR32, uchar2>
445        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
446        (dpNv12, nNv12Pitch, dpBgrp, nBgrpPitch, nWidth, nHeight);
447}
448
449template <class COLOR32>
450void P016ToColorPlanar(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
451    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
452    YuvToRgbPlanarKernel<ushort2, COLOR32, uchar2>
453        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
454        (dpP016, nP016Pitch, dpBgrp, nBgrpPitch, nWidth, nHeight);
455}
456
457template <class COLOR32>
458void Nv16ToColorPlanar(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
459    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
460    Yuv422ToRgbPlanarKernel<uchar2, COLOR32, uchar2>
461        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
462        (dpNv16, nNv16Pitch, dpBgrp, nBgrpPitch, nWidth, nHeight);
463}
464
465template <class COLOR32>
466void P216ToColorPlanar(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
467    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
468    Yuv422ToRgbPlanarKernel<ushort2, COLOR32, uchar2>
469        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 1) / 2), dim3(32, 2)>>>
470        (dpP216, nP216Pitch, dpBgrp, nBgrpPitch, nWidth, nHeight);
471}
472
473template <class COLOR32>
474void YUV444ToColorPlanar(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
475    SetMatYuv2Rgb(iMatrix, 1, video_full_range);
476    Yuv444ToRgbPlanarKernel<uchar2, COLOR32, uchar2>
477        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >>>
478        (dpYUV444, nPitch, dpBgrp, nBgrpPitch, nWidth, nHeight);
479}
480
481template <class COLOR32>
482void YUV444P16ToColorPlanar(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range = 0) {
483    SetMatYuv2Rgb(iMatrix, 2, video_full_range);
484    Yuv444ToRgbPlanarKernel<ushort2, COLOR32, uchar2>
485        << <dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2), dim3(32, 2) >> >
486        (dpYUV444, nPitch, dpBgrp, nBgrpPitch, nWidth, nHeight);
487}
488
489// Explicit Instantiation
490template void Nv12ToColor24<RGB24>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
491template void Nv12ToColor24<BGR24>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
492template void Nv12ToColor32<BGRA32>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
493template void Nv12ToColor32<RGBA32>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
494template void Nv12ToColor64<BGRA64>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
495template void Nv12ToColor64<RGBA64>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
496template void Nv16ToColor32<BGRA32>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
497template void Nv16ToColor32<RGBA32>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
498template void Nv16ToColor24<BGR24>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
499template void Nv16ToColor24<RGB24>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
500template void Nv16ToColor64<BGRA64>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
501template void Nv16ToColor64<RGBA64>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
502template void YUV444ToColor32<BGRA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
503template void YUV444ToColor32<RGBA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
504template void YUV444ToColor24<BGR24>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
505template void YUV444ToColor24<RGB24>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
506template void YUV444ToColor64<BGRA64>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
507template void YUV444ToColor64<RGBA64>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
508template void P016ToColor32<BGRA32>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
509template void P016ToColor32<RGBA32>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
510template void P016ToColor24<BGR24>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
511template void P016ToColor24<RGB24>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
512template void P016ToColor64<BGRA64>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
513template void P016ToColor64<RGBA64>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
514template void P216ToColor32<BGRA32>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
515template void P216ToColor32<RGBA32>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
516template void P216ToColor24<BGR24>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
517template void P216ToColor24<RGB24>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
518template void P216ToColor64<BGRA64>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
519template void P216ToColor64<RGBA64>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
520template void YUV444P16ToColor32<BGRA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
521template void YUV444P16ToColor32<RGBA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
522template void YUV444P16ToColor24<BGR24>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
523template void YUV444P16ToColor24<RGB24>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
524template void YUV444P16ToColor64<BGRA64>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
525template void YUV444P16ToColor64<RGBA64>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgra, int nBgraPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
526template void Nv12ToColorPlanar<BGRA32>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
527template void Nv12ToColorPlanar<RGBA32>(uint8_t *dpNv12, int nNv12Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
528template void P016ToColorPlanar<BGRA32>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
529template void P016ToColorPlanar<RGBA32>(uint8_t *dpP016, int nP016Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
530template void Nv16ToColorPlanar<BGRA32>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
531template void Nv16ToColorPlanar<RGBA32>(uint8_t *dpNv16, int nNv16Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
532template void P216ToColorPlanar<BGRA32>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
533template void P216ToColorPlanar<RGBA32>(uint8_t *dpP216, int nP216Pitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
534template void YUV444ToColorPlanar<BGRA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
535template void YUV444ToColorPlanar<RGBA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
536template void YUV444P16ToColorPlanar<BGRA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
537template void YUV444P16ToColorPlanar<RGBA32>(uint8_t *dpYUV444, int nPitch, uint8_t *dpBgrp, int nBgrpPitch, int nWidth, int nHeight, int iMatrix, bool video_full_range);
538
539template<class YuvUnit, class RgbUnit>
540__device__ inline YuvUnit RgbToY(RgbUnit r, RgbUnit g, RgbUnit b) {
541    return matRgb2Yuv[0][0] * r + matRgb2Yuv[0][1] * g + matRgb2Yuv[0][2] * b + yuvOffsets[0];
542}
543
544template<class YuvUnit, class RgbUnit>
545__device__ inline YuvUnit RgbToU(RgbUnit r, RgbUnit g, RgbUnit b) {
546    return matRgb2Yuv[1][0] * r + matRgb2Yuv[1][1] * g + matRgb2Yuv[1][2] * b + yuvOffsets[1];
547}
548
549template<class YuvUnit, class RgbUnit>
550__device__ inline YuvUnit RgbToV(RgbUnit r, RgbUnit g, RgbUnit b) {
551    return matRgb2Yuv[2][0] * r + matRgb2Yuv[2][1] * g + matRgb2Yuv[2][2] * b + yuvOffsets[1];
552}
553
554template<class YuvUnitx2, class Rgb, class RgbIntx2>
555__global__ static void RgbToYuvKernel(uint8_t *pRgb, int nRgbPitch, uint8_t *pYuv, int nYuvPitch, int nWidth, int nHeight) {
556    int x = (threadIdx.x + blockIdx.x * blockDim.x) * 2;
557    int y = (threadIdx.y + blockIdx.y * blockDim.y) * 2;
558    if (x + 1 >= nWidth || y + 1 >= nHeight) {
559        return;
560    }
561
562    uint8_t *pSrc = pRgb + x * sizeof(Rgb) + y * nRgbPitch;
563    RgbIntx2 int2a = *(RgbIntx2 *)pSrc;
564    RgbIntx2 int2b = *(RgbIntx2 *)(pSrc + nRgbPitch);
565
566    Rgb rgb[4] = {int2a.x, int2a.y, int2b.x, int2b.y};
567    decltype(Rgb::c.r)
568        r = (rgb[0].c.r + rgb[1].c.r + rgb[2].c.r + rgb[3].c.r) / 4,
569        g = (rgb[0].c.g + rgb[1].c.g + rgb[2].c.g + rgb[3].c.g) / 4,
570        b = (rgb[0].c.b + rgb[1].c.b + rgb[2].c.b + rgb[3].c.b) / 4;
571
572    uint8_t *pDst = pYuv + x * sizeof(YuvUnitx2) / 2 + y * nYuvPitch;
573    *(YuvUnitx2 *)pDst = YuvUnitx2 {
574        RgbToY<decltype(YuvUnitx2::x)>(rgb[0].c.r, rgb[0].c.g, rgb[0].c.b),
575        RgbToY<decltype(YuvUnitx2::x)>(rgb[1].c.r, rgb[1].c.g, rgb[1].c.b),
576    };
577    *(YuvUnitx2 *)(pDst + nYuvPitch) = YuvUnitx2 {
578        RgbToY<decltype(YuvUnitx2::x)>(rgb[2].c.r, rgb[2].c.g, rgb[2].c.b),
579        RgbToY<decltype(YuvUnitx2::x)>(rgb[3].c.r, rgb[3].c.g, rgb[3].c.b),
580    };
581    *(YuvUnitx2 *)(pDst + (nHeight - y / 2) * nYuvPitch) = YuvUnitx2 {
582        RgbToU<decltype(YuvUnitx2::x)>(r, g, b), 
583        RgbToV<decltype(YuvUnitx2::x)>(r, g, b),
584    };
585}
586
587void Bgra64ToP016(uint8_t *dpBgra, int nBgraPitch, uint8_t *dpP016, int nP016Pitch, int nWidth, int nHeight, int iMatrix, bool video_full_range) {
588    SetMatRgb2Yuv(iMatrix, 2, video_full_range);
589    RgbToYuvKernel<ushort2, BGRA64, ulonglong2>
590        <<<dim3((nWidth + 63) / 32 / 2, (nHeight + 3) / 2 / 2), dim3(32, 2)>>>
591        (dpBgra, nBgraPitch, dpP016, nP016Pitch, nWidth, nHeight);
592}
593 
codekingpro/portable-devtools · Team Ai