codekingpro/portable-devtools
114k
1/*
2 * This copyright notice applies to this header file only:
3 *
4 * Copyright (c) 2010-2024 NVIDIA Corporation
5 *
6 * Permission is hereby granted, free of charge, to any person
7 * obtaining a copy of this software and associated documentation
8 * files (the "Software"), to deal in the Software without
9 * restriction, including without limitation the rights to use,
10 * copy, modify, merge, publish, distribute, sublicense, and/or sell
11 * copies of the software, and to permit persons to whom the
12 * software is furnished to do so, subject to the following
13 * conditions:
14 *
15 * The above copyright notice and this permission notice shall be
16 * included in all copies or substantial portions of the Software.
17 *
18 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
19 * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES
20 * OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
21 * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
22 * HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
23 * WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
24 * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
25 * OTHER DEALINGS IN THE SOFTWARE.
26 */
27
28#include "Metrics.h"
29
30__global__ void squaredErrorKernel(const uint8_t* image1, const uint8_t* image2, uint32_t width, uint32_t height, uint32_t pitch, float* sse) {
31 extern __shared__ float sdata[];
32
33 unsigned int tid = threadIdx.x;
34 unsigned int idx = blockIdx.x * blockDim.x + threadIdx.x;
35 unsigned int stride = gridDim.x * blockDim.x;
36
37 float sum = 0.0f;
38
39 // Accumulate squared error for this thread's part of the image
40 while (idx < width * height) {
41 int row = idx / width;
42 int col = idx % width;
43
44 int idx1 = row * pitch + col;
45 int idx2 = row * pitch + col;
46
47 float diff = static_cast<float>(image1[idx1]) - static_cast<float>(image2[idx2]);
48 sum += diff * diff;
49 idx += stride;
50 }
51
52 // Store the result in shared memory
53 sdata[tid] = sum;
54 __syncthreads();
55
56 // Reduce within the block
57 for (unsigned int s = blockDim.x / 2; s > 0; s >>= 1) {
58 if (tid < s) {
59 sdata[tid] += sdata[tid + s];
60 }
61 __syncthreads();
62 }
63
64 // Write the result of this block to the output array
65 if (tid == 0) {
66 atomicAdd(sse, sdata[0]);
67 }
68}
69
70void squaredError(uint8_t* image1, uint8_t* image2, uint32_t width, uint32_t height, uint32_t pitch, float &sse) {
71 int threadsPerBlock = 256;
72 int blocksPerGrid = (width * height + threadsPerBlock - 1) / threadsPerBlock;
73 int sharedMemSize = threadsPerBlock * sizeof(float);
74 float *sseDevice;
75
76 ck(cudaMalloc(&sseDevice, sizeof(float)));
77 ck(cudaMemset(sseDevice, 0, sizeof(float)));
78
79 squaredErrorKernel<<<blocksPerGrid, threadsPerBlock, sharedMemSize>>>(image1, image2, width, height, pitch, sseDevice);
80 CudaCheckError();
81 ck(cudaDeviceSynchronize());
82
83 ck(cudaMemcpy(&sse, sseDevice, sizeof(float), cudaMemcpyDeviceToHost));
84 ck(cudaFree(sseDevice));
85}
86
87void calcPSNRY(uint8_t* ref, uint8_t* dis, uint32_t width, uint32_t height, uint32_t pitch, float &psnr) {
88 float sse = 0.0f;
89 float mse = 0.0f;
90 uint32_t MAX = 255;
91
92 squaredError(ref, dis, width, height, pitch, sse);
93 mse = sse / (width * height);
94 if (mse)
95 psnr = 10.0f * log10f(MAX * MAX / mse);
96 else
97 psnr = 100.0;
98
99}