1 // This file is part of OpenCV project.
2 // It is subject to the license terms in the LICENSE file found in the top-level directory
3 // of this distribution and at http://opencv.org/license.html.
5 // Copyright (C) 2014, Itseez, Inc., all rights reserved.
6 // Third party copyrights are property of their respective owners.
10 #pragma OPENCL EXTENSION cl_amd_fp64:enable
11 #elif defined (cl_khr_fp64)
12 #pragma OPENCL EXTENSION cl_khr_fp64:enable
18 #define MAX_VAL UCHAR_MAX
20 #define MIN_VAL SCHAR_MIN
21 #define MAX_VAL SCHAR_MAX
24 #define MAX_VAL USHRT_MAX
26 #define MIN_VAL SHRT_MIN
27 #define MAX_VAL SHRT_MAX
29 #define MIN_VAL INT_MIN
30 #define MAX_VAL INT_MAX
32 #define MIN_VAL (-FLT_MAX)
33 #define MAX_VAL FLT_MAX
35 #define MIN_VAL (-DBL_MAX)
36 #define MAX_VAL DBL_MAX
40 #define INDEX_MAX UINT_MAX
43 #define loadpix(addr) *(__global const srcT *)(addr)
44 #define srcTSIZE (int)sizeof(srcT1)
46 #define loadpix(addr) vload3(0, (__global const srcT1 *)(addr))
47 #define srcTSIZE ((int)sizeof(srcT1))
51 #define CALC_MINLOC(inc) minloc = id + inc
53 #define CALC_MINLOC(inc)
57 #define CALC_MAXLOC(inc) maxloc = id + inc
59 #define CALC_MAXLOC(inc)
63 #define CALC_MIN(p, inc) \
64 if (minval > temp.p) \
70 #define CALC_MIN(p, inc)
74 #define CALC_MAX(p, inc) \
75 if (maxval < temp.p) \
81 #define CALC_MAX(p, inc)
85 #define CALC_MAX2(p) \
86 if (maxval2 < temp.p) \
92 #define CALC_P(p, inc) \
97 __kernel void minmaxloc(__global const uchar * srcptr, int src_step, int src_offset, int cols,
98 int total, int groupnum, __global uchar * dstptr
100 , __global const uchar * mask, int mask_step, int mask_offset
103 , __global const uchar * src2ptr, int src2_step, int src2_offset
107 int lid = get_local_id(0);
108 int gid = get_group_id(0);
109 int id = get_global_id(0) * kercn;
111 srcptr += src_offset;
116 src2ptr += src2_offset;
120 __local dstT1 localmem_min[WGS2_ALIGNED];
121 dstT1 minval = MAX_VAL;
123 __local uint localmem_minloc[WGS2_ALIGNED];
124 uint minloc = INDEX_MAX;
128 dstT1 maxval = MIN_VAL;
129 __local dstT1 localmem_max[WGS2_ALIGNED];
131 __local uint localmem_maxloc[WGS2_ALIGNED];
132 uint maxloc = INDEX_MAX;
136 __local dstT1 localmem_max2[WGS2_ALIGNED];
137 dstT1 maxval2 = MIN_VAL;
153 for (int grain = groupnum * WGS * kercn; id < total; id += grain)
156 #ifdef HAVE_MASK_CONT
159 mask_index = mad24(id / cols, mask_step, id % cols);
161 if (mask[mask_index])
165 src_index = mul24(id, srcTSIZE);
167 src_index = mad24(id / cols, src_step, mul24(id % cols, srcTSIZE));
169 temp = convertToDT(loadpix(srcptr + src_index));
171 temp = temp >= (dstT)(0) ? temp : -temp;
175 #ifdef HAVE_SRC2_CONT
176 src2_index = mul24(id, srcTSIZE);
178 src2_index = mad24(id / cols, src2_step, mul24(id % cols, srcTSIZE));
180 temp2 = convertToDT(loadpix(src2ptr + src2_index));
181 temp = temp > temp2 ? temp - temp2 : (temp2 - temp);
183 temp2 = temp2 >= (dstT)(0) ? temp2 : -temp2;
239 if (lid < WGS2_ALIGNED)
242 localmem_min[lid] = minval;
245 localmem_max[lid] = maxval;
248 localmem_minloc[lid] = minloc;
251 localmem_maxloc[lid] = maxloc;
254 localmem_max2[lid] = maxval2;
257 barrier(CLK_LOCAL_MEM_FENCE);
259 if (lid >= WGS2_ALIGNED && total >= WGS2_ALIGNED)
261 int lid3 = lid - WGS2_ALIGNED;
263 if (localmem_min[lid3] >= minval)
266 if (localmem_min[lid3] == minval)
267 localmem_minloc[lid3] = min(localmem_minloc[lid3], minloc);
269 localmem_minloc[lid3] = minloc,
271 localmem_min[lid3] = minval;
275 if (localmem_max[lid3] <= maxval)
278 if (localmem_max[lid3] == maxval)
279 localmem_maxloc[lid3] = min(localmem_maxloc[lid3], maxloc);
281 localmem_maxloc[lid3] = maxloc,
283 localmem_max[lid3] = maxval;
287 if (localmem_max2[lid3] < maxval2)
288 localmem_max2[lid3] = maxval2;
291 barrier(CLK_LOCAL_MEM_FENCE);
293 for (int lsize = WGS2_ALIGNED >> 1; lsize > 0; lsize >>= 1)
297 int lid2 = lsize + lid;
300 if (localmem_min[lid] >= localmem_min[lid2])
303 if (localmem_min[lid] == localmem_min[lid2])
304 localmem_minloc[lid] = min(localmem_minloc[lid2], localmem_minloc[lid]);
306 localmem_minloc[lid] = localmem_minloc[lid2],
308 localmem_min[lid] = localmem_min[lid2];
312 if (localmem_max[lid] <= localmem_max[lid2])
315 if (localmem_max[lid] == localmem_max[lid2])
316 localmem_maxloc[lid] = min(localmem_maxloc[lid2], localmem_maxloc[lid]);
318 localmem_maxloc[lid] = localmem_maxloc[lid2],
320 localmem_max[lid] = localmem_max[lid2];
324 if (localmem_max2[lid] < localmem_max2[lid2])
325 localmem_max2[lid] = localmem_max2[lid2];
328 barrier(CLK_LOCAL_MEM_FENCE);
335 *(__global dstT1 *)(dstptr + mad24(gid, (int)sizeof(dstT1), pos)) = localmem_min[0];
336 pos = mad24(groupnum, (int)sizeof(dstT1), pos);
339 *(__global dstT1 *)(dstptr + mad24(gid, (int)sizeof(dstT1), pos)) = localmem_max[0];
340 pos = mad24(groupnum, (int)sizeof(dstT1), pos);
343 *(__global uint *)(dstptr + mad24(gid, (int)sizeof(uint), pos)) = localmem_minloc[0];
344 pos = mad24(groupnum, (int)sizeof(uint), pos);
347 *(__global uint *)(dstptr + mad24(gid, (int)sizeof(uint), pos)) = localmem_maxloc[0];
349 pos = mad24(groupnum, (int)sizeof(uint), pos);
353 *(__global dstT1 *)(dstptr + mad24(gid, (int)sizeof(dstT1), pos)) = localmem_max2[0];