FazBrowse GitHub Viewer | Trending |
URL:
| Home
Tools: [Download Repo ZIP]   [Original HTTPS Page]

GitHub Viewer

/*M/////////////////////////////////////////////////////////////////////////////////////// // // IMPORTANT: READ BEFORE DOWNLOADING, COPYING, INSTALLING OR USING. // // By downloading, copying, installing or using the software you agree to this license. // If you do not agree to this license, do not download, install, // copy or use the software. // // // License Agreement // For Open Source Computer Vision Library // // Copyright (C) 2000-2008, Intel Corporation, all rights reserved. // Copyright (C) 2009-2011, Willow Garage Inc., all rights reserved. // Copyright (C) 2014-2015, Itseez Inc., all rights reserved. // Third party copyrights are property of their respective owners. // // Redistribution and use in source and binary forms, with or without modification, // are permitted provided that the following conditions are met: // // * Redistribution's of source code must retain the above copyright notice, // this list of conditions and the following disclaimer. // // * Redistribution's in binary form must reproduce the above copyright notice, // this list of conditions and the following disclaimer in the documentation // and/or other materials provided with the distribution. // // * The name of the copyright holders may not be used to endorse or promote products // derived from this software without specific prior written permission. // // This software is provided by the copyright holders and contributors "as is" and // any express or implied warranties, including, but not limited to, the implied // warranties of merchantability and fitness for a particular purpose are disclaimed. // In no event shall the Intel Corporation or contributors be liable for any direct, // indirect, incidental, special, exemplary, or consequential damages // (including, but not limited to, procurement of substitute goods or services; // loss of use, data, or profits; or business interruption) however caused // and on any theory of liability, whether in contract, strict liability, // or tort (including negligence or otherwise) arising in any way out of // the use of this software, even if advised of the possibility of such damage. // //M*/ #include #include "precomp.hpp" #include "opencl_kernels_core.hpp" #include "opencv2/core/opencl/runtime/opencl_clamdblas.hpp" #include "opencv2/core/opencl/runtime/opencl_core.hpp" #include "intel_gpu_gemm.inl.hpp" namespace cv { /****************************************************************************************\ * GEMM * \****************************************************************************************/ static void GEMM_CopyBlock( const uchar* src, size_t src_step, uchar* dst, size_t dst_step, Size size, size_t pix_size ) { int j; size.width *= (int)(pix_size / sizeof(int)); for( ; size.height--; src += src_step, dst += dst_step ) { j=0; #if CV_ENABLE_UNROLLED for( ; j 1 && n > 1 ) { _a_buf.allocate(n); a_buf = _a_buf.data(); } } if( n == 1 ) /* external product */ { cv::AutoBuffer _b_buf; T* b_buf = 0; if( a_step > 1 && a_size.height > 1 ) { _a_buf.allocate(drows); a_buf = _a_buf.data(); for( k = 0; k < drows; k++ ) a_buf[k] = a_data[a_step*k]; a_data = a_buf; } if( b_step > 1 ) { _b_buf.allocate(d_size.width); b_buf = _b_buf.data(); for( j = 0; j < d_size.width; j++ ) b_buf[j] = b_data[j*b_step]; b_data = b_buf; } for( i = 0; i < drows; i++, _c_data += c_step0, d_data += d_step ) { WT al = WT(a_data[i])*alpha; c_data = _c_data; for( j = 0; j step/sizeof(double); _alpha.re = alpha; _alpha.im = 0; _beta.re = C->data.ptr ? beta : 0; _beta.im = 0; if( CV_MAT_CN(type) == 2 ) lda /= 2, ldb /= 2, ldd /= 2; blas_func( transb, transa, &d_size.width, &d_size.height, &len, &_alpha, B->data.ptr, &ldb, A->data.ptr, &lda, &_beta, D->data.ptr, &ldd ); } } else*/ if( ((d_size.height step + j*elem_size; const uchar* _c = Cdata + i*c_step0 + j*c_step1; size_t _d_step = matD->step; dj = dn0; if( j + dj >= d_size.width || 8*(j + dj) + dj > 8*d_size.width ) dj = d_size.width - j; flags &= 15; if( dk0 < len ) { _d = d_buf; _d_step = dj*work_elem_size; } for( k = 0; k < len; k += dk ) { const uchar* _a = A.ptr() + i*a_step0 + k*a_step1; size_t _a_step = A.step; const uchar* _b = B.ptr() + k*b_step0 + j*b_step1; size_t _b_step = b_step; Size a_bl_size; dk = dk0; if( k + dk >= len || 8*(k + dk) + dk > 8*len ) dk = len - k; if( !is_a_t ) a_bl_size.width = dk, a_bl_size.height = di; else a_bl_size.width = di, a_bl_size.height = dk; if( a_buf && is_a_t ) { _a_step = dk*elem_size; GEMM_TransposeBlock( _a, A.step, a_buf, _a_step, a_bl_size, elem_size ); std::swap( a_bl_size.width, a_bl_size.height ); _a = a_buf; } if( dj < d_size.width ) { Size b_size; if( !is_b_t ) b_size.width = dj, b_size.height = dk; else b_size.width = dk, b_size.height = dj; _b_step = b_size.width*elem_size; GEMM_CopyBlock( _b, b_step, b_buf, _b_step, b_size, elem_size ); _b = b_buf; } if( dk0 < len ) blockMulFunc( _a, _a_step, _b, _b_step, _d, _d_step, a_bl_size, Size(dj,di), flags ); else singleMulFunc( _a, _a_step, _b, _b_step, _c, Cstep, _d, _d_step, a_bl_size, Size(dj,di), alpha, beta, flags ); flags |= 16; } if( dk0 < len ) storeFunc( _c, Cstep, _d, _d_step, matD->ptr(i) + j*elem_size, matD->step, Size(dj,di), alpha, beta, flags ); } } } } } template inline static void callGemmImpl(const fptype *src1, size_t src1_step, const fptype *src2, size_t src2_step, fptype alpha, const fptype *src3, size_t src3_step, fptype beta, fptype *dst, size_t dst_step, int m_a, int n_a, int n_d, int flags, int type) { CV_StaticAssert(GEMM_1_T == CV_HAL_GEMM_1_T, "Incompatible GEMM_1_T flag in HAL"); CV_StaticAssert(GEMM_2_T == CV_HAL_GEMM_2_T, "Incompatible GEMM_2_T flag in HAL"); CV_StaticAssert(GEMM_3_T == CV_HAL_GEMM_3_T, "Incompatible GEMM_3_T flag in HAL"); int b_m, b_n, c_m, c_n, m_d; if(flags & GEMM_2_T) { b_m = n_d; if(flags & GEMM_1_T ) { b_n = m_a; m_d = n_a; } else { b_n = n_a; m_d = m_a; } } else { b_n = n_d; if(flags & GEMM_1_T ) { b_m = m_a; m_d = n_a; } else { m_d = m_a; b_m = n_a; } } if(flags & GEMM_3_T) { c_m = n_d; c_n = m_d; } else { c_m = m_d; c_n = n_d; } Mat A, B, C; if(src1 != NULL) A = Mat(m_a, n_a, type, (void*)src1, src1_step); if(src2 != NULL) B = Mat(b_m, b_n, type, (void*)src2, src2_step); if(src3 != NULL && beta != 0.0) C = Mat(c_m, c_n, type, (void*)src3, src3_step); Mat D(m_d, n_d, type, (void*)dst, dst_step); gemmImpl(A, B, alpha, C, beta, D, flags); } } void cv::hal::gemm32f(const float* src1, size_t src1_step, const float* src2, size_t src2_step, float alpha, const float* src3, size_t src3_step, float beta, float* dst, size_t dst_step, int m_a, int n_a, int n_d, int flags) { CALL_HAL(gemm32f, cv_hal_gemm32f, src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags) callGemmImpl(src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags, CV_32F); } void cv::hal::gemm64f(const double* src1, size_t src1_step, const double* src2, size_t src2_step, double alpha, const double* src3, size_t src3_step, double beta, double* dst, size_t dst_step, int m_a, int n_a, int n_d, int flags) { CALL_HAL(gemm64f, cv_hal_gemm64f, src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags) callGemmImpl(src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags, CV_64F); } CV_EXPORTS void cv::hal::gemm32fc(const float* src1, size_t src1_step, const float* src2, size_t src2_step, float alpha, const float* src3, size_t src3_step, float beta, float* dst, size_t dst_step, int m_a, int n_a, int n_d, int flags) { CALL_HAL(gemm32fc, cv_hal_gemm32fc, src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags) callGemmImpl(src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags, CV_32FC2); } CV_EXPORTS void cv::hal::gemm64fc(const double* src1, size_t src1_step, const double* src2, size_t src2_step, double alpha, const double* src3, size_t src3_step, double beta, double* dst, size_t dst_step, int m_a, int n_a, int n_d, int flags) { CALL_HAL(gemm64fc, cv_hal_gemm64fc, src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags) callGemmImpl(src1, src1_step, src2, src2_step, alpha, src3, src3_step, beta, dst, dst_step, m_a, n_a, n_d, flags, CV_64FC2); } void cv::gemm( InputArray matA, InputArray matB, double alpha, InputArray matC, double beta, OutputArray _matD, int flags ) { #ifdef HAVE_CLAMDBLAS CV_OCL_RUN(ocl::haveAmdBlas() && matA.dims() 20, // since it works incorrect for small sizes ocl_gemm_amdblas(matA, matB, alpha, matC, beta, _matD, flags)) #endif #ifdef HAVE_OPENCL CV_OCL_RUN(_matD.isUMat() && matA.dims() cols, flags); else if( type == CV_64FC1 ) hal::gemm64f(A.ptr(), A.step, B.ptr(), B.step, alpha, C.ptr(), C.step, beta, DProxyPtr->ptr(), DProxyPtr->step, a_size.height, a_size.width, DProxyPtr->cols, flags); else if( type == CV_32FC2 ) hal::gemm32fc(A.ptr(), A.step, B.ptr(), B.step, static_cast(alpha), C.ptr(), C.step, static_cast(beta), DProxyPtr->ptr(), DProxyPtr->step, a_size.height, a_size.width, DProxyPtr->cols, flags); else { CV_Assert( type == CV_64FC2 ); hal::gemm64fc(A.ptr(), A.step, B.ptr(), B.step, alpha, C.ptr(), C.step, beta, D.ptr(), D.step, a_size.height, a_size.width, DProxyPtr->cols, flags); } if(DProxyPtr != &D) DProxyPtr->copyTo(D); } /****************************************************************************************\ * Transform * \****************************************************************************************/ namespace cv { template static void transform_( const T* src, T* dst, const WT* m, int len, int scn, int dcn ) { int x; if( scn == 2 && dcn == 2 ) { for( x = 0; x < len*2; x += 2 ) { WT v0 = src[x], v1 = src[x+1]; T t0 = saturate_cast(m[0]*v0 + m[1]*v1 + m[2]); T t1 = saturate_cast(m[3]*v0 + m[4]*v1 + m[5]); dst[x] = t0; dst[x+1] = t1; } } else if( scn == 3 && dcn == 3 ) { for( x = 0; x < len*3; x += 3 ) { WT v0 = src[x], v1 = src[x+1], v2 = src[x+2]; T t0 = saturate_cast(m[0]*v0 + m[1]*v1 + m[2]*v2 + m[3]); T t1 = saturate_cast(m[4]*v0 + m[5]*v1 + m[6]*v2 + m[7]); T t2 = saturate_cast(m[8]*v0 + m[9]*v1 + m[10]*v2 + m[11]); dst[x] = t0; dst[x+1] = t1; dst[x+2] = t2; } } else if( scn == 3 && dcn == 1 ) { for( x = 0; x < len; x++, src += 3 ) dst[x] = saturate_cast(m[0]*src[0] + m[1]*src[1] + m[2]*src[2] + m[3]); } else if( scn == 4 && dcn == 4 ) { for( x = 0; x < len*4; x += 4 ) { WT v0 = src[x], v1 = src[x+1], v2 = src[x+2], v3 = src[x+3]; T t0 = saturate_cast(m[0]*v0 + m[1]*v1 + m[2]*v2 + m[3]*v3 + m[4]); T t1 = saturate_cast(m[5]*v0 + m[6]*v1 + m[7]*v2 + m[8]*v3 + m[9]); dst[x] = t0; dst[x+1] = t1; t0 = saturate_cast(m[10]*v0 + m[11]*v1 + m[12]*v2 + m[13]*v3 + m[14]); t1 = saturate_cast(m[15]*v0 + m[16]*v1 + m[17]*v2 + m[18]*v3 + m[19]); dst[x+2] = t0; dst[x+3] = t1; } } else { for( x = 0; x < len; x++, src += scn, dst += dcn ) { const WT* _m = m; int j, k; for( j = 0; j < dcn; j++, _m += scn + 1 ) { WT s = _m[scn]; for( k = 0; k < scn; k++ ) s += _m[k]*src[k]; dst[j] = saturate_cast(s); } } } } #if CV_SIMD128 static inline void load3x3Matrix(const float* m, v_float32x4& m0, v_float32x4& m1, v_float32x4& m2, v_float32x4& m3) { m0 = v_float32x4(m[0], m[4], m[8], 0); m1 = v_float32x4(m[1], m[5], m[9], 0); m2 = v_float32x4(m[2], m[6], m[10], 0); m3 = v_float32x4(m[3], m[7], m[11], 0); } static inline v_int16x8 v_matmulvec(const v_int16x8 &v0, const v_int16x8 &m0, const v_int16x8 &m1, const v_int16x8 &m2, const v_int32x4 &m3, const int BITS) { // v0 : 0 b0 g0 r0 b1 g1 r1 ? v_int32x4 t0 = v_dotprod(v0, m0); // a0 b0 a1 b1 v_int32x4 t1 = v_dotprod(v0, m1); // c0 d0 c1 d1 v_int32x4 t2 = v_dotprod(v0, m2); // e0 f0 e1 f1 v_int32x4 t3 = v_setzero_s32(); v_int32x4 s0, s1, s2, s3; v_transpose4x4(t0, t1, t2, t3, s0, s1, s2, s3); s0 = s0 + s1 + m3; // B0 G0 R0 ? s2 = s2 + s3 + m3; // B1 G1 R1 ? s0 = s0 >> BITS; s2 = s2 >> BITS; v_int16x8 result = v_pack(s0, v_setzero_s32()); // B0 G0 R0 0 0 0 0 0 result = v_reinterpret_as_s16(v_reinterpret_as_s64(result) 40) | v_reinterpret_as_u8(v_reinterpret_as_u64(z3) >BITS); uchar t1 = saturate_cast((m10*v0 + m11*v1 + m12*v2 + m13)>>BITS); uchar t2 = saturate_cast((m20*v0 + m21*v1 + m22*v2 + m23)>>BITS); dst[x] = t0; dst[x+1] = t1; dst[x+2] = t2; } return; } #endif transform_(src, dst, m, len, scn, dcn); } static void transform_16u( const ushort* src, ushort* dst, const float* m, int len, int scn, int dcn ) { #if CV_SIMD128 && !defined(__aarch64__) if( hasSIMD128() && scn == 3 && dcn == 3 ) { const int nChannels = 3; const int cWidth = v_float32x4::nlanes; v_int16x8 delta = v_int16x8(0, -32768, -32768, -32768, -32768, -32768, -32768, 0); v_float32x4 m0, m1, m2, m3; load3x3Matrix(m, m0, m1, m2, m3); m3 -= v_float32x4(32768.f, 32768.f, 32768.f, 0.f); int x = 0; for( ; x eps ) isDiag = false; } } } TransformFunc func = isDiag ? getDiagTransformFunc(depth): getTransformFunc(depth); CV_Assert( func != 0 ); const Mat* arrays[] = {&src, &dst, 0}; uchar* ptrs[2]; NAryMatIterator it(arrays, ptrs); size_t i, total = it.size; for( i = 0; i < it.nplanes; i++, ++it ) func( ptrs[0], ptrs[1], (uchar*)mbuf, (int)total, scn, dcn ); } /****************************************************************************************\ * Perspective Transform * \****************************************************************************************/ namespace cv { template static void perspectiveTransform_( const T* src, T* dst, const double* m, int len, int scn, int dcn ) { const double eps = FLT_EPSILON; int i; if( scn == 2 && dcn == 2 ) { for( i = 0; i < len*2; i += 2 ) { T x = src[i], y = src[i + 1]; double w = x*m[6] + y*m[7] + m[8]; if( fabs(w) > eps ) { w = 1./w; dst[i] = (T)((x*m[0] + y*m[1] + m[2])*w); dst[i+1] = (T)((x*m[3] + y*m[4] + m[5])*w); } else dst[i] = dst[i+1] = (T)0; } } else if( scn == 3 && dcn == 3 ) { for( i = 0; i < len*3; i += 3 ) { T x = src[i], y = src[i + 1], z = src[i + 2]; double w = x*m[12] + y*m[13] + z*m[14] + m[15]; if( fabs(w) > eps ) { w = 1./w; dst[i] = (T)((x*m[0] + y*m[1] + z*m[2] + m[3]) * w); dst[i+1] = (T)((x*m[4] + y*m[5] + z*m[6] + m[7]) * w); dst[i+2] = (T)((x*m[8] + y*m[9] + z*m[10] + m[11]) * w); } else dst[i] = dst[i+1] = dst[i+2] = (T)0; } } else if( scn == 3 && dcn == 2 ) { for( i = 0; i < len; i++, src += 3, dst += 2 ) { T x = src[0], y = src[1], z = src[2]; double w = x*m[8] + y*m[9] + z*m[10] + m[11]; if( fabs(w) > eps ) { w = 1./w; dst[0] = (T)((x*m[0] + y*m[1] + z*m[2] + m[3])*w); dst[1] = (T)((x*m[4] + y*m[5] + z*m[6] + m[7])*w); } else dst[0] = dst[1] = (T)0; } } else { for( i = 0; i < len; i++, src += scn, dst += dcn ) { const double* _m = m + dcn*(scn + 1); double w = _m[scn]; int j, k; for( k = 0; k < scn; k++ ) w += _m[k]*src[k]; if( fabs(w) > eps ) { _m = m; for( j = 0; j < dcn; j++, _m += scn + 1 ) { double s = _m[scn]; for( k = 0; k < scn; k++ ) s += _m[k]*src[k]; dst[j] = (T)(s*w); } } else for( j = 0; j < dcn; j++ ) dst[j] = 0; } } } static void perspectiveTransform_32f(const float* src, float* dst, const double* m, int len, int scn, int dcn) { perspectiveTransform_(src, dst, m, len, scn, dcn); } static void perspectiveTransform_64f(const double* src, double* dst, const double* m, int len, int scn, int dcn) { perspectiveTransform_(src, dst, m, len, scn, dcn); } } void cv::perspectiveTransform( InputArray _src, OutputArray _dst, InputArray _mtx ) { CV_INSTRUMENT_REGION() Mat src = _src.getMat(), m = _mtx.getMat(); int depth = src.depth(), scn = src.channels(), dcn = m.rows-1; CV_Assert( scn + 1 == m.cols ); CV_Assert( depth == CV_32F || depth == CV_64F ); _dst.create( src.size(), CV_MAKETYPE(depth, dcn) ); Mat dst = _dst.getMat(); const int mtype = CV_64F; AutoBuffer _mbuf; double* mbuf = m.ptr(); if( !m.isContinuous() || m.type() != mtype ) { _mbuf.allocate((dcn+1)*(scn+1)); mbuf = _mbuf.data(); Mat tmp(dcn+1, scn+1, mtype, mbuf); m.convertTo(tmp, mtype); m = tmp; } TransformFunc func = depth == CV_32F ? (TransformFunc)perspectiveTransform_32f : (TransformFunc)perspectiveTransform_64f; CV_Assert( func != 0 ); const Mat* arrays[] = {&src, &dst, 0}; uchar* ptrs[2]; NAryMatIterator it(arrays, ptrs); size_t i, total = it.size; for( i = 0; i < it.nplanes; i++, ++it ) func( ptrs[0], ptrs[1], (uchar*)mbuf, (int)total, scn, dcn ); } /****************************************************************************************\ * ScaleAdd * \****************************************************************************************/ namespace cv { static void scaleAdd_32f(const float* src1, const float* src2, float* dst, int len, float* _alpha) { float alpha = *_alpha; int i = 0; #if CV_SIMD128 if (hasSIMD128()) { v_float32x4 v_alpha = v_setall_f32(alpha); const int cWidth = v_float32x4::nlanes; for (; i 0 ); Size size = src[0].size(); int type = src[0].type(); ctype = std::max(std::max(CV_MAT_DEPTH(ctype >= 0 ? ctype : type), _mean.depth()), CV_32F); Mat _data(static_cast(src.size()), size.area(), type); int i = 0; for(std::vector::iterator each = src.begin(); each != src.end(); ++each, ++i ) { CV_Assert( (*each).size() == size, (*each).type() == type ); Mat dataRow(size.height, size.width, type, _data.ptr(i)); (*each).copyTo(dataRow); } Mat mean; if( (flags & CV_COVAR_USE_AVG) != 0 ) { CV_Assert( _mean.size() == size ); if( mean.type() != ctype ) { mean = _mean.getMat(); _mean.create(mean.size(), ctype); Mat tmp = _mean.getMat(); mean.convertTo(tmp, ctype); mean = tmp; } mean = _mean.getMat().reshape(1, 1); } calcCovarMatrix( _data, _covar, mean, (flags & ~(CV_COVAR_ROWS|CV_COVAR_COLS)) | CV_COVAR_ROWS, ctype ); if( (flags & CV_COVAR_USE_AVG) == 0 ) { mean = mean.reshape(1, size.height); mean.copyTo(_mean); } return; } Mat data = _src.getMat(), mean; CV_Assert( ((flags & CV_COVAR_ROWS) != 0) ^ ((flags & CV_COVAR_COLS) != 0) ); bool takeRows = (flags & CV_COVAR_ROWS) != 0; int type = data.type(); int nsamples = takeRows ? data.rows : data.cols; CV_Assert( nsamples > 0 ); Size size = takeRows ? Size(data.cols, 1) : Size(1, data.rows); if( (flags & CV_COVAR_USE_AVG) != 0 ) { mean = _mean.getMat(); ctype = std::max(std::max(CV_MAT_DEPTH(ctype >= 0 ? ctype : type), mean.depth()), CV_32F); CV_Assert( mean.size() == size ); if( mean.type() != ctype ) { _mean.create(mean.size(), ctype); Mat tmp = _mean.getMat(); mean.convertTo(tmp, ctype); mean = tmp; } } else { ctype = std::max(CV_MAT_DEPTH(ctype >= 0 ? ctype : type), CV_32F); reduce( _src, _mean, takeRows ? 0 : 1, CV_REDUCE_AVG, ctype ); mean = _mean.getMat(); } mulTransposed( data, _covar, ((flags & CV_COVAR_NORMAL) == 0) ^ takeRows, mean, (flags & CV_COVAR_SCALE) != 0 ? 1./nsamples : 1, ctype ); } /****************************************************************************************\ * Mahalanobis * \****************************************************************************************/ double cv::Mahalanobis( InputArray _v1, InputArray _v2, InputArray _icovar ) { CV_INSTRUMENT_REGION() Mat v1 = _v1.getMat(), v2 = _v2.getMat(), icovar = _icovar.getMat(); int type = v1.type(), depth = v1.depth(); Size sz = v1.size(); int i, j, len = sz.width*sz.height*v1.channels(); AutoBuffer buf(len); double result = 0; CV_Assert( type == v2.type(), type == icovar.type(), sz == v2.size(), len == icovar.rows && len == icovar.cols ); sz.width *= v1.channels(); if( v1.isContinuous() && v2.isContinuous() ) { sz.width *= sz.height; sz.height = 1; } if( depth == CV_32F ) { const float* src1 = v1.ptr(); const float* src2 = v2.ptr(); size_t step1 = v1.step/sizeof(src1[0]); size_t step2 = v2.step/sizeof(src2[0]); double* diff = buf.data(); const float* mat = icovar.ptr(); size_t matstep = icovar.step/sizeof(mat[0]); for( ; sz.height--; src1 += step1, src2 += step2, diff += sz.width ) { for( i = 0; i < sz.width; i++ ) diff[i] = src1[i] - src2[i]; } diff = buf.data(); for( i = 0; i < len; i++, mat += matstep ) { double row_sum = 0; j = 0; #if CV_ENABLE_UNROLLED for(; j = gemm_level && src.rows >= gemm_level))) { Mat src2; const Mat* tsrc = &src; if( !delta.empty() ) { if( delta.size() == src.size() ) subtract( src, delta, src2 ); else { repeat(delta, src.rows/delta.rows, src.cols/delta.cols, src2); subtract( src, src2, src2 ); } tsrc = &src2; } gemm( *tsrc, *tsrc, scale, Mat(), 0, dst, ata ? GEMM_1_T : GEMM_2_T ); } else { MulTransposedFunc func = 0; if(stype == CV_8U && dtype == CV_32F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_8U && dtype == CV_64F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_16U && dtype == CV_32F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_16U && dtype == CV_64F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_16S && dtype == CV_32F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_16S && dtype == CV_64F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_32F && dtype == CV_32F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_32F && dtype == CV_64F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } else if(stype == CV_64F && dtype == CV_64F) { if(ata) func = MulTransposedR; else func = MulTransposedL; } if( !func ) CV_Error( CV_StsUnsupportedFormat, "" ); func( src, dst, delta, scale ); completeSymm( dst, false ); } } /****************************************************************************************\ * Dot Product * \****************************************************************************************/ namespace cv { template double dotProd_(const T* src1, const T* src2, int len) { int i = 0; double result = 0; #if CV_ENABLE_UNROLLED for( ; i 201800 || cv::ipp::getIppTopFeatures() != ippCPUID_SSE42, CV_INSTRUMENT_FUN_IPP(ippiDotProd_8u64f_C1R, src1, len*sizeof(uchar), src2, len*sizeof(uchar), ippiSize(len, 1), &r) >= 0, r); #endif int i = 0; #if CV_SIMD128 if (hasSIMD128()) { int len0 = len & -8, blockSize0 = (1

Back | FazBrowse Home | New Git URL