diff --git a/docs/2026.html b/docs/2026.html
index 5b78d6e49b..a512d4b18b 100644
--- a/docs/2026.html
+++ b/docs/2026.html
@@ -96,6 +96,7 @@
New features
C++ wrapper Simd::SynetDeconvolution32f.
C++ wrapper Simd::SynetDeconvolution16b.
C++ wrapper Simd::SynetMergedConvolution32f.
+ C++ wrapper Simd::SynetMergedConvolution16b.
Improving
diff --git a/prj/txt/DoxygenGroups.txt b/prj/txt/DoxygenGroups.txt
index d760559190..0ca1cb4b9c 100644
--- a/prj/txt/DoxygenGroups.txt
+++ b/prj/txt/DoxygenGroups.txt
@@ -93,7 +93,7 @@
/*! @ingroup cpp_types
@defgroup cpp_synet Synet Wrappers
- \short Simd::SynetAdd16b, Simd::SynetQuantizedAdd, Simd::SynetQuantizedMul, Simd::SynetGatherElements, Simd::SynetPermute, Simd::SynetInnerProduct32f, Simd::SynetInnerProduct16b, Simd::SynetQuantizedInnerProduct, Simd::SynetDeconvolution32f, Simd::SynetDeconvolution16b and Simd::SynetMergedConvolution32f classes (C++ wrappers of Synet operations).
+ \short Simd::SynetAdd16b, Simd::SynetQuantizedAdd, Simd::SynetQuantizedMul, Simd::SynetGatherElements, Simd::SynetPermute, Simd::SynetInnerProduct32f, Simd::SynetInnerProduct16b, Simd::SynetQuantizedInnerProduct, Simd::SynetDeconvolution32f, Simd::SynetDeconvolution16b, Simd::SynetMergedConvolution32f and Simd::SynetMergedConvolution16b classes (C++ wrappers of Synet operations).
*/
/*! @ingroup cpp_types
diff --git a/src/Simd/SimdLib.h b/src/Simd/SimdLib.h
index 862b561166..edea2901f5 100644
--- a/src/Simd/SimdLib.h
+++ b/src/Simd/SimdLib.h
@@ -9140,6 +9140,8 @@ extern "C"
three-convolution sequence, the source tensor is added to the final output and therefore must
have the same shape as the final destination tensor.
+ \note This function has a C++ wrapper: Simd::SynetMergedConvolution16b.
+
\param [in] batch - a batch size.
\param [in] convs - an array with convolution parameters in execution order.
\param [in] count - a number of merged convolutions. It must be 2 or 3.
@@ -9156,6 +9158,8 @@ extern "C"
\short Gets the size in bytes of the optional external temporary buffer for BF16 merged convolution.
+ \note This function has a C++ wrapper: Simd::SynetMergedConvolution16b.
+
\param [in] context - a pointer to BF16 merged convolution context. It must be created by function ::SimdSynetMergedConvolution16bInit and released by function ::SimdRelease.
\return size in bytes of the external temporary buffer passed to ::SimdSynetMergedConvolution16bForward.
*/
@@ -9167,6 +9171,8 @@ extern "C"
\short Gets the size in bytes of internal storage used by a BF16 merged convolution context.
+ \note This function has a C++ wrapper: Simd::SynetMergedConvolution16b.
+
\param [in] context - a pointer to BF16 merged convolution context. It must be created by function ::SimdSynetMergedConvolution16bInit and released by function ::SimdRelease.
\return size in bytes of internal temporary storage, reordered weights, biases and activation parameters.
*/
@@ -9178,6 +9184,8 @@ extern "C"
\short Gets a textual description of the selected BF16 merged convolution implementation.
+ \note This function has a C++ wrapper: Simd::SynetMergedConvolution16b.
+
\param [in] context - a pointer to BF16 merged convolution context. It must be created by function ::SimdSynetMergedConvolution16bInit and released by function ::SimdRelease.
\return a zero-terminated string with the selected implementation name.
*/
@@ -9189,6 +9197,8 @@ extern "C"
\short Sets FP32 weights, biases and activation parameters for BF16 merged convolution.
+ \note This function has a C++ wrapper: Simd::SynetMergedConvolution16b.
+
\param [in, out] context - a pointer to BF16 merged convolution context. It must be created by function ::SimdSynetMergedConvolution16bInit and released by function ::SimdRelease.
\param [in] weight - an array of pointers to FP32 convolution weights. The array size must be equal to the number of merged convolutions.
\param [in] bias - an array of pointers to FP32 bias arrays, one per convolution. Each pointer can be NULL.
@@ -9202,6 +9212,8 @@ extern "C"
\short Performs forward propagation through the fused BF16 merged convolution sequence.
+ \note This function has a C++ wrapper: Simd::SynetMergedConvolution16b.
+
\param [in] context - a pointer to BF16 merged convolution context. It must be created by function ::SimdSynetMergedConvolution16bInit and released by function ::SimdRelease.
\param [in] src - a pointer to the input tensor bytes. The tensor type is determined by convs[0].srcT (FP32 or BF16).
\param [out] buf - a pointer to an external temporary byte buffer. Its size in bytes is determined by function ::SimdSynetMergedConvolution16bExternalBufferSize. Can be NULL (it causes usage of internal buffer).
diff --git a/src/Simd/SimdSynet.hpp b/src/Simd/SimdSynet.hpp
index 46e8e2b97a..7d21e888ff 100644
--- a/src/Simd/SimdSynet.hpp
+++ b/src/Simd/SimdSynet.hpp
@@ -2314,6 +2314,307 @@ namespace Simd
SimdBool _add;
SimdConvolutionParameters _convs[3];
};
+
+ //-------------------------------------------------------------------------------------------------
+
+ /*! @ingroup cpp_synet
+
+ \short The SynetMergedConvolution16b class is a C++ wrapper of BF16/FP32 merged convolution.
+
+ The class wraps C API functions ::SimdSynetMergedConvolution16bInit, ::SimdSynetMergedConvolution16bExternalBufferSize,
+ ::SimdSynetMergedConvolution16bInternalBufferSize, ::SimdSynetMergedConvolution16bInfo, ::SimdSynetMergedConvolution16bSetParams and
+ ::SimdSynetMergedConvolution16bForward. It fuses a sequence of two or three NHWC convolutions into one forward call:
+ convolution + depthwise convolution, depthwise convolution + convolution, or
+ convolution + depthwise convolution + convolution. Internal convolution data is processed in BF16.
+ Source and destination tensors can be FP32 or BF16 according to the corresponding ::SimdConvolutionParameters fields.
+ Supported kernels are 1x1 or 3x3 for ordinary convolutions, 3x3, 5x5 or 7x7 for depthwise
+ convolutions; dilation must be 1 and stride must be 1, 2 or 3. If add is ::SimdTrue for a
+ three-convolution sequence, the source tensor is added to the final output and therefore must
+ have the same shape as the final destination tensor.
+
+ Call Init() and SetParams() before Forward(). Use Enable() to check that a context was created.
+ The context is released by Clear() or by the destructor.
+
+ Using example:
+ \verbatim
+ #include "Simd/SimdSynet.hpp"
+
+ int main()
+ {
+ const size_t batch = 1, srcC = 4, srcH = 8, srcW = 8, midC = 8, count = 2;
+ SimdConvolutionParameters convs[2] = {};
+
+ convs[0].srcC = srcC;
+ convs[0].srcH = srcH;
+ convs[0].srcW = srcW;
+ convs[0].srcT = SimdTensorData32f;
+ convs[0].srcF = SimdTensorFormatNhwc;
+ convs[0].dstC = midC;
+ convs[0].kernelY = 1;
+ convs[0].kernelX = 1;
+ convs[0].dilationY = 1;
+ convs[0].dilationX = 1;
+ convs[0].strideY = 1;
+ convs[0].strideX = 1;
+ convs[0].padY = 0;
+ convs[0].padX = 0;
+ convs[0].padH = 0;
+ convs[0].padW = 0;
+ convs[0].group = 1;
+ convs[0].activation = SimdConvolutionActivationIdentity;
+ convs[0].dstH = srcH;
+ convs[0].dstW = srcW;
+ convs[0].dstT = SimdTensorData32f;
+ convs[0].dstF = SimdTensorFormatNhwc;
+
+ convs[1].srcC = midC;
+ convs[1].srcH = convs[0].dstH;
+ convs[1].srcW = convs[0].dstW;
+ convs[1].srcT = SimdTensorData32f;
+ convs[1].srcF = SimdTensorFormatNhwc;
+ convs[1].dstC = midC;
+ convs[1].kernelY = 3;
+ convs[1].kernelX = 3;
+ convs[1].dilationY = 1;
+ convs[1].dilationX = 1;
+ convs[1].strideY = 1;
+ convs[1].strideX = 1;
+ convs[1].padY = 1;
+ convs[1].padX = 1;
+ convs[1].padH = 1;
+ convs[1].padW = 1;
+ convs[1].group = midC;
+ convs[1].activation = SimdConvolutionActivationIdentity;
+ convs[1].dstH = convs[1].srcH;
+ convs[1].dstW = convs[1].srcW;
+ convs[1].dstT = SimdTensorData32f;
+ convs[1].dstF = SimdTensorFormatNhwc;
+
+ std::vector src(batch * srcH * srcW * srcC);
+ std::vector weight0(convs[0].kernelY * convs[0].kernelX * convs[0].srcC * convs[0].dstC);
+ std::vector weight1(convs[1].kernelY * convs[1].kernelX * convs[1].dstC);
+ std::vector bias0(convs[0].dstC, 0.0f), bias1(convs[1].dstC, 0.0f);
+ std::vector dst(batch * convs[1].dstH * convs[1].dstW * convs[1].dstC, 0.0f);
+ const float * weight[2] = { weight0.data(), weight1.data() };
+ const float * bias[2] = { bias0.data(), bias1.data() };
+ const float * params[2] = { NULL, NULL };
+ for (size_t i = 0; i < src.size(); ++i)
+ src[i] = float(i) * 0.01f;
+ for (size_t i = 0; i < weight0.size(); ++i)
+ weight0[i] = float(i) * 0.02f;
+ for (size_t i = 0; i < weight1.size(); ++i)
+ weight1[i] = float(i) * 0.03f;
+
+ Simd::SynetMergedConvolution16b mergedConvolution;
+ mergedConvolution.Init(batch, convs, count, SimdFalse);
+ if (mergedConvolution.Enable())
+ {
+ mergedConvolution.SetParams(weight, bias, params);
+ mergedConvolution.Forward((const uint8_t*)src.data(), NULL, (uint8_t*)dst.data());
+ }
+
+ return 0;
+ }
+ \endverbatim
+ */
+ class SynetMergedConvolution16b
+ {
+ public:
+ /*!
+ Creates a new empty SynetMergedConvolution16b class.
+ */
+ SynetMergedConvolution16b()
+ : _context(NULL)
+ , _batch(0)
+ , _count(0)
+ , _add(SimdFalse)
+ {
+ SimdConvolutionParameters conv = {};
+ _convs[0] = conv;
+ _convs[1] = conv;
+ _convs[2] = conv;
+ }
+
+ /*!
+ SynetMergedConvolution16b class destructor. Releases internal context.
+ */
+ virtual ~SynetMergedConvolution16b()
+ {
+ Clear();
+ }
+
+ /*!
+ Initializes (or re-initializes) a BF16/FP32 merged convolution context.
+
+ Creates an internal context with using of function ::SimdSynetMergedConvolution16bInit.
+ The context is recreated only if batch size, convolution count, residual-add flag or
+ any convolution parameters were changed.
+
+ \note This function is a C++ wrapper for function ::SimdSynetMergedConvolution16bInit.
+
+ \param [in] batch - a batch size.
+ \param [in] convs - an array with convolution parameters in execution order.
+ \param [in] count - a number of merged convolutions. It must be 2 or 3.
+ \param [in] add - a flag that enables adding the source tensor to the final output tensor.
+ */
+ SIMD_INLINE void Init(size_t batch, const SimdConvolutionParameters * convs, size_t count, SimdBool add)
+ {
+ if (convs == NULL || count < 2 || count > 3)
+ return;
+ if (_batch != batch || _count != count || _add != add || Changed(convs, count))
+ {
+ Clear();
+ _batch = batch;
+ _count = count;
+ _add = add;
+ for (size_t i = 0; i < count; ++i)
+ _convs[i] = convs[i];
+ _context = SimdSynetMergedConvolution16bInit(_batch, _convs, _count, _add);
+ }
+ }
+
+ /*!
+ Checks that the internal merged convolution context was created.
+
+ \return true if the context exists and Forward() can be called.
+ */
+ SIMD_INLINE bool Enable() const
+ {
+ return _context != NULL;
+ }
+
+ /*!
+ Gets the size in bytes of caller-provided temporary buffer for BF16 merged convolution.
+
+ The returned value is a number of bytes. It depends on the implementation selected
+ during initialization and can be used when allocating the \a buf argument of Forward().
+ Some implementations return 1 when they do not need external temporary storage.
+
+ \note This function is a C++ wrapper for function ::SimdSynetMergedConvolution16bExternalBufferSize.
+
+ \return a number of bytes required for external temporary buffer.
+ */
+ SIMD_INLINE size_t ExternalBufferSize() const
+ {
+ return _context ? SimdSynetMergedConvolution16bExternalBufferSize(_context) : 0;
+ }
+
+ /*!
+ Gets the size in bytes of internal storage used by the merged convolution context.
+
+ The returned value reports internal temporary buffers and implementation-specific
+ reordered weights, bias or activation parameters already allocated by the context.
+
+ \note This function is a C++ wrapper for function ::SimdSynetMergedConvolution16bInternalBufferSize.
+
+ \return a number of bytes used by internal buffers.
+ */
+ SIMD_INLINE size_t InternalBufferSize() const
+ {
+ return _context ? SimdSynetMergedConvolution16bInternalBufferSize(_context) : 0;
+ }
+
+ /*!
+ Gets a short description of the selected BF16 merged convolution implementation.
+
+ The returned string contains the implementation extension and algorithm name.
+ The returned pointer is owned by the context and remains valid until the next call
+ of this function or until the context is released.
+
+ \note This function is a C++ wrapper for function ::SimdSynetMergedConvolution16bInfo.
+
+ \return a string with description of internal implementation. NULL if the context was not created.
+ */
+ SIMD_INLINE const char * Info() const
+ {
+ return _context ? SimdSynetMergedConvolution16bInfo(_context) : NULL;
+ }
+
+ /*!
+ Sets FP32 weights, biases and activation parameters for BF16 merged convolution.
+
+ This function must be called before Forward(). The \a weight array contains pointers to FP32
+ convolution weights, one per merged convolution. The selected implementation transforms
+ weights to its internal BF16/reordered representation. Bias is copied to an internal FP32
+ array; when a bias pointer is NULL, zeros are used. Activation parameters are copied or
+ expanded to the internal FP32 array according to ::SimdConvolutionActivationType.
+
+ \note This function is a C++ wrapper for function ::SimdSynetMergedConvolution16bSetParams.
+
+ \param [in] weight - an array of pointers to FP32 convolution weights. The array size must be equal to the number of merged convolutions.
+ \param [in] bias - an array of pointers to FP32 bias arrays, one per convolution. Each pointer can be NULL.
+ \param [in] params - an array of pointers to activation parameters (see ::SimdConvolutionActivationType), one per convolution. The array itself must be valid; each element can be NULL for activations that do not use parameters.
+ */
+ SIMD_INLINE void SetParams(const float * const * weight, const float * const * bias, const float * const * params)
+ {
+ if (_context)
+ SimdSynetMergedConvolution16bSetParams(_context, weight, bias, params);
+ }
+
+ /*!
+ Performs BF16/FP32 merged convolution forward propagation.
+
+ The function converts FP32 input to BF16 when the context source type is FP32, uses BF16 input
+ directly when the source type is BF16, applies the fused convolution sequence stored in the
+ context created by Init() and SetParams(), and writes FP32 or BF16 output according to the
+ last convolution destination type. The \a buf argument can be NULL (it causes usage of internal buffer).
+ If Init() was called with add equal to ::SimdTrue, the source tensor is added to the final output.
+
+ \note This function is a C++ wrapper for function ::SimdSynetMergedConvolution16bForward.
+
+ \param [in] src - a pointer to the input tensor bytes. The tensor type is determined by convs[0].srcT (FP32 or BF16).
+ \param [out] buf - a pointer to an external temporary byte buffer. Can be NULL.
+ \param [out] dst - a pointer to the output tensor bytes. The tensor type is determined by convs[count - 1].dstT (FP32 or BF16).
+ */
+ SIMD_INLINE void Forward(const uint8_t * src, uint8_t * buf, uint8_t * dst)
+ {
+ if (_context)
+ SimdSynetMergedConvolution16bForward(_context, src, buf, dst);
+ }
+
+ /*!
+ Releases internal context and clears stored merged convolution parameters.
+ */
+ SIMD_INLINE void Clear()
+ {
+ if (_context)
+ SimdRelease(_context), _context = NULL;
+ _batch = 0;
+ _count = 0;
+ _add = SimdFalse;
+ SimdConvolutionParameters conv = {};
+ _convs[0] = conv;
+ _convs[1] = conv;
+ _convs[2] = conv;
+ }
+
+ private:
+ SIMD_INLINE bool Changed(const SimdConvolutionParameters * convs, size_t count) const
+ {
+ for (size_t i = 0; i < count; ++i)
+ {
+ const SimdConvolutionParameters & conv = convs[i];
+ const SimdConvolutionParameters & prev = _convs[i];
+ if (prev.srcC != conv.srcC || prev.srcH != conv.srcH || prev.srcW != conv.srcW ||
+ prev.srcT != conv.srcT || prev.srcF != conv.srcF ||
+ prev.dstC != conv.dstC || prev.dstH != conv.dstH || prev.dstW != conv.dstW ||
+ prev.dstT != conv.dstT || prev.dstF != conv.dstF ||
+ prev.kernelY != conv.kernelY || prev.kernelX != conv.kernelX ||
+ prev.dilationY != conv.dilationY || prev.dilationX != conv.dilationX ||
+ prev.strideY != conv.strideY || prev.strideX != conv.strideX ||
+ prev.padY != conv.padY || prev.padX != conv.padX ||
+ prev.padH != conv.padH || prev.padW != conv.padW ||
+ prev.group != conv.group || prev.activation != conv.activation)
+ return true;
+ }
+ return false;
+ }
+
+ void * _context;
+ size_t _batch, _count;
+ SimdBool _add;
+ SimdConvolutionParameters _convs[3];
+ };
}
#endif
diff --git a/src/Test/TestCheckCpp.cpp b/src/Test/TestCheckCpp.cpp
index cfc3a94d8d..384206520b 100644
--- a/src/Test/TestCheckCpp.cpp
+++ b/src/Test/TestCheckCpp.cpp
@@ -36,6 +36,7 @@
#include "Simd/SimdLib.hpp"
#include "Simd/SimdFrame.hpp"
#include "Simd/SimdPyramid.hpp"
+#include "Simd/SimdDefs.h"
#include "Simd/SimdSynet.hpp"
#include
@@ -719,6 +720,105 @@ namespace Test
std::cout << "TestSynetMergedConvolution32f is failed at " << i << " : " << dst1[i] << " != " << dst2[i] << std::endl;
}
}
+
+ static void TestSynetMergedConvolution16b()
+ {
+ const size_t batch = 1, srcC = 4, srcH = 8, srcW = 8, midC = 8, count = 2;
+ SimdConvolutionParameters convs[2] = {};
+
+ convs[0].srcC = srcC;
+ convs[0].srcH = srcH;
+ convs[0].srcW = srcW;
+ convs[0].srcT = SimdTensorData32f;
+ convs[0].srcF = SimdTensorFormatNhwc;
+ convs[0].dstC = midC;
+ convs[0].kernelY = 1;
+ convs[0].kernelX = 1;
+ convs[0].dilationY = 1;
+ convs[0].dilationX = 1;
+ convs[0].strideY = 1;
+ convs[0].strideX = 1;
+ convs[0].padY = 0;
+ convs[0].padX = 0;
+ convs[0].padH = 0;
+ convs[0].padW = 0;
+ convs[0].group = 1;
+ convs[0].activation = SimdConvolutionActivationIdentity;
+ convs[0].dstH = srcH;
+ convs[0].dstW = srcW;
+ convs[0].dstT = SimdTensorData32f;
+ convs[0].dstF = SimdTensorFormatNhwc;
+
+ convs[1].srcC = midC;
+ convs[1].srcH = convs[0].dstH;
+ convs[1].srcW = convs[0].dstW;
+ convs[1].srcT = SimdTensorData32f;
+ convs[1].srcF = SimdTensorFormatNhwc;
+ convs[1].dstC = midC;
+ convs[1].kernelY = 3;
+ convs[1].kernelX = 3;
+ convs[1].dilationY = 1;
+ convs[1].dilationX = 1;
+ convs[1].strideY = 1;
+ convs[1].strideX = 1;
+ convs[1].padY = 1;
+ convs[1].padX = 1;
+ convs[1].padH = 1;
+ convs[1].padW = 1;
+ convs[1].group = midC;
+ convs[1].activation = SimdConvolutionActivationIdentity;
+ convs[1].dstH = convs[1].srcH;
+ convs[1].dstW = convs[1].srcW;
+ convs[1].dstT = SimdTensorData32f;
+ convs[1].dstF = SimdTensorFormatNhwc;
+
+ const size_t srcSize = batch * srcH * srcW * srcC;
+ const size_t weight0Size = convs[0].kernelY * convs[0].kernelX * convs[0].srcC * convs[0].dstC / convs[0].group;
+ const size_t weight1Size = convs[1].kernelY * convs[1].kernelX * convs[1].srcC * convs[1].dstC / convs[1].group;
+ const size_t dstSize = batch * convs[1].dstH * convs[1].dstW * convs[1].dstC;
+ std::vector src(srcSize), weight0(weight0Size), weight1(weight1Size);
+ std::vector bias0(convs[0].dstC, 0.1f), bias1(convs[1].dstC, 0.2f);
+ std::vector dst1(dstSize, 0.0f), dst2(dstSize, 0.0f);
+ const float* weight[2] = { weight0.data(), weight1.data() };
+ const float* bias[2] = { bias0.data(), bias1.data() };
+ const float* params[2] = { NULL, NULL };
+ for (size_t i = 0; i < src.size(); ++i)
+ src[i] = float(i) * 0.01f;
+ for (size_t i = 0; i < weight0.size(); ++i)
+ weight0[i] = float(i) * 0.02f;
+ for (size_t i = 0; i < weight1.size(); ++i)
+ weight1[i] = float(i) * 0.03f;
+
+ Simd::SynetMergedConvolution16b mergedConvolution;
+ mergedConvolution.Init(batch, convs, count, SimdFalse);
+ if (mergedConvolution.Enable())
+ {
+ mergedConvolution.SetParams(weight, bias, params);
+ mergedConvolution.Forward((const uint8_t*)src.data(), NULL, (uint8_t*)dst1.data());
+ }
+
+ void* context = SimdSynetMergedConvolution16bInit(batch, convs, count, SimdFalse);
+ if (context)
+ {
+ SimdSynetMergedConvolution16bSetParams(context, weight, bias, params);
+ SimdSynetMergedConvolution16bForward(context, (const uint8_t*)src.data(), NULL, (uint8_t*)dst2.data());
+ if (mergedConvolution.InternalBufferSize() != SimdSynetMergedConvolution16bInternalBufferSize(context))
+ std::cout << "TestSynetMergedConvolution16b is failed : InternalBufferSize mismatch" << std::endl;
+ if (mergedConvolution.ExternalBufferSize() != SimdSynetMergedConvolution16bExternalBufferSize(context))
+ std::cout << "TestSynetMergedConvolution16b is failed : ExternalBufferSize mismatch" << std::endl;
+ const char* info1 = mergedConvolution.Info();
+ const char* info2 = SimdSynetMergedConvolution16bInfo(context);
+ if ((info1 == NULL) != (info2 == NULL) || (info1 && info2 && std::strcmp(info1, info2) != 0))
+ std::cout << "TestSynetMergedConvolution16b is failed : Info mismatch" << std::endl;
+ SimdRelease(context);
+ }
+
+ for (size_t i = 0; i < dst1.size(); ++i)
+ {
+ if (dst1[i] != dst2[i])
+ std::cout << "TestSynetMergedConvolution16b is failed at " << i << " : " << dst1[i] << " != " << dst2[i] << std::endl;
+ }
+ }
#endif
void CheckCpp()
@@ -763,6 +863,7 @@ namespace Test
TestSynetDeconvolution32f();
TestSynetDeconvolution16b();
TestSynetMergedConvolution32f();
+ TestSynetMergedConvolution16b();
#endif
}
}