diff --git a/master/doctrees/environment.pickle b/master/doctrees/environment.pickle index 13b2fb1..ed85ef8 100644 Binary files a/master/doctrees/environment.pickle and b/master/doctrees/environment.pickle differ diff --git a/master/doctrees/functionlib/dsplib/activation.doctree b/master/doctrees/functionlib/dsplib/activation.doctree new file mode 100644 index 0000000..80c1aff Binary files /dev/null and b/master/doctrees/functionlib/dsplib/activation.doctree differ diff --git a/master/doctrees/functionlib/dsplib/adamweightdecay.doctree b/master/doctrees/functionlib/dsplib/adamweightdecay.doctree new file mode 100644 index 0000000..0ecedd4 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/adamweightdecay.doctree differ diff --git a/master/doctrees/functionlib/dsplib/adder.doctree b/master/doctrees/functionlib/dsplib/adder.doctree new file mode 100644 index 0000000..749b2e6 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/adder.doctree differ diff --git a/master/doctrees/functionlib/dsplib/applymomentum.doctree b/master/doctrees/functionlib/dsplib/applymomentum.doctree new file mode 100644 index 0000000..9439431 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/applymomentum.doctree differ diff --git a/master/doctrees/functionlib/dsplib/assert.doctree b/master/doctrees/functionlib/dsplib/assert.doctree new file mode 100644 index 0000000..ea6d3b4 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/assert.doctree differ diff --git a/master/doctrees/functionlib/dsplib/attention.doctree b/master/doctrees/functionlib/dsplib/attention.doctree new file mode 100644 index 0000000..5b878b8 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/attention.doctree differ diff --git a/master/doctrees/functionlib/dsplib/avgpoolinggrad.doctree b/master/doctrees/functionlib/dsplib/avgpoolinggrad.doctree new file mode 100644 index 0000000..232a32a Binary files /dev/null and b/master/doctrees/functionlib/dsplib/avgpoolinggrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/batchtospace.doctree b/master/doctrees/functionlib/dsplib/batchtospace.doctree new file mode 100644 index 0000000..abe7340 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/batchtospace.doctree differ diff --git a/master/doctrees/functionlib/dsplib/batchtospacend.doctree b/master/doctrees/functionlib/dsplib/batchtospacend.doctree new file mode 100644 index 0000000..863b26f Binary files /dev/null and b/master/doctrees/functionlib/dsplib/batchtospacend.doctree differ diff --git a/master/doctrees/functionlib/dsplib/broadcastto.doctree b/master/doctrees/functionlib/dsplib/broadcastto.doctree new file mode 100644 index 0000000..006b6a2 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/broadcastto.doctree differ diff --git a/master/doctrees/functionlib/dsplib/conv2d.doctree b/master/doctrees/functionlib/dsplib/conv2d.doctree new file mode 100644 index 0000000..e51c7ef Binary files /dev/null and b/master/doctrees/functionlib/dsplib/conv2d.doctree differ diff --git a/master/doctrees/functionlib/dsplib/conv2d_transpose.doctree b/master/doctrees/functionlib/dsplib/conv2d_transpose.doctree new file mode 100644 index 0000000..5284231 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/conv2d_transpose.doctree differ diff --git a/master/doctrees/functionlib/dsplib/conv2dbackpropfilterfusion.doctree b/master/doctrees/functionlib/dsplib/conv2dbackpropfilterfusion.doctree new file mode 100644 index 0000000..317aa9c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/conv2dbackpropfilterfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/conv2dbackpropinputfusion.doctree b/master/doctrees/functionlib/dsplib/conv2dbackpropinputfusion.doctree new file mode 100644 index 0000000..88f6490 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/conv2dbackpropinputfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/crop.doctree b/master/doctrees/functionlib/dsplib/crop.doctree new file mode 100644 index 0000000..124fc19 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/crop.doctree differ diff --git a/master/doctrees/functionlib/dsplib/crop_and_resize.doctree b/master/doctrees/functionlib/dsplib/crop_and_resize.doctree new file mode 100644 index 0000000..34c0284 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/crop_and_resize.doctree differ diff --git a/master/doctrees/functionlib/dsplib/depthtospace.doctree b/master/doctrees/functionlib/dsplib/depthtospace.doctree new file mode 100644 index 0000000..fad7b8a Binary files /dev/null and b/master/doctrees/functionlib/dsplib/depthtospace.doctree differ diff --git a/master/doctrees/functionlib/dsplib/dsplib_index.doctree b/master/doctrees/functionlib/dsplib/dsplib_index.doctree index 915c696..a7d05d7 100644 Binary files a/master/doctrees/functionlib/dsplib/dsplib_index.doctree and b/master/doctrees/functionlib/dsplib/dsplib_index.doctree differ diff --git a/master/doctrees/functionlib/dsplib/eltwise.doctree b/master/doctrees/functionlib/dsplib/eltwise.doctree new file mode 100644 index 0000000..8dd4880 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/eltwise.doctree differ diff --git a/master/doctrees/functionlib/dsplib/embeddinglookup.doctree b/master/doctrees/functionlib/dsplib/embeddinglookup.doctree new file mode 100644 index 0000000..1845acc Binary files /dev/null and b/master/doctrees/functionlib/dsplib/embeddinglookup.doctree differ diff --git a/master/doctrees/functionlib/dsplib/equal.doctree b/master/doctrees/functionlib/dsplib/equal.doctree index ba94dc2..cd5d0bf 100644 Binary files a/master/doctrees/functionlib/dsplib/equal.doctree and b/master/doctrees/functionlib/dsplib/equal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/expand_dims.doctree b/master/doctrees/functionlib/dsplib/expand_dims.doctree new file mode 100644 index 0000000..6887337 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/expand_dims.doctree differ diff --git a/master/doctrees/functionlib/dsplib/expfusion.doctree b/master/doctrees/functionlib/dsplib/expfusion.doctree new file mode 100644 index 0000000..a1a8773 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/expfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fillv2.doctree b/master/doctrees/functionlib/dsplib/fillv2.doctree new file mode 100644 index 0000000..a29189a Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fillv2.doctree differ diff --git a/master/doctrees/functionlib/dsplib/floor.doctree b/master/doctrees/functionlib/dsplib/floor.doctree new file mode 100644 index 0000000..524eb94 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/floor.doctree differ diff --git a/master/doctrees/functionlib/dsplib/floordiv.doctree b/master/doctrees/functionlib/dsplib/floordiv.doctree new file mode 100644 index 0000000..6d36978 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/floordiv.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fusedbatchnorm.doctree b/master/doctrees/functionlib/dsplib/fusedbatchnorm.doctree new file mode 100644 index 0000000..76e3305 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fusedbatchnorm.doctree differ diff --git a/master/doctrees/functionlib/dsplib/groupnormfusion.doctree b/master/doctrees/functionlib/dsplib/groupnormfusion.doctree new file mode 100644 index 0000000..d58a68b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/groupnormfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/gru.doctree b/master/doctrees/functionlib/dsplib/gru.doctree new file mode 100644 index 0000000..4b74ede Binary files /dev/null and b/master/doctrees/functionlib/dsplib/gru.doctree differ diff --git a/master/doctrees/functionlib/dsplib/leaky_relu.doctree b/master/doctrees/functionlib/dsplib/leaky_relu.doctree new file mode 100644 index 0000000..d7adaa7 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/leaky_relu.doctree differ diff --git a/master/doctrees/functionlib/dsplib/linspace.doctree b/master/doctrees/functionlib/dsplib/linspace.doctree new file mode 100644 index 0000000..d624e3e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/linspace.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lstm.doctree b/master/doctrees/functionlib/dsplib/lstm.doctree new file mode 100644 index 0000000..74cf74b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lstm.doctree differ diff --git a/master/doctrees/functionlib/dsplib/matmulfusion.doctree b/master/doctrees/functionlib/dsplib/matmulfusion.doctree new file mode 100644 index 0000000..b543602 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/matmulfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/raggedrange.doctree b/master/doctrees/functionlib/dsplib/raggedrange.doctree new file mode 100644 index 0000000..ef8853d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/raggedrange.doctree differ diff --git a/master/doctrees/functionlib/dsplib/range.doctree b/master/doctrees/functionlib/dsplib/range.doctree new file mode 100644 index 0000000..36f5946 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/range.doctree differ diff --git a/master/doctrees/functionlib/dsplib/reduce.doctree b/master/doctrees/functionlib/dsplib/reduce.doctree new file mode 100644 index 0000000..e2e1137 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/reduce.doctree differ diff --git a/master/doctrees/functionlib/dsplib/resize.doctree b/master/doctrees/functionlib/dsplib/resize.doctree new file mode 100644 index 0000000..16710de Binary files /dev/null and b/master/doctrees/functionlib/dsplib/resize.doctree differ diff --git a/master/doctrees/functionlib/dsplib/reverse_sequence.doctree b/master/doctrees/functionlib/dsplib/reverse_sequence.doctree new file mode 100644 index 0000000..6f9a5e7 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/reverse_sequence.doctree differ diff --git a/master/doctrees/functionlib/dsplib/reversev2.doctree b/master/doctrees/functionlib/dsplib/reversev2.doctree new file mode 100644 index 0000000..ed50596 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/reversev2.doctree differ diff --git a/master/doctrees/functionlib/dsplib/scalefusion.doctree b/master/doctrees/functionlib/dsplib/scalefusion.doctree new file mode 100644 index 0000000..a08a345 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/scalefusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/scatter_elements.doctree b/master/doctrees/functionlib/dsplib/scatter_elements.doctree new file mode 100644 index 0000000..ab3f051 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/scatter_elements.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sgd.doctree b/master/doctrees/functionlib/dsplib/sgd.doctree new file mode 100644 index 0000000..beacfbe Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sgd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/spacetobatch.doctree b/master/doctrees/functionlib/dsplib/spacetobatch.doctree new file mode 100644 index 0000000..d7bc9ce Binary files /dev/null and b/master/doctrees/functionlib/dsplib/spacetobatch.doctree differ diff --git a/master/doctrees/functionlib/dsplib/spacetobatchnd.doctree b/master/doctrees/functionlib/dsplib/spacetobatchnd.doctree new file mode 100644 index 0000000..14e9bb4 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/spacetobatchnd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/spacetodepth.doctree b/master/doctrees/functionlib/dsplib/spacetodepth.doctree new file mode 100644 index 0000000..5c1db74 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/spacetodepth.doctree differ diff --git a/master/doctrees/functionlib/dsplib/squeeze.doctree b/master/doctrees/functionlib/dsplib/squeeze.doctree new file mode 100644 index 0000000..b1912a0 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/squeeze.doctree differ diff --git a/master/doctrees/functionlib/dsplib/unsqueeze.doctree b/master/doctrees/functionlib/dsplib/unsqueeze.doctree new file mode 100644 index 0000000..d5477d4 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/unsqueeze.doctree differ diff --git a/master/html/_sources/functionlib/dsplib/activation.rst.txt b/master/html/_sources/functionlib/dsplib/activation.rst.txt new file mode 100644 index 0000000..c33dc8a --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/activation.rst.txt @@ -0,0 +1,314 @@ +Activation +================= + +对输入的数组中每一个元素执行激活函数计算,激活函数可选,具体函数见以下说明。 + +- ``Relu`` - 标准Relu函数。 + + .. math:: + + output_i = \max(0, input_i) + +- ``Relu6`` - 在标准Relu函数的基础上进行输出上限限制。 + + .. math:: + + output_i = \min(\max(0, input_i),6) + +- ``Clip`` - 将输入裁剪到区间 [min_val, max_val] + + .. math:: + + output_i = \min(\max(input_i, \text{min_val}), \text{max_val}) + +- ``LRelu`` - 带泄露的线性整流单元(Leaky Rectified Linear Unit),它在输入为正时保持线性,在输入为负时也保留一个很小的斜率,以避免标准 ReLU 中的“死亡神经元”问题。 + + .. math:: + + output_i = + \begin{cases} + input_i, & input_i \ge 0 \\ + \alpha \cdot input_i, & input_i < 0 + \end{cases} + +- ``Sigmoid`` - 常用的平滑非线性激活函数(又称逻辑函数),可以将任意实数映射到区间 :math:`(0, 1)`,常用于二分类问题的输出层,表示概率意义的结果。 + + .. math:: + + output_i = \frac{1}{1 + e^{-input_i}} + +- ``Tanh`` - 双曲正切激活函数(Hyperbolic Tangent),其输出范围为 :math:`(-1, 1)` 。 + + .. math:: + + output_i = \tanh(input_i) = \frac{e^{input_i} - e^{-input_i}}{e^{input_i} + e^{-input_i}} + +- ``HSigmoid`` - 硬 Sigmoid 激活函数(Hard Sigmoid),是 ``Sigmoid`` 函数的近似形式,计算简单、效率更高。 + + .. math:: + + output_i = \text{clip}\left(\frac{input_i + 3}{6}, 0, 1\right) + + 其中 ``clip(a, 0, 1)`` 表示将 ``a`` 限制在区间 :math:`[0, 1]` 内。 +- ``Swish`` - 自门控(Self-Gated)激活函数,由 Google 提出,结合了 ``Sigmoid`` 与线性特性,具有平滑且非单调的特点。 + + .. math:: + + output_i = input_i \cdot \sigma(input_i) = \frac{input_i}{1 + e^{-input_i}} + + 其中 :math:`\sigma(x)` 为标准 ``Sigmoid`` 函数。``Swish`` 在深层网络中通常表现优于 ``ReLU``。 +- ``HSwish`` - 硬 Swish 激活函数(Hard Swish),是 ``Swish`` 函数的近似形式,计算简单且在移动端模型(如 MobileNetV3)中被广泛采用。 + + .. math:: + + output_i = input_i \cdot \text{clip}\left(\frac{input_i + 3}{6}, 0, 1\right) + + 其中 ``clip(a, 0, 1)`` 表示将 ``a`` 限制在区间 :math:`[0, 1]` 内。 +- ``HardTanh`` - 硬双曲正切激活函数(Hard Tanh),是 ``Tanh`` 函数的分段线性近似形式,计算简单、梯度稳定,常用于量化或轻量网络中。 + + .. math:: + + output_i = \text{clip}(input_i, min\_val, max\_val) + + 其中 ``clip(x, min_val, max_val)`` 表示当 :math:`x < min\_val` 时输出 min_val,当 :math:`x > max\_val` 时输出 max_val,否则输出 :math:`x` 本身。 + +- ``Gelu`` - 高斯误差线性单元(Gaussian Error Linear Unit),是一种平滑的非线性激活函数,结合了 ``ReLU`` 与概率特性。 该函数支持精确计算及非近似计算模式,近似算法由 *Hendrycks & Gimpel (2016)* 提出,用以替代精确形式 :math:`output_i=x\Phi(x)` ,计算速度更快且精度损失极小。 + + .. math:: + + \begin{aligned} + output_i = + \begin{cases} + 0.5\,input_i \Bigl[ 1 + \tanh\!\Bigl( + \sqrt{\frac{2}{\pi}}\,(input_i + 0.044715\,input_i^3) + \Bigr) \Bigr], & flag = true, \\[6pt] + input_i \,\Phi(input_i) + = \tfrac{1}{2}x \Bigl[ + 1 + \mathrm{erf}\!\Bigl(\tfrac{input_i}{\sqrt{2}}\Bigr) + \Bigr], & flag = false. + \end{cases} + \end{aligned} + + + 其中 :math:`\Phi(x)` 为标准正态分布的累积分布函数。 +- ``Softplus`` - ``ReLU`` 的平滑近似形式,能在零点处保持可导性。 + + .. math:: + + output_i = + \begin{cases} + input_i, & input_i \gt 88.0 \\ + \ln(1 + e^{input_i}), & \text{otherwise} + \end{cases} + +- ``Elu`` - 在输入为正时保持线性,在输入为负时呈指数衰减,可缓解 ReLU 的“死亡神经元”问题。 + + .. math:: + + output_i = + \begin{cases} + input_i, & input_i \ge 0 \\ + \alpha (e^{input_i} - 1), & input_i < 0 + \end{cases} + + 其中 :math:`\alpha` 为超参数,通常取 :math:`\alpha = 1.0`。 +- ``Celu`` - 连续指数线性单元(Continuously Differentiable ELU),是 ``ELU`` 的改进版本,保证在零点处连续可导。 + + .. math:: + output_i = + \begin{cases} + input_i, & input_i \ge 0 \\ + \alpha (e^{\frac{input_i}{alpha}} - 1), & input_i < 0 + \end{cases} + + 其中 :math:`\alpha` 为可调超参数,用于控制负区间的平滑程度。 +- ``HardShrink`` - 硬收缩激活函数(Hard Shrinkage),用于稀疏化输出。 + + .. math:: + + output_i = + \begin{cases} + input_i, & \text{if } |input_i| > \lambda \\ + 0, & \text{otherwise} + \end{cases} + + 其中 :math:`\lambda` 为阈值常数。 + +- ``SoftShrink`` - 软收缩激活函数(Soft Shrinkage),与 ``HardShrink`` 类似,但收缩过程更加平滑。 + + .. math:: + + output_i = + \begin{cases} + input_i - \lambda, & \text{if } input_i > \lambda \\ + input_i + \lambda, & \text{if } input_i < -\lambda \\ + 0, & \text{otherwise} + \end{cases} + +- ``SoftsignOpt`` - 优化的软符号函数(Optimized Softsign),是一种平滑的压缩函数,用于将输入映射到有限区间。 + + .. math:: + + output_i = \frac{input_i}{1 + |input_i|} + +输入: + - **Input0** - 输入数据地址。 + - **length** - 数组长度。 + - **args(部分激活函数)** - 激活函数计算参数(仅适用于部分函数)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_relu_s(int8_t* Input0, int8_t* output,int length, int core_mask) +.. c:function:: void fp_relu_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_relu_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_relu6_s(int8_t* Input0, int8_t* output,int length, int core_mask) +.. c:function:: void fp_relu6_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_relu6_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_clip_s(int8_t* Input0, int8_t* output,int length, int8_t min_val, int8_t max_val, int core_mask) +.. c:function:: void fp_clip_s(float* Input0, float* output,int length, float min_val, float max_val, int core_mask) +.. c:function:: void hp_clip_s(half* Input0, half* output,int length, half min_val, half max_val, int core_mask) +.. c:function:: void i8_lrelu_s(int8_t* Input0, int8_t* output,int length, float alpha, int core_mask) +.. c:function:: void fp_lrelu_s(float* Input0, float* output,int length, float alpha, int core_mask) +.. c:function:: void hp_lrelu_s(half* Input0, half* output,int length, half alpha, int core_mask) +.. c:function:: void i8_sigmoid_s(int8_t* Input0, float* output,int length, int core_mask) +.. c:function:: void fp_sigmoid_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_sigmoid_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_tanh_s(int8_t* Input0, float* output,int length, int core_mask) +.. c:function:: void fp_tanh_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_tanh_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_hsigmoid_s(int8_t* Input0, float* output,int length, int core_mask) +.. c:function:: void fp_hsigmoid_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_hsigmoid_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_swish_s(int8_t* Input0, float* output,int length, int core_mask) +.. c:function:: void fp_swish_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_swish_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_hswish_s(int8_t* Input0, float* output,int length, int core_mask) +.. c:function:: void fp_hswish_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_hswish_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_hardtanh_s(int8_t* Input0, int8_t* output,int length, int8_t min_val, int8_t max_val, int core_mask) +.. c:function:: void fp_hardtanh_s(float* Input0, float* output,int length, float min_val, float max_val, int core_mask) +.. c:function:: void hp_hardtanh_s(half* Input0, half* output,int length, half min_val, half max_val, int core_mask) +.. c:function:: void i8_gelu_s(int8_t* Input0, float* output,int length, int approximate, int core_mask) +.. c:function:: void fp_gelu_s(float* Input0, float* output,int length, int approximate, int core_mask) +.. c:function:: void hp_gelu_s(half* Input0, half* output,int length, int approximate, int core_mask) +.. c:function:: void i8_softplus_s(int8_t* Input0, float* output,int length, int core_mask) +.. c:function:: void fp_softplus_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_softplus_s(half* Input0, half* output,int length, int core_mask) +.. c:function:: void i8_elu_s(int8_t* Input0, float* output,int length, float alpha, int core_mask) +.. c:function:: void fp_elu_s(float* Input0, float* output,int length, float alpha, int core_mask) +.. c:function:: void hp_elu_s(half* Input0, half* output,int length, half alpha, int core_mask) +.. c:function:: void i8_celu_s(int8_t* Input0, float* output,int length, float alpha, int core_mask) +.. c:function:: void fp_celu_s(float* Input0, float* output,int length, float alpha, int core_mask) +.. c:function:: void hp_celu_s(half* Input0, half* output,int length, half alpha, int core_mask) +.. c:function:: void i8_hardshrink_s(int8_t* Input0, int8_t* output,int length, int8_t lambd, int core_mask) +.. c:function:: void fp_hardshrink_s(float* Input0, float* output,int length, float lambd, int core_mask) +.. c:function:: void hp_hardshrink_s(half* Input0, half* output,int length, half lambd, int core_mask) +.. c:function:: void i8_softshrink_s(int8_t* Input0, int8_t* output,int length, int8_t lambd, int core_mask) +.. c:function:: void fp_softshrink_s(float* Input0, float* output,int length, float lambd, int core_mask) +.. c:function:: void hp_softshrink_s(half* Input0, half* output,int length, half lambd, int core_mask) +.. c:function:: void i8_softsignopt_s(int8_t* Input0, float* output,int length, int core_mask) +.. c:function:: void fp_softsignopt_s(float* Input0, float* output,int length, int core_mask) +.. c:function:: void hp_softsignopt_s(half* Input0, half* output,int length, int core_mask) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + fp_tanh_s(input0, output, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_relu_p(int8_t* Input0, int8_t* output,int length) +.. c:function:: void fp_relu_p(float* Input0, float* output,int length) +.. c:function:: void hp_relu_p(half* Input0, half* output,int length) +.. c:function:: void i8_relu6_p(int8_t* Input0, int8_t* output,int length) +.. c:function:: void fp_relu6_p(float* Input0, float* output,int length) +.. c:function:: void hp_relu6_p(half* Input0, half* output,int length) +.. c:function:: void i8_clip_p(int8_t* Input0, int8_t* output,int length, int8_t min_val, int8_t max_val) +.. c:function:: void fp_clip_p(float* Input0, float* output,int length, float min_val, float max_val) +.. c:function:: void hp_clip_p(half* Input0, half* output,int length, half min_val, half max_val) +.. c:function:: void i8_lrelu_p(int8_t* Input0, int8_t* output,int length, float alpha) +.. c:function:: void fp_lrelu_p(float* Input0, float* output,int length, float alpha) +.. c:function:: void hp_lrelu_p(half* Input0, half* output,int length, half alpha) +.. c:function:: void i8_sigmoid_p(int8_t* Input0, float* output,int length) +.. c:function:: void fp_sigmoid_p(float* Input0, float* output,int length) +.. c:function:: void hp_sigmoid_p(half* Input0, half* output,int length) +.. c:function:: void i8_tanh_p(int8_t* Input0, float* output,int length) +.. c:function:: void fp_tanh_p(float* Input0, float* output,int length) +.. c:function:: void hp_tanh_p(half* Input0, half* output,int length) +.. c:function:: void i8_hsigmoid_p(int8_t* Input0, float* output,int length) +.. c:function:: void fp_hsigmoid_p(float* Input0, float* output,int length) +.. c:function:: void hp_hsigmoid_p(half* Input0, half* output,int length) +.. c:function:: void i8_swish_p(int8_t* Input0, float* output,int length) +.. c:function:: void fp_swish_p(float* Input0, float* output,int length) +.. c:function:: void hp_swish_p(half* Input0, half* output,int length) +.. c:function:: void i8_hswish_p(int8_t* Input0, float* output,int length) +.. c:function:: void fp_hswish_p(float* Input0, float* output,int length) +.. c:function:: void hp_hswish_p(half* Input0, half* output,int length) +.. c:function:: void i8_hardtanh_p(int8_t* Input0, int8_t* output,int length, int8_t min_val, int8_t max_val) +.. c:function:: void fp_hardtanh_p(float* Input0, float* output,int length, float min_val, float max_val) +.. c:function:: void hp_hardtanh_p(half* Input0, half* output,int length, half min_val, half max_val) +.. c:function:: void i8_gelu_p(int8_t* Input0, float* output,int length, int approximate) +.. c:function:: void fp_gelu_p(float* Input0, float* output,int length, int approximate) +.. c:function:: void hp_gelu_p(half* Input0, half* output,int length, int approximate) +.. c:function:: void i8_softplus_p(int8_t* Input0, float* output,int length) +.. c:function:: void fp_softplus_p(float* Input0, float* output,int length) +.. c:function:: void hp_softplus_p(half* Input0, half* output,int length) +.. c:function:: void i8_elu_p(int8_t* Input0, float* output,int length, float alpha) +.. c:function:: void fp_elu_p(float* Input0, float* output,int length, float alpha) +.. c:function:: void hp_elu_p(half* Input0, half* output,int length, half alpha) +.. c:function:: void i8_celu_p(int8_t* Input0, float* output,int length, float alpha) +.. c:function:: void fp_celu_p(float* Input0, float* output,int length, float alpha) +.. c:function:: void hp_celu_p(half* Input0, half* output,int length, half alpha) +.. c:function:: void i8_hardshrink_p(int8_t* Input0, int8_t* output,int length, int8_t lambd) +.. c:function:: void fp_hardshrink_p(float* Input0, float* output,int length, float lambd) +.. c:function:: void hp_hardshrink_p(half* Input0, half* output,int length, half lambd) +.. c:function:: void i8_softshrink_p(int8_t* Input0, int8_t* output,int length, int8_t lambd) +.. c:function:: void fp_softshrink_p(float* Input0, float* output,int length, float lambd) +.. c:function:: void hp_softshrink_p(half* Input0, half* output,int length, half lambd) +.. c:function:: void i8_softsignopt_p(int8_t* Input0, float* output,int length) +.. c:function:: void fp_softsignopt_p(float* Input0, float* output,int length) +.. c:function:: void hp_softsignopt_p(half* Input0, half* output,int length) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; //input在DDR空间 + float *output = (float *)0x10004000; + int length = 1000; + fp_tanh_p(input0, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/adamweightdecay.rst.txt b/master/html/_sources/functionlib/dsplib/adamweightdecay.rst.txt new file mode 100644 index 0000000..a83c44d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/adamweightdecay.rst.txt @@ -0,0 +1,108 @@ +AdamWeightDecay +================= +对权重张量执行 Adam Weight Decay 优化更新。 + + .. math:: + + \begin{aligned} + m_t &= \beta_1 \cdot m_{t-1} + (1 - \beta_1) \cdot g_t \\ + v_t &= \beta_2 \cdot v_{t-1} + (1 - \beta_2) \cdot g_t^2 \\ + \hat{m}_t &= \frac{m_t}{\sqrt{v_t} + \epsilon} \\ + var_t &= var_{t-1} - lr \cdot (\hat{m}_t + decay \cdot var_{t-1}) + \end{aligned} + + + 输入: + - **var** - 待更新权重张量首地址。 + - **m** - 一阶动量张量首地址。 + - **v** - 二阶动量张量首地址。 + - **gradient** - 梯度张量首地址。 + - **lr** - 学习率。 + - **beta1** - 一阶动量衰减系数。 + - **beta2** - 二阶动量衰减系数。 + - **epsilon** - 数值稳定项。 + - **decay** - 权重衰减系数。 + - **start** - 参与计算的起始索引(闭区间)。 + - **end** - 参与计算的结束索引(开区间)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **var** - 原地写回更新后的权重张量。 + - **m** - 原地写回更新后的一阶动量张量。 + - **v** - 原地写回更新后的二阶动量张量。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + +**共享存储版本:** + +.. c:function:: void hp_adamweightdecay_s(half *var, half *m, half *v, const half *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int start, int end, int core_mask) +.. c:function:: void fp_adamweightdecay_s(float *var, float *m, float *v, const float *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int start, int end, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 17 + + // FT78NE 多核示例 + #include + + int main(void) { + float *var = (float *)0xA0000000; // DDR 存储 + float *m = (float *)0xB0000000; + float *v = (float *)0xC0000000; + float *gradient = (float *)0xD0000000; + int start = 0; + int end = 4096; + int core_mask = 0xff; + float lr = 1e-3f; + float beta1 = 0.9f; + float beta2 = 0.999f; + float epsilon = 1e-8f; + float decay = 1e-2f; + fp_adamweightdecay_s(var, m, v, gradient, lr, + beta1, beta2, epsilon, decay, + start, end, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_adamweightdecay_p(half *var, half *m, half *v, const half *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int length) +.. c:function:: void fp_adamweightdecay_p(float *var, float *m, float *v, const float *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int length) + + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // MT7004 单核示例 + #include + + int main(void) { + half *var = (half *)0x10000000; // L2 存储 + half *m = (half *)0x10002000; + half *v = (half *)0x10004000; + half *gradient = (half *)0x10006000; + int length = 2048; + float lr = 5e-4f; + float beta1 = 0.9f; + float beta2 = 0.999f; + float epsilon = 1e-6f; + float decay = 5e-3f; + hp_adamweightdecay_p(var, m, v, gradient, lr, + beta1, beta2, epsilon, decay, + length); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/adder.rst.txt b/master/html/_sources/functionlib/dsplib/adder.rst.txt new file mode 100644 index 0000000..75a9e7e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/adder.rst.txt @@ -0,0 +1,170 @@ +Adder +================= + +Adder是一种卷积替代算子,它使用L1距离度量(绝对差的和)代替传统卷积中的点积操作。与标准卷积不同,Adder通过计算特征与卷积核之间的绝对差的负和来进行特征提取。假定输入X,filter表示为F,它按以下公式计算: + +.. math:: + + Y(m,n,t) = - \sum_{i=0}^{d} \sum_{j=0}^{d} \sum_{k=0}^{C_{in}} |X(m+i, n+j, k) - F(i,j,k,t)| + +输入: + - **input_x** - 输入数据的地址 + - **input_w** - 输入卷积核权重的地址 + - **bias** - 输入偏置的地址 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**AdderParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct AdderParameter { + void* workspace_; // 用于存放中间计算结果 + int output_batch_; // 输出数据总批次 + int input_batch_; // 输入数据总批次 + int input_h_; // 输入数据h维度大小 + int input_w_; // 输入数据w维度大小 + int output_h_; // 输出数据h维度大小 + int output_w_; // 输出数据w维度大小 + int input_channel_; // 输入数据通道数 + int output_channel_; // 输出数据通道数 + int kernel_h_; // 卷积核h维度大小 + int kernel_w_; // 卷积核w维度大小 + int group_; // 组数 + int pad_l_; // 左填充大小 + int pad_u_; // 上填充大小 + int dilation_h_; // 卷积核h维度膨胀尺寸大小 + int dilation_w_; // 卷积核w维度膨胀尺寸大小 + int stride_h_; // 卷积核h维度步长 + int stride_w_; // 卷积核w维度步长 + int buffer_size_; // 为分块计算所分配的缓存大小 + } AdderParameter; + +输出: + - **out_y** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_adder_s(int8_t* input_x, int8_t* input_w, int8_t* out_y, int* bias, AdderParameter *param, int core_mask) +.. c:function:: void hp_adder_s(half* input_x, half* input_w, half* out_y, half* bias, AdderParameter *param, int core_mask) +.. c:function:: void fp_adder_s(float* input_x, float* input_w, float* out_y, float* bias, AdderParameter *param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 32 + + void TestAdderSMCFp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) { + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int core_num = GetCoreNum(core_mask); + float* input_data = (float*)0x88000000; + float* weight = (float*)0x89000000; + float* output_data = (float*)0x90000000; + float* bias_data = (float*)0x91000000; + AdderParameter* param = (AdderParameter*)0x92000000; + if (logic_core_id == 0) { + memcpy(bias_data, bias, sizeof(float) * output_shape[3]); + param->dilation_h_ = dilation[0]; + param->dilation_w_ = dilation[1]; + param->group_ = groups; + param->input_batch_ = input_shape[0]; + param->input_h_ = input_shape[1]; + param->input_w_ = input_shape[2]; + param->input_channel_ = input_shape[3]; + param->kernel_h_ = weight_shape[1]; + param->kernel_w_ = weight_shape[2]; + param->output_batch_ = output_shape[0]; + param->output_h_ = output_shape[1]; + param->output_w_ = output_shape[2]; + param->output_channel_ = output_shape[3]; + param->stride_h_ = stride[0]; + param->stride_w_ = stride[0]; + param->pad_u_ = padding[0]; + param->pad_l_ = padding[2]; + param->workspace_ = (float*)0x10000000; // workspace空间需分配在AM内,计算过程中会将数据搬运到workspace空间内进行计算 + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_adder_s(input_data, weight, output_data, bias_data, param, core_mask); + } + + void main(){ + int in_channel = 4; + int out_channel = 4; + int groups = 4; + int input_shape[4] = {1, 30, 30, in_channel}; // NHWC + int weight_shape[4] = {out_channel, 3, 3, in_channel / groups}; + int output_shape[4] = {1, 10, 10, out_channel}; // NHWC + int stride[2] = {2, 2}; + int padding[4] = {1, 1, 1, 1}; + int dilation[2]= {2, 2}; + float bias[4] = {0, 0, 0, 0}; + int core_mask = 0b1111; + TestAdderSMCFp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask); + } + +**私有存储版本:** + +.. c:function:: void i8_adder_p(int8_t* input_x, int8_t* input_w, int8_t* out_y, int* bias, ConvParameter *conv_param, ConvQuantParameter quant_param, int core_mask) +.. c:function:: void hp_adder_p(half* input_x, half* input_w, half* out_y, half* bias, ConvParameter *conv_param, int core_mask) +.. c:function:: void fp_adder_p(float* input_x, float* input_w, float* out_y, float* bias, ConvParameter *conv_param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 27 + + void TestAdderL2Fp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) { + float* input_data = (float*)0x10010000; // 私有存储版本地址设置在AM内 + float* weight = (float*)0x10020000; + float* output_data = (float*)0x10030000; + float* bias_data = (float*)0x10040000; + AdderParameter* param = (AdderParameter*)0x10060000; + memcpy(bias_data, bias, sizeof(float) * output_shape[3]); + param->dilation_h_ = dilation[0]; + param->dilation_w_ = dilation[1]; + param->group_ = groups; + param->input_batch_ = input_shape[0]; + param->input_h_ = input_shape[1]; + param->input_w_ = input_shape[2]; + param->input_channel_ = input_shape[3]; + param->kernel_h_ = weight_shape[1]; + param->kernel_w_ = weight_shape[2]; + param->output_batch_ = output_shape[0]; + param->output_h_ = output_shape[1]; + param->output_w_ = output_shape[2]; + param->output_channel_ = output_shape[3]; + param->stride_h_ = stride[0]; + param->stride_w_ = stride[0]; + param->pad_u_ = padding[0]; + param->pad_l_ = padding[2]; + param->workspace_ = (float*)0x10070000; + param->buffer_size_ = 2048; // 私有存储版本中,必须设置该参数,用于确定分块计算的大小 + fp_adder_p(input_data, weight, output_data, bias_data, param, core_mask); + } + + void main(){ + int in_channel = 4; + int out_channel = 4; + int groups = 4; + int input_shape[4] = {1, 30, 30, in_channel}; // NHWC + int weight_shape[4] = {out_channel, 3, 3, in_channel / groups}; + int output_shape[4] = {1, 10, 10, out_channel}; // NHWC + int stride[2] = {2, 2}; + int padding[4] = {1, 1, 1, 1}; + int dilation[2]= {2, 2}; + float bias[4] = {0, 0, 0, 0}; + int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动 + TestAdderL2Fp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/applymomentum.rst.txt b/master/html/_sources/functionlib/dsplib/applymomentum.rst.txt new file mode 100644 index 0000000..4c291ab --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/applymomentum.rst.txt @@ -0,0 +1,101 @@ +ApplyMomentum +================= +对权重张量执行 Momentum/改进动量优化更新。 + + .. math:: + + \begin{aligned} + accu_t &= moment \cdot accu_{t-1} + g_t \\ + update_t &= \begin{cases} + (accu_t \cdot moment + g_t), & \text{if nesterov = True} \\ + accu_t, & \text{otherwise} + \end{cases} \\ + weight_t &= weight_{t-1} - learning\_rate \cdot update_t + \end{aligned} + + + 输入: + - **weight** - 待更新权重张量首地址。 + - **accumulate** - 动量累积张量首地址。 + - **gradient** - 梯度张量首地址。 + - **learning_rate** - 学习率。 + - **moment** - 动量系数。 + - **nesterov** - 是否启用 Nesterov 动量。 + - **start** - 参与计算的起始索引(闭区间)。 + - **end** - 参与计算的结束索引(开区间)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **weight** - 原地写回更新后的权重张量。 + - **accumulate** - 原地写回更新后的动量张量。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + + +**共享存储版本:** + +.. c:function:: void hp_applymomentum_s(half *weight, half *accumulate, const half *gradient, float learning_rate, float moment, bool nesterov, int start, int end, int core_mask) +.. c:function:: void fp_applymomentum_s(float *weight, float *accumulate, const float *gradient, float learning_rate, float moment, bool nesterov, int start, int end, int core_mask) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 多核示例 + #include + #include + + int main(void) { + float *weight = (float *)0xA0000000; // DDR 存储 + float *accumulate = (float *)0xB0000000; + float *gradient = (float *)0xC0000000; + int start = 0; + int end = 4096; + int core_mask = 0xff; + float learning_rate = 1e-2f; + float moment = 0.99f; + bool nesterov = false; + fp_applymomentum_s(weight, accumulate, gradient, + learning_rate, moment, nesterov, + start, end, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_applymomentum_p(half *weight, half *accumulate, const half *gradient, float learning_rate, float moment, bool nesterov, int length) +.. c:function:: void fp_applymomentum_p(float *weight, float *accumulate, const float *gradient, float learning_rate, float moment, bool nesterov, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + // MT7004 单核示例 + #include + #include + + int main(void) { + half *weight = (half *)0x10000000; // L2 存储 + half *accumulate = (half *)0x10002000; + half *gradient = (half *)0x10004000; + int length = 2048; + float learning_rate = 5e-3f; + float moment = 0.9f; + bool nesterov = true; + hp_applymomentum_p(weight, accumulate, gradient, + learning_rate, moment, nesterov, + length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/assert.rst.txt b/master/html/_sources/functionlib/dsplib/assert.rst.txt new file mode 100644 index 0000000..2efdcd0 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/assert.rst.txt @@ -0,0 +1,45 @@ +Assert +================= + + +.. c:function:: void assert(bool* Input, bool* output) + +判断输入是否为 True。 + +.. math:: + + output_i = \begin{cases} + \text{True}, & \text{if } Input_i = \text{True} \\ + \text{False}, & \text{if } Input_i = \text{False} + \end{cases} + +输入: + - **Input** - 输入数据地址(布尔类型)。 + +输出: + - **output** - 计算结果地址(布尔类型)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 本算子只有一个版本 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE/MT7004 示例(共享存储) + #include + #include + + int main(int argc, char* argv[]) { + bool *input = (bool *)0xA0000000; + bool *output = (bool *)0xC0000000; + assert_s(input, output); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/attention.rst.txt b/master/html/_sources/functionlib/dsplib/attention.rst.txt new file mode 100644 index 0000000..5863eea --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/attention.rst.txt @@ -0,0 +1,79 @@ +Attention +================= + +多头缩放点积注意力机制(Scaled Dot-Product Attention) + +.. math:: + + \text{Attention}(Q, K, V) = \operatorname{softmax}\left(\frac{Q K^\top}{\sqrt{d_k}}\right) V + +输入: + - **Q** - 查询矩阵地址(行优先),形状 :math:`[B, H, L, D]` 展平。 + - **K** - 键矩阵地址(行优先),形状 :math:`[B, H, L, D]` 展平。 + - **V** - 值矩阵地址(行优先),形状 :math:`[B, H, L, D]` 展平。 + - **batch_size (B)** - 批大小。 + - **seq_len (L)** - 序列长度。 + - **head_num (H)** - 多头数量。 + - **head_dim (D)** - 每头通道维数。 + - **QK, softmax_out** - 中间缓冲区地址,容量不小于 :math:`B\times H\times L\times L`。 + - **core_mask(可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 输出地址(行优先),形状 :math:`[B, H, L, D]` 展平。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 当前实现基于 fp32;输入/中间/输出缓冲区不应重叠。 + - 内存布局为行优先(row-major)。 + +**共享存储版本:** + +.. c:function:: void fp_attention_s(float *Q, float *K, float *V, float *output, int batch_size, int seq_len, int head_num, int head_dim, float *QK, float *softmax_out, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + #include + + int main(int argc, char* argv[]) { + int B = 2, L = 128, H = 8, D = 64; + float *Q = (float *)0xA0000000; // DDR + float *K = (float *)0xA1000000; // DDR + float *V = (float *)0xA2000000; // DDR + float *O = (float *)0xA3000000; // DDR + float *QK = (float *)0xA4000000; // DDR + float *SM = (float *)0xA5000000; // DDR + int core_mask = 0xff; + fp_attention_s(Q, K, V, O, B, L, H, D, QK, SM, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_attention_p(float *Q, float *K, float *V, float *output, int batch_size, int seq_len, int head_num, int head_dim, float *QK, float *softmax_out) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + + int main(int argc, char* argv[]) { + int B = 1, L = 64, H = 4, D = 32; + float *Q = (float *)0x10000000; // L2 + float *K = (float *)0x10040000; // L2 + float *V = (float *)0x10080000; // L2 + float *O = (float *)0x100C0000; // L2 + float *QK = (float *)0x10100000; // L2 + float *SM = (float *)0x10200000; // L2 + fp_attention_p(Q, K, V, O, B, L, H, D, QK, SM); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/avgpoolinggrad.rst.txt b/master/html/_sources/functionlib/dsplib/avgpoolinggrad.rst.txt new file mode 100644 index 0000000..9a4093d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/avgpoolinggrad.rst.txt @@ -0,0 +1,120 @@ +AvgPoolingGrad +================= +根据输入梯度对平均池化前的特征图计算反向传播梯度。 + + .. math:: + + output_{b, x_h, x_w, c} += \frac{input_{b, y_h, y_w, c}}{\text{window}_h \cdot \text{window}_w} + + 其中 :math:`(y_h, y_w)` 与 :math:`(x_h, x_w)` 之间满足窗口与步长的映射关系。 + + 输入: + - **input** - 反向传播输入梯度张量首地址。 + - **output** - 反向传播输出梯度张量首地址。 + - **batch** - 批大小。 + - **output_h** - 池化输出高度。 + - **output_w** - 池化输出宽度。 + - **channel** - 通道数。 + - **input_w** - 池化输入宽度。 + - **input_h** - 池化输入高度。 + - **stride_w** - 水平步长。 + - **stride_h** - 垂直步长。 + - **pad_l** - 左侧填充大小。 + - **pad_u** - 上侧填充大小。 + - **window_w** - 池化窗口宽度。 + - **window_h** - 池化窗口高度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 原地累加平均池化反向传播梯度。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + + +**共享存储版本:** + +.. c:function:: void hp_avgpoolinggrad_s(const half *input, half *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int core_mask) +.. c:function:: void fp_avgpoolinggrad_s(const float *input, float *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int core_mask) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 20 + + // FT78NE 多核示例 + #include + + int main(void) { + const float *input = (const float *)0xA0000000; // DDR 存储 + float *output = (float *)0xB0000000; + int batch = 16; + int output_h = 7; + int output_w = 7; + int channel = 64; + int input_w = 14; + int input_h = 14; + int stride_w = 2; + int stride_h = 2; + int pad_l = 0; + int pad_u = 0; + int window_w = 2; + int window_h = 2; + int core_mask = 0xff; + fp_avgpoolinggrad_s(input, output, batch, output_h, output_w, + channel, input_w, input_h, + stride_w, stride_h, + pad_l, pad_u, + window_w, window_h, + core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_avgpoolinggrad_p(const half *input, half *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int start_idx, int end_idx) +.. c:function:: void fp_avgpoolinggrad_p(const float *input, float *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int start_idx, int end_idx) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 21 + + // MT7004 单核示例 + #include + + int main(void) { + const half *input = (const half *)0x10000000; // L2 存储 + half *output = (half *)0x10004000; + int batch = 4; + int output_h = 4; + int output_w = 4; + int channel = 128; + int input_w = 8; + int input_h = 8; + int stride_w = 2; + int stride_h = 2; + int pad_l = 0; + int pad_u = 0; + int window_w = 2; + int window_h = 2; + int start_idx = 0; + int end_idx = batch * output_h * output_w * channel; + hp_avgpoolinggrad_p(input, output, batch, output_h, output_w, + channel, input_w, input_h, + stride_w, stride_h, + pad_l, pad_u, + window_w, window_h, + start_idx, end_idx); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/batchtospace.rst.txt b/master/html/_sources/functionlib/dsplib/batchtospace.rst.txt new file mode 100644 index 0000000..251c107 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/batchtospace.rst.txt @@ -0,0 +1,96 @@ +BatchToSpace +================= +将输入张量在批维度上分块并重新分布到空间维度,同时按照 ``crops`` 对输出空间范围进行裁剪。 + + .. math:: + + \begin{aligned} + N_{\text{out}} &= \frac{N}{b_h \times b_w}, \\ + H_{\text{out}} &= b_h \times H - c_{\text{top}} - c_{\text{bottom}}, \\ + W_{\text{out}} &= b_w \times W - c_{\text{left}} - c_{\text{right}}, \\ + ext{output}[n, h, w, c] &= \text{input}[n', h', w', c] + \end{aligned} + + 其中 :math:`N, H, W, C` 分别表示输入的 batch、高度、宽度和通道数;:math:`b_h, b_w` 为 ``block_size``;:math:`c_{*}` 来源于 ``crops``;:math:`n', h', w'` 由 ``BatchToSpace`` 映射关系确定。 + + 输入: + - **input** - 输入数据地址。 + - **input_shape** - 输入形状,格式为 ``[batch, height, width, channel]``。 + - **block_size** - 分块因子,格式为 ``[block_h, block_w]``。 + - **crops** - 裁剪参数,格式为 ``[top, bottom, left, right]``。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 输出数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32、fp64、cplx64、cplx128、int16、int8、int32 数据类型。 + - MT7004 支持 fp32、fp16、cplx64、int16、int32 数据类型。 + + +**共享存储版本:** + +.. c:function:: void i8_batchtospace_s(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void i16_batchtospace_s(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void i32_batchtospace_s(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void hp_batchtospace_s(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void fp_batchtospace_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void dp_batchtospace_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void c64_batchtospace_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void c128_batchtospace_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) + + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // 多核(共享存储)示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0xA0000000; // 输入在 DDR 空间 + float *output = (float *)0xB0000000; + int input_shape[4] = {400, 2, 2, 3}; + int block_size[2] = {2, 2}; + int crops[4] = {0, 0, 0, 0}; + int core_mask = 0xff; + fp_batchtospace_s(input, output, input_shape, block_size, crops, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_batchtospace_p(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void i16_batchtospace_p(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void i32_batchtospace_p(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void hp_batchtospace_p(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void fp_batchtospace_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void dp_batchtospace_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void c64_batchtospace_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void c128_batchtospace_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // 单核(私有存储)示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0x10000000; // 输入在 L2 空间 + float *output = (float *)0x10010000; + int input_shape[4] = {400, 2, 2, 3}; + int block_size[2] = {2, 2}; + int crops[4] = {0, 0, 0, 0}; + fp_batchtospace_p(input, output, input_shape, block_size, crops, sizeof(float)); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/batchtospacend.rst.txt b/master/html/_sources/functionlib/dsplib/batchtospacend.rst.txt new file mode 100644 index 0000000..4ccc630 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/batchtospacend.rst.txt @@ -0,0 +1,91 @@ +BatchToSpaceND +================= +将输入张量的 batch 维度按 block 因子分解并重排到空间维度,随后根据 ``crops`` 参数对输出空间范围进行裁剪。 + + + - 输入形状: ``[batch, height, width, channel]`` + - ``block_size``: ``[block_h, block_w]`` + - ``crops``: ``[top, bottom, left, right]`` + + 输入: + - **input** - 输入数据地址。 + - **input_shape** - 输入形状,格式为 ``[batch, height, width, channel]``。 + - **block_size** - 分块因子,格式为 ``[block_h, block_w]``。 + - **crops** - 裁剪参数,格式为 ``[top, bottom, left, right]``。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 输出数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。 + - MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。 + + +**共享存储版本:** + +.. c:function:: void i8_batchtospacend_s(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void i16_batchtospacend_s(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void i32_batchtospacend_s(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void hp_batchtospacend_s(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void fp_batchtospacend_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void dp_batchtospacend_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void c64_batchtospacend_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) +.. c:function:: void c128_batchtospacend_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 多核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0xA0000000; // 多核版本:输入放在 DDR 地址 0xA0000000 + float *output = (float *)0xB0000000; // 多核版本:输出放在 DDR 地址 0xB0000000 + int input_shape[4] = {400, 2, 2, 3}; + int block_size[2] = {2, 2}; + int crops[4] = {0, 0, 0, 0}; + int core_mask = 0xff; + fp_batchtospacend_s(input, output, input_shape, block_size, crops, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_batchtospacend_p(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void i16_batchtospacend_p(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void i32_batchtospacend_p(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void hp_batchtospacend_p(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void fp_batchtospacend_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void dp_batchtospacend_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void c64_batchtospacend_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size) +.. c:function:: void c128_batchtospacend_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 单核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0x10000000; // 单核版本:输入放在 L2 地址 0x10000000 + float *output = (float *)0x10010000; // 单核版本:输出放在 L2 地址 0x10010000 + int input_shape[4] = {400, 2, 2, 3}; + int block_size[2] = {2, 2}; + int crops[4] = {0, 0, 0, 0}; + fp_batchtospacend_p(input, output, input_shape, block_size, crops, sizeof(float)); + return 0; + } + + diff --git a/master/html/_sources/functionlib/dsplib/broadcastto.rst.txt b/master/html/_sources/functionlib/dsplib/broadcastto.rst.txt new file mode 100644 index 0000000..8866a98 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/broadcastto.rst.txt @@ -0,0 +1,84 @@ +BroadcastTo +================= +将较小张量按 N-D 规则广播到目标形状并写入输出。 + + 输入: + - **input** - 输入数据地址。 + - **input_shape** - 输入形状数组。 + - **input_shape_size** - 输入形状长度。 + - **output_shape** - 目标输出形状数组。 + - **output_shape_size** - 输出形状长度。 + - **data_size** - 单个元素字节数(例如 sizeof(float))。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 输出数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。 + - MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。 + +**共享存储版本:** + +.. c:function:: void i8_broadcastto_s(int8_t *input, int8_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) +.. c:function:: void i16_broadcastto_s(int16_t *input, int16_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) +.. c:function:: void i32_broadcastto_s(int32_t *input, int32_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) +.. c:function:: void hp_broadcastto_s(half *input, half *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) +.. c:function:: void fp_broadcastto_s(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) +.. c:function:: void dp_broadcastto_s(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) +.. c:function:: void c64_broadcastto_s(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) +.. c:function:: void c128_broadcastto_s(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 多核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0xA0000000; // 输入在 DDR 地址 0xA0000000 + float *output = (float *)0xB0000000; // 输出在 DDR 地址 0xB0000000 + int input_shape[4] = {1, 50, 1, 1}; + int output_shape[4] = {10, 50, 20, 1}; + int core_mask = 0xff; + fp_broadcastto_s(input, output, input_shape, 4, output_shape, 4, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_broadcastto_p(int8_t *input, int8_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) +.. c:function:: void i16_broadcastto_p(int16_t *input, int16_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) +.. c:function:: void i32_broadcastto_p(int32_t *input, int32_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) +.. c:function:: void hp_broadcastto_p(half *input, half *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) +.. c:function:: void fp_broadcastto_p(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) +.. c:function:: void dp_broadcastto_p(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) +.. c:function:: void c64_broadcastto_p(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) +.. c:function:: void c128_broadcastto_p(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + // FT78NE 单核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0x10000000; // 单核版本:输入放在 L2 地址 0x10000000 + float *output = (float *)0x10020000; // 单核版本:输出放在 L2 地址 0x10020000 + int input_shape[4] = {1, 50, 1, 1}; + int output_shape[4] = {10, 50, 20, 1}; + fp_broadcastto_p(input, output, input_shape, 4, output_shape, 4, sizeof(float)); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/conv2d.rst.txt b/master/html/_sources/functionlib/dsplib/conv2d.rst.txt new file mode 100644 index 0000000..0c0e537 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/conv2d.rst.txt @@ -0,0 +1,192 @@ +Conv2d +================= + +对输入 Tensor 计算二维卷积,输入的 shape 为 :math:`(N, H_{in}, W_{in}, C_{in})`,其中 :math:`N` 为 batch size,:math:`C` 为通道数,:math:`H` 为特征图的高度,:math:`W` 为特征图的宽度。 + +根据以下公式计算输出: + +.. math:: + + out(N_i, C_{out_j}) = bias(C_{out_j}) + \sum_{k=0}^{C_{in}-1} \text{ccor}(\text{weight}(C_{out_j}, k), X(N_i, k)) + +其中,:math:`bias` 为输出偏置,:math:`\text{ccor}` 为 cross-correlation 操作,:math:`weight` 为卷积核的值,:math:`X` 为输入的特征图。 + +- :math:`i` 对应 batch 数,其范围为 :math:`[0, N-1]`,其中 :math:`N` 为输入 batch。 +- :math:`j` 对应输出通道,其范围为 :math:`[0, C_{out}-1]`,其中 :math:`C_{out}` 为输出通道数,该值也等于卷积核的个数。 +- :math:`k` 对应输入通道数,其范围为 :math:`[0, C_{in}-1]`,其中 :math:`C_{in}` 为输入通道数,该值也等于卷积核的通道数。 + +因此,上面的公式中,:math:`bias(C_{out_j})` 为第 :math:`j` 个输出通道的偏置,:math:`weight(C_{out_j}, k)` 表示第 :math:`j` 个卷积核在第 :math:`k` 个输入通道的卷积核切片,:math:`X(N_i, k)` 为特征图第 :math:`i` 个 batch 第 :math:`k` 个输入通道的切片。卷积核 shape 为 :math:`(\text{kernel_size}[0], \text{kernel_size}[1])`,其中 kernel_size[0] 和 kernel_size[1] 是卷积核的高度和宽度。若考虑到输入输出通道以及 group,则完整卷积核的 shape 为 :math:`(C_{out}, \text{kernel_size}[0], \text{kernel_size}[1], C_{in}/\text{group})`,其中 group 是分组卷积时在通道上分割输入 :math:`x` 的组数。 + +输入: + - **input_x** - 输入数据的地址 + - **input_w** - 输入卷积核权重的地址 + - **bias** - 输入偏置的地址 + - **conv_param** - 算子计算所需参数的结构体。其各成员见下述。 + - **quant_param** - 对int8类型进行量化计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**ConvParameter及ConvQuantParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct ConvParameter { + void* workspace_; // 用于存放中间计算结果 + int output_batch_; // 输出数据总批次 + int input_batch_; // 输入数据总批次 + int input_h_; // 输入数据h维度大小 + int input_w_; // 输入数据w维度大小 + int output_h_; // 输出数据h维度大小 + int output_w_; // 输出数据w维度大小 + int input_channel_; // 输入数据通道数 + int output_channel_; // 输出数据通道数 + int kernel_h_; // 卷积核h维度大小 + int kernel_w_; // 卷积核w维度大小 + int group_; // 组数 + int pad_l_; // 左填充大小 + int pad_u_; // 上填充大小 + int dilation_h_; // 卷积核h维度膨胀尺寸大小 + int dilation_w_; // 卷积核w维度膨胀尺寸大小 + int stride_h_; // 卷积核h维度步长 + int stride_w_; // 卷积核w维度步长 + int buffer_size_; // 为分块计算所分配的缓存大小 + } ConvParameter; + + typedef struct ConvQuantParameter { + int32_t* left_shift_; + int32_t* right_shift_; + int32_t* multiplier_; + int32_t* filter_zp_ptr_; + int32_t output_zp_; + int32_t mini_; + int32_t maxi_; + int per_channel_; + } ConvQuantParameter; + +输出: + - **out_y** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_conv2d_s(int8_t* input_x, int8_t* input_w, int8_t* out_y, int* bias, ConvParameter *conv_param, ConvQuantParameter quant_param, int core_mask) +.. c:function:: void hp_conv2d_s(half* input_x, half* input_w, half* out_y, half* bias, ConvParameter *conv_param, int core_mask) +.. c:function:: void fp_conv2d_s(float* input_x, float* input_w, float* out_y, float* bias, ConvParameter *conv_param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 32 + + void TestConvSMCFp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) { + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int core_num = GetCoreNum(core_mask); + float* input_data = (float*)0x88000000; + float* weight = (float*)0x89000000; + float* output_data = (float*)0x90000000; + float* bias_data = (float*)0x91000000; + ConvParameter* param = (ConvParameter*)0x92000000; + if (logic_core_id == 0) { + memcpy(bias_data, bias, sizeof(float) * output_shape[3]); + param->dilation_h_ = dilation[0]; + param->dilation_w_ = dilation[1]; + param->group_ = groups; + param->input_batch_ = input_shape[0]; + param->input_h_ = input_shape[1]; + param->input_w_ = input_shape[2]; + param->input_channel_ = input_shape[3]; + param->kernel_h_ = weight_shape[1]; + param->kernel_w_ = weight_shape[2]; + param->output_batch_ = output_shape[0]; + param->output_h_ = output_shape[1]; + param->output_w_ = output_shape[2]; + param->output_channel_ = output_shape[3]; + param->stride_h_ = stride[0]; + param->stride_w_ = stride[0]; + param->pad_u_ = padding[0]; + param->pad_l_ = padding[2]; + param->workspace_ = (float*)0x10000000; // workspace空间需分配在AM内,计算过程中会将数据搬运到workspace空间内进行计算 + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_conv2d_s(input_data, weight, output_data, bias_data, param, core_mask); + } + + void main(){ + int in_channel = 4; + int out_channel = 4; + int groups = 4; + int input_shape[4] = {1, 30, 30, in_channel}; // NHWC + int weight_shape[4] = {out_channel, 3, 3, in_channel / groups}; + int output_shape[4] = {1, 10, 10, out_channel}; // NHWC + int stride[2] = {2, 2}; + int padding[4] = {1, 1, 1, 1}; + int dilation[2]= {2, 2}; + float bias[4] = {0, 0, 0, 0}; + int core_mask = 0b1111; + TestConvSMCFp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask); + } + +**私有存储版本:** + +.. c:function:: void i8_conv2d_p(int8_t* input_x, int8_t* input_w, int8_t* out_y, int* bias, ConvParameter *conv_param, ConvQuantParameter quant_param, int core_mask) +.. c:function:: void hp_conv2d_p(half* input_x, half* input_w, half* out_y, half* bias, ConvParameter *conv_param, int core_mask) +.. c:function:: void fp_conv2d_p(float* input_x, float* input_w, float* out_y, float* bias, ConvParameter *conv_param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 27 + + void TestConvL2Fp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) { + float* input_data = (float*)0x10010000; // 私有存储版本地址设置在AM内 + float* weight = (float*)0x10020000; + float* output_data = (float*)0x10030000; + float* bias_data = (float*)0x10040000; + ConvParameter* param = (ConvParameter*)0x10060000; + memcpy(bias_data, bias, sizeof(float) * output_shape[3]); + param->dilation_h_ = dilation[0]; + param->dilation_w_ = dilation[1]; + param->group_ = groups; + param->input_batch_ = input_shape[0]; + param->input_h_ = input_shape[1]; + param->input_w_ = input_shape[2]; + param->input_channel_ = input_shape[3]; + param->kernel_h_ = weight_shape[1]; + param->kernel_w_ = weight_shape[2]; + param->output_batch_ = output_shape[0]; + param->output_h_ = output_shape[1]; + param->output_w_ = output_shape[2]; + param->output_channel_ = output_shape[3]; + param->stride_h_ = stride[0]; + param->stride_w_ = stride[0]; + param->pad_u_ = padding[0]; + param->pad_l_ = padding[2]; + param->workspace_ = (float*)0x10070000; + param->buffer_size_ = 2048; // 私有存储版本中,必须设置该参数,用于确定分块计算的大小 + fp_conv2d_p(input_data, weight, output_data, bias_data, param, core_mask); + } + + void main(){ + int in_channel = 4; + int out_channel = 4; + int groups = 4; + int input_shape[4] = {1, 30, 30, in_channel}; // NHWC + int weight_shape[4] = {out_channel, 3, 3, in_channel / groups}; + int output_shape[4] = {1, 10, 10, out_channel}; // NHWC + int stride[2] = {2, 2}; + int padding[4] = {1, 1, 1, 1}; + int dilation[2]= {2, 2}; + float bias[4] = {0, 0, 0, 0}; + int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动 + TestConvL2Fp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/conv2d_transpose.rst.txt b/master/html/_sources/functionlib/dsplib/conv2d_transpose.rst.txt new file mode 100644 index 0000000..b2a0cb0 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/conv2d_transpose.rst.txt @@ -0,0 +1,179 @@ +Conv2dTranspose +================= + +计算二维转置卷积,可以视为 Conv2d 对输入求梯度,也称为反卷积(实际不是真正的反卷积)。 + +输入的 shape 通常为 :math:`(N, H_{in}, W_{in}, C_{in})`,其中: + + - :math:`N` 是 batch size + - :math:`C_{in}` 是空间维度 + - :math:`H_{in}, W_{in}` 分别为特征层的高度和宽度 + +输入: + - **input_x** - 输入数据的地址 + - **input_w** - 输入卷积核权重的地址 + - **bias** - 输入偏置的地址 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**ConvTransposeParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct ConvTransposeParameter { + void* workspace_; // 用于存放中间计算结果 + int output_batch_; // 输出数据总批次 + int input_batch_; // 输入数据总批次 + int input_h_; // 输入数据h维度大小 + int input_w_; // 输入数据w维度大小 + int output_h_; // 输出数据h维度大小 + int output_w_; // 输出数据w维度大小 + int input_channel_; // 输入数据通道数 + int output_channel_; // 输出数据通道数 + int kernel_h_; // 卷积核h维度大小 + int kernel_w_; // 卷积核w维度大小 + int group_; // 组数 + int pad_l_; // 左填充大小 + int pad_u_; // 上填充大小 + int dilation_h_; // 卷积核h维度膨胀尺寸大小 + int dilation_w_; // 卷积核w维度膨胀尺寸大小 + int stride_h_; // 卷积核h维度步长 + int stride_w_; // 卷积核w维度步长 + int buffer_size_; // 为分块计算所分配的缓存大小 + } ConvTransposeParameter; + +输出: + - **out_y** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_convtranspose_s(int8_t* input_x, int8_t* input_w, int8_t* out_y, int* bias, ConvTransposeParameter *conv_param, int core_mask) +.. c:function:: void hp_convtranspose_s(half* input_x, half* input_w, half* out_y, half* bias, ConvTransposeParameter *conv_param, int core_mask) +.. c:function:: void fp_convtranspose_s(float* input_x, float* input_w, float* out_y, float* bias, ConvTransposeParameter *conv_param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 35 + + void TestConvTransposeSMCFp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) { + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int core_num = GetCoreNum(core_mask); + float* input_data = (float*)0x88000000; + float* weight = (float*)0x89000000; + float* output_data = (float*)0x90000000; + float* bias_data = (float*)0x91000000; + float* check = (float*)0x94000000; + ConvTransposeParameter* param = (ConvTransposeParameter*)0x92000000; + if (logic_core_id == 0) { + memcpy(bias_data, bias, sizeof(float) * output_shape[3]); + memset(output_data, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float)); + memset(check, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float)); + param->dilation_h_ = dilation[0]; + param->dilation_w_ = dilation[1]; + param->group_ = groups; + param->input_batch_ = input_shape[0]; + param->input_h_ = input_shape[1]; + param->input_w_ = input_shape[2]; + param->input_channel_ = input_shape[3]; + param->kernel_h_ = weight_shape[1]; + param->kernel_w_ = weight_shape[2]; + param->output_batch_ = output_shape[0]; + param->output_h_ = output_shape[1]; + param->output_w_ = output_shape[2]; + param->output_channel_ = output_shape[3]; + param->stride_h_ = stride[0]; + param->stride_w_ = stride[0]; + param->pad_u_ = padding[0]; + param->pad_l_ = padding[2]; + param->workspace_ = (float*)0xA0000000; + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_convtranspose_s(input_data, weight, output_data, bias_data, param, core_mask); + } + + void main(){ + int in_channel = 6; + int out_channel = 6; + int groups = 6; + int input_shape[4] = {2, 5, 7, in_channel}; // NHWC + int weight_shape[4] = {in_channel, 3, 3, out_channel / groups}; + int output_shape[4] = {2, 7, 9, out_channel}; // NHWC + int stride[2] = {1, 1}; + int padding[4] = {0, 0, 0, 0}; + int dilation[2]= {1, 1}; + float bias[] = {0, 0, 0, 0, 0, 0}; + int core_mask = 0b1111; + TestConvTransposeSMCFp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask); + } + +**私有存储版本:** + +.. c:function:: void i8_convtranspose_p(int8_t* input_x, int8_t* input_w, int8_t* out_y, int* bias, ConvTransposeParameter *conv_param, int core_mask) +.. c:function:: void hp_convtranspose_p(half* input_x, half* input_w, half* out_y, half* bias, ConvTransposeParameter *conv_param, int core_mask) +.. c:function:: void fp_convtranspose_p(float* input_x, float* input_w, float* out_y, float* bias, ConvTransposeParameter *conv_param, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 30 + + void TestConvTransposeL2Fp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) { + float* input_data = (float*)0x10000000; // 私有存储版本地址设置在AM内 + float* weight = (float*)0x10001000; + float* output_data = (float*)0x10002000; + float* bias_data = (float*)0x10003000; + float* check = (float*)0x10004000; + ConvTransposeParameter* param = (ConvTransposeParameter*)0x10005000; + memcpy(bias_data, bias, sizeof(float) * output_shape[3]); + memset(output_data, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float)); + memset(check, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float)); + param->dilation_h_ = dilation[0]; + param->dilation_w_ = dilation[1]; + param->group_ = groups; + param->input_batch_ = input_shape[0]; + param->input_h_ = input_shape[1]; + param->input_w_ = input_shape[2]; + param->input_channel_ = input_shape[3]; + param->kernel_h_ = weight_shape[1]; + param->kernel_w_ = weight_shape[2]; + param->output_batch_ = output_shape[0]; + param->output_h_ = output_shape[1]; + param->output_w_ = output_shape[2]; + param->output_channel_ = output_shape[3]; + param->stride_h_ = stride[0]; + param->stride_w_ = stride[0]; + param->pad_u_ = padding[0]; + param->pad_l_ = padding[2]; + param->workspace_ = (float*)0x10006000; + param->buffer_size_ = 1024; // 私有存储版本中,必须设置该参数,用于确定分块计算的大小 + fp_convtranspose_p(input_data, weight, output_data, bias_data, param, core_mask); + } + + void main(){ + int in_channel = 6; + int out_channel = 6; + int groups = 6; + int input_shape[4] = {2, 5, 7, in_channel}; // NHWC + int weight_shape[4] = {in_channel, 3, 3, out_channel / groups}; + int output_shape[4] = {2, 7, 9, out_channel}; // NHWC + int stride[2] = {1, 1}; + int padding[4] = {0, 0, 0, 0}; + int dilation[2]= {1, 1}; + float bias[] = {0, 0, 0, 0, 0, 0}; + int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动 + TestConvTransposeL2Fp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/conv2dbackpropfilterfusion.rst.txt b/master/html/_sources/functionlib/dsplib/conv2dbackpropfilterfusion.rst.txt new file mode 100644 index 0000000..4a6ae12 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/conv2dbackpropfilterfusion.rst.txt @@ -0,0 +1,139 @@ +Conv2DBackpropFilterFusion +========================== +计算二维卷积反向传播的权重梯度(Conv2D backprop filter fusion),支持常规卷积、Depthwise 卷积以及 1x1 优化路径,多核按批次与空间维度分块协同完成。 + + .. math:: + + dw = \text{Conv2DGradFilter}(x, dy) + + 输入: + - **dy** - 输出梯度张量首地址,形状 ``[batch, out_h, out_w, out_channel]``。 + - **x** - 正向输入张量首地址,形状 ``[batch, in_h, in_w, in_channel]``。 + - **conv_param** - 卷积参数结构体地址,包含 ``stride``、``pad``、``dilation``、``group``、输入输出维度及共享工作空间指针等信息。 + + ConvParameter 字段说明: + + - ``workspace_`` - 指向算子运行时使用的临时工作空间,需满足对齐与容量要求。 + - ``output_batch_`` - 输出梯度 ``dy`` 的批次数(通常等于输入批次数)。 + - ``input_batch_`` - 正向输入 ``x`` 的批次数,用于与 ``output_batch_`` 校验。 + - ``input_h_`` / ``input_w_`` - 正向输入特征图的高度与宽度。 + - ``output_h_`` / ``output_w_`` - 输出梯度特征图的高度与宽度。 + - ``input_channel_`` / ``output_channel_`` - 输入与输出通道数,需与 ``group_`` 配合满足整除关系。 + - ``kernel_h_`` / ``kernel_w_`` - 卷积核的高与宽。 + - ``group_`` - 组卷积数量,``group_ = 1`` 表示普通卷积。 + - ``pad_l_`` / ``pad_r_`` / ``pad_u_`` / ``pad_d_`` - 分别表示左右上下方向的填充大小。 + - ``dilation_h_`` / ``dilation_w_`` - 核心采样间隔(膨胀系数)。 + - ``stride_h_`` / ``stride_w_`` - 滑动窗口在高、宽方向的步长。 + - ``buffer_size_`` - 分配给 ``workspace_`` 的缓冲区字节数,在运行前需要正确设置。 + - ``nweights_`` - 卷积权重 ``w`` 的元素总数,用于内部分块和校验。 + + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **dw** - 卷积核梯度张量首地址,形状 ``[out_channel, in_channel/group, kernel_h, kernel_w]``。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + - 需在 ``conv_param->workspace_`` 中预先分配共享工作空间,长度不少于 ``conv_param->buffer_size_``。 + + +**共享存储版本:** + +.. c:function:: void hp_conv2dbackpropfilterfusion_s(const half *dy, const half *x, half *dw, ConvParameter *conv_param, int core_mask) +.. c:function:: void fp_conv2dbackpropfilterfusion_s(const float *dy, const float *x, float *dw, ConvParameter *conv_param, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 34 + + // FT78NE 多核示例 + #include + #include "conv_parameter.h" + + int main(void) { + const float *dy = (const float *)0xA0000000; // DDR 存储 + const float *x = (const float *)0xB0000000; + float *dw = (float *)0xC0000000; + ConvParameter *param = (ConvParameter *)0xB0001000; + // 设置 ConvParameter 字段 + param->workspace_ = (void *)0xB0002000; + param->buffer_size_ = 0x20000; + param->input_batch_ = 1; + param->input_h_ = 3; + param->input_w_ = 3; + param->input_channel_ = 4; + param->output_batch_ = 1; + param->output_h_ = 3; + param->output_w_ = 3; + param->output_channel_ = 4; + param->kernel_h_ = 2; + param->kernel_w_ = 2; + param->group_ = 2; + param->pad_u_ = 1; + param->pad_d_ = 0; + param->pad_l_ = 1; + param->pad_r_ = 0; + param->dilation_h_ = 1; + param->dilation_w_ = 1; + param->stride_h_ = 1; + param->stride_w_ = 1; + param->nweights_ = 4 * 2 * 2 * 2; // 示例值 + int core_mask = 0xff; + fp_conv2dbackpropfilterfusion_s(dy, x, dw, param, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_conv2dbackpropfilterfusion_p(const half *dy, const half *x, half *dw, ConvParameter *conv_param) +.. c:function:: void fp_conv2dbackpropfilterfusion_p(const float *dy, const float *x, float *dw, ConvParameter *conv_param) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 33 + + // MT7004 单核示例 + #include + #include "conv_parameter.h" + + int main(void) { + const half *dy = (const half *)0x10000000; // L2 存储 + const half *x = (const half *)0x10020000; + half *dw = (half *)0x10040000; + ConvParameter *param = (ConvParameter *)0x10060000; + // 设置 ConvParameter 字段 + param->workspace_ = (void *)0x10070000; + param->buffer_size_ = 0x10000; + param->input_batch_ = 1; + param->input_h_ = 3; + param->input_w_ = 3; + param->input_channel_ = 4; + param->output_batch_ = 1; + param->output_h_ = 3; + param->output_w_ = 3; + param->output_channel_ = 4; + param->kernel_h_ = 2; + param->kernel_w_ = 2; + param->group_ = 2; + param->pad_u_ = 1; + param->pad_d_ = 0; + param->pad_l_ = 1; + param->pad_r_ = 0; + param->dilation_h_ = 1; + param->dilation_w_ = 1; + param->stride_h_ = 1; + param->stride_w_ = 1; + param->nweights_ = 4 * 2 * 2 * 2; // 示例值 + hp_conv2dbackpropfilterfusion_p(dy, x, dw, param); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/conv2dbackpropinputfusion.rst.txt b/master/html/_sources/functionlib/dsplib/conv2dbackpropinputfusion.rst.txt new file mode 100644 index 0000000..b63263b --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/conv2dbackpropinputfusion.rst.txt @@ -0,0 +1,137 @@ +Conv2DBackpropInputFusion +========================= +计算二维卷积反向传播的输入梯度(Conv2D backprop input fusion),支持普通卷积、Depthwise 卷积以及 1x1 优化路径,多个核心通过核掩码协同完成批次并行。 + + .. math:: + + dx = \text{Conv2D}^\top(dy, w) + + 输入: + - **dy** - 输出梯度张量首地址,形状 ``[batch, out_h, out_w, out_channel]``。 + - **w** - 卷积权重张量首地址,形状 ``[out_channel, kernel_h, kernel_w, in_channel/group]``。 + - **conv_param** - 卷积参数结构体地址,包含 ``stride``、``pad``、``dilation``、``group``、输入输出维度、批次数及共享工作空间指针等信息。 + + ConvParameter 字段说明: + + - ``workspace_`` - 指向算子运行时使用的临时工作空间,需满足对齐与容量要求。 + - ``output_batch_`` - 输出梯度 ``dy`` 的批次数(通常等于输入批次数)。 + - ``input_batch_`` - 正向输入 ``x`` 的批次数,用于与 ``output_batch_`` 校验。 + - ``input_h_`` / ``input_w_`` - 正向输入特征图的高度与宽度。 + - ``output_h_`` / ``output_w_`` - 输出梯度特征图的高度与宽度。 + - ``input_channel_`` / ``output_channel_`` - 输入与输出通道数,需与 ``group_`` 配合满足整除关系。 + - ``kernel_h_`` / ``kernel_w_`` - 卷积核的高与宽。 + - ``group_`` - 组卷积数量,``group_ = 1`` 表示普通卷积。 + - ``pad_l_`` / ``pad_r_`` / ``pad_u_`` / ``pad_d_`` - 分别表示左右上下方向的填充大小。 + - ``dilation_h_`` / ``dilation_w_`` - 核心采样间隔(膨胀系数)。 + - ``stride_h_`` / ``stride_w_`` - 滑动窗口在高、宽方向的步长。 + - ``buffer_size_`` - 分配给 ``workspace_`` 的缓冲区字节数,在运行前需要正确设置。 + - ``nweights_`` - 卷积权重 ``w`` 的元素总数,用于内部分块和校验。 + + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **dx** - 输入梯度张量首地址,形状 ``[batch, in_h, in_w, in_channel]``。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + - 需在 ``conv_param->workspace_`` 中预先分配共享工作空间,并设置 ``conv_param->buffer_size_``。 + +**共享存储版本:** + +.. c:function:: void hp_conv2dbackpropinputfusion_s(const half *dy, const half *w, half *dx, ConvParameter *conv_param, int core_mask) +.. c:function:: void fp_conv2dbackpropinputfusion_s(const float *dy, const float *w, float *dx, ConvParameter *conv_param, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 34 + + // FT78NE 多核示例 + #include + #include "conv_parameter.h" + + int main(void) { + const float *dy = (const float *)0xA0000000; // DDR 存储 + const float *w = (const float *)0xB0000000; + float *dx = (const float *)0xC0000000; + ConvParameter *param = (ConvParameter *)0xB0001000; // 卷积参数共享区域 + // 设置 ConvParameter 字段 + param->workspace_ = (void *)0xB0002000; // 共享工作空间 + param->buffer_size_ = 0x20000; + param->input_batch_ = 1; + param->input_h_ = 5; + param->input_w_ = 5; + param->input_channel_ = 4; + param->output_batch_ = 1; + param->output_h_ = 3; + param->output_w_ = 3; + param->output_channel_ = 8; + param->kernel_h_ = 3; + param->kernel_w_ = 3; + param->group_ = 1; + param->pad_u_ = 1; + param->pad_d_ = 1; + param->pad_l_ = 1; + param->pad_r_ = 1; + param->dilation_h_ = 1; + param->dilation_w_ = 1; + param->stride_h_ = 2; + param->stride_w_ = 2; + param->nweights_ = 8 * 3 * 3 * 4; // 示例值 + int core_mask = 0xff; + fp_conv2dbackpropinputfusion_s(dy, w, dx, param, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void hp_conv2dbackpropinputfusion_p(const half *dy, const half *w, half *dx, ConvParameter *conv_param) +.. c:function:: void fp_conv2dbackpropinputfusion_p(const float *dy, const float *w, float *dx, ConvParameter *conv_param) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 33 + + // MT7004 单核示例 + #include + #include "conv_parameter.h" + + int main(void) { + const half *dy = (const half *)0x10000000; // L2 存储 + const half *w = (const half *)0x10020000; + half *dx = (half *)0x10040000; + ConvParameter *param = (ConvParameter *)0x10060000; + // 设置 ConvParameter 字段 + param->workspace_ = (void *)0x10070000; + param->buffer_size_ = 0x10000; + param->input_batch_ = 1; + param->input_h_ = 5; + param->input_w_ = 5; + param->input_channel_ = 4; + param->output_batch_ = 1; + param->output_h_ = 3; + param->output_w_ = 3; + param->output_channel_ = 8; + param->kernel_h_ = 3; + param->kernel_w_ = 3; + param->group_ = 1; + param->pad_u_ = 1; + param->pad_d_ = 1; + param->pad_l_ = 1; + param->pad_r_ = 1; + param->dilation_h_ = 1; + param->dilation_w_ = 1; + param->stride_h_ = 2; + param->stride_w_ = 2; + param->nweights_ = 8 * 3 * 3 * 4; // 示例值 + hp_conv2dbackpropinputfusion_p(dy, w, dx, param); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/crop.rst.txt b/master/html/_sources/functionlib/dsplib/crop.rst.txt new file mode 100644 index 0000000..1a8fdf9 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/crop.rst.txt @@ -0,0 +1,64 @@ +Crop +================= + +类似于slice(张量切片)。不过只支持四维。 + +输入: + - **input** - 输入数据地址。 + - **in_shape** - 输入张量形状。 + - **out_shape** - 输出张量形状。 + - **type_size** - 输入和输出张量数据类型的长度。 + - **offset** - 每一维度裁剪开始的偏移量 + - **axis** - 裁剪开始的维度 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享/私有存储版本:** + +.. c:function:: void anytype_crop_anycore(void *input, void *output, int *in_shape, int *out_shape, int type_size, int* offset, int axis, int core_mask) + +各种数据类型、私有及共享空间版本均使用该函数。对于不同数据类型,改变type_size参数即可。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + void TestCropSMCFp32(int* in_shape, int* out_shape, int axis, int* offset_, int core_mask) { + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int core_num = GetCoreNum(core_mask); + float* input = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可 + float* output = (float*)0x98000000; + int* input_shape = (int*)0xA8000000; + int* output_shape = (int*)0xA8200000; + int type_size = sizeof(float); + int* offset = (int*)0xA8410000; + if (logic_core_id == 0) { + memcpy(offset, offset_, sizeof(int) * (4 - axis)); + memcpy(input_shape, in_shape, sizeof(int) * 4); + memcpy(output_shape, out_shape, sizeof(int) * 4); + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + anytype_crop_anycore(input, output, in_shape, out_shape, type_size, offset, axis, core_mask); + } + + void main(){ + int in_shape[4] = {2, 3, 3, 5}; + int out_shape[4] = {2, 2, 2, 5}; + int axis = 1; + int offset[3] = {1, 1, 0}; + int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可 + TestCropSMCFp32(in_shape, out_shape, axis, offset, core_mask); + } diff --git a/master/html/_sources/functionlib/dsplib/crop_and_resize.rst.txt b/master/html/_sources/functionlib/dsplib/crop_and_resize.rst.txt new file mode 100644 index 0000000..f8daebd --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/crop_and_resize.rst.txt @@ -0,0 +1,94 @@ +CropAndResize +================= + +从输入图像Tensor中提取切片并调整其大小。仅支持双线性插值方法。 + +输入: + - **src** - 输入数据的地址。 + - **box_idx** - boxes的索引,box_idx[i]的值表示第i个框的图像的值。 + - **boxes** - 第i行表示box_index[i]图像区域的坐标,并且坐标[y1,x1,y2,x2]是归一化后的值。归一化后的坐标值y,映射到图像y*(image_height-1)处,因此归一化后的图像高度范围为[0,1],映射到实际图像高度范围为[0,image_height-1]。我们允许y1>y2,在这种情况下,视为原始图像的上下翻转变换。宽度尺寸的处理类似。坐标取值允许在[0,1]范围之外,在这种情况下,我们使用extrapolation_value外插值进行补齐。 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **extrapolation_value** - 外插值。 + - **core_mask** - 核掩码。 + +**CropAndResizeParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct CropAndResizeParameter { + int* input_shape_; // 输入张量形状 + int* output_shape_; // 输出张量形状 + int* x_lefts_; // 用于存储预处理结果 + int* x_rights_; // 用于存储预处理结果 + int* y_tops_; // 用于存储预处理结果 + int* y_bottoms_; // 用于存储预处理结果 + void* x_weights_; // 用于存储预处理结果 + void* y_weights_; // 用于存储预处理结果 + void* line_buffers_; // 用于存储中间结果 + } CropAndResizeParameter; + +输出: + - **output** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享/私有存储版本:** + +.. c:function:: void i8_crop_and_resize_anycore(int8_t* src, int8_t* dst, int *box_idx, float *boxes, CropAndResizeParameter* param, float extrapolation_value, int core_mask) +.. c:function:: void hp_crop_and_resize_anycore(half* src, half* dst, int *box_idx, float *boxes, CropAndResizeParameter* param, float extrapolation_value, int core_mask) +.. c:function:: void fp_crop_and_resize_anycore(float* src, float* dst, int *box_idx, float *boxes, CropAndResizeParameter* param, half extrapolation_value, int core_mask) + +私有及共享空间版本均使用这些函数。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 28 + + void TestCropAndResizeSMCFp32(int* input_shape, int* output_shape, float* inp_boxes, int32_t* inp_box_idx, float extrapolation_value, int core_mask) { + int core_id = get_core_id(); + int core_num = GetCoreNum(core_mask); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + float* input = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可 + float* output = (float*)0x89000000; + float* boxes = (float*)0x8A000000; + int* box_idx = (int*)0x8B000000; + CropAndResizeParameter* param = (CropAndResizeParameter*)0x8C000000; + if (logic_core_id == 0) { + memcpy(boxes, inp_boxes, sizeof(float) * output_shape[0] * 4); + memcpy(box_idx, inp_box_idx, sizeof(int) * output_shape[0]); + param->input_shape_ = (int*)0x8D000000; + memcpy(param->input_shape_, input_shape, sizeof(int) * 4); + param->output_shape_ = (int*)0x8E000000; + memcpy(param->output_shape_, output_shape, sizeof(int) * 4); + param->line_buffers_ = (void*)0x8F000000; + param->x_lefts_ = (int*)0x90000000; + param->x_rights_ = (int*)0x91000000; + param->y_bottoms_ = (int*)0x92000000; + param->y_tops_ = (int*)0x93000000; + param->x_weights_ = (void*)0x94000000; + param->y_weights_ = (void*)0x95000000; + PrepareCropAndResizeBilinear(param->input_shape_, boxes, param->output_shape_, param->y_bottoms_, param->y_tops_, + param->x_lefts_, param->x_rights_, param->y_weights_, param->x_weights_); // 做预处理 + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_crop_and_resize_anycore(input, output, box_idx, boxes, param, extrapolation_value, core_mask); + } + + void main(){ + int input_shape[4] = {1, 4, 4, 4}; + int output_shape[4] = {1, 8, 8, 4}; + float boxes[4] = {0, 0, 0.5, 0.5}; + int box_idx[1] = {0}; + int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可 + float extrapolation_value = 0.5; + TestCropAndResizeSMCFp32(input_shape, output_shape, boxes, box_idx, extrapolation_value, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/depthtospace.rst.txt b/master/html/_sources/functionlib/dsplib/depthtospace.rst.txt new file mode 100644 index 0000000..4ff302e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/depthtospace.rst.txt @@ -0,0 +1,84 @@ +DepthToSpace +================= +将输入张量的深度通道按 block_size 分解并重排到空间维度(Depth -> Space)。 + + 输入: + - **input** - 输入数据地址。 + - **in_shape** - 输入形状,格式为 ``[batch, height, width, channel]``。 + - **block_size** - block 因子(单个整数)。 + - **data_size** - 单个元素字节数(例如 sizeof(float))。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 输出数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。 + - MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。 + + +**共享存储版本:** + +.. c:function:: void i8_depthtospace_s(int8_t *input, int8_t *output, const int *in_shape, int block_size, int data_size, int core_mask) +.. c:function:: void i16_depthtospace_s(int16_t *input, int16_t *output, const int *in_shape, int block_size, int data_size, int core_mask) +.. c:function:: void i32_depthtospace_s(int32_t *input, int32_t *output, const int *in_shape, int block_size, int data_size, int core_mask) +.. c:function:: void hp_depthtospace_s(half *input, half *output, const int *in_shape, int block_size, int data_size, int core_mask) +.. c:function:: void fp_depthtospace_s(float *input, float *output, const int *in_shape, int block_size, int data_size, int core_mask) +.. c:function:: void dp_depthtospace_s(double *input, double *output, const int *in_shape, int block_size, int data_size, int core_mask) +.. c:function:: void c64_depthtospace_s(float *input, float *output, const int *in_shape, int block_size, int data_size, int core_mask) +.. c:function:: void c128_depthtospace_s(double *input, double *output, const int *in_shape, int block_size, int data_size, int core_mask) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 多核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0xA0000000; // 多核版本:输入放在 DDR 地址 0xA0000000 + float *output = (float *)0xB0000000; // 多核版本:输出放在 DDR 地址 0xB0000000 + int in_shape[4] = {10, 16, 16, 4}; + int block_size = 2; + int core_mask = 0xff; + fp_depthtospace_s(input, output, in_shape, block_size, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_depthtospace_p(int8_t *input, int8_t *output, const int *in_shape, int block_size, int data_size) +.. c:function:: void i16_depthtospace_p(int16_t *input, int16_t *output, const int *in_shape, int block_size, int data_size) +.. c:function:: void i32_depthtospace_p(int32_t *input, int32_t *output, const int *in_shape, int block_size, int data_size) +.. c:function:: void hp_depthtospace_p(half *input, half *output, const int *in_shape, int block_size, int data_size) +.. c:function:: void fp_depthtospace_p(float *input, float *output, const int *in_shape, int block_size, int data_size) +.. c:function:: void dp_depthtospace_p(double *input, double *output, const int *in_shape, int block_size, int data_size) +.. c:function:: void c64_depthtospace_p(float *input, float *output, const int *in_shape, int block_size, int data_size) +.. c:function:: void c128_depthtospace_p(double *input, double *output, const int *in_shape, int block_size, int data_size) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + // FT78NE 单核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0x10000000; // 单核版本:输入放在 L2 地址 0x10000000 + float *output = (float *)0x10040000; // 单核版本:输出放在 L2 地址 0x10040000 + int in_shape[4] = {10, 16, 16, 4}; + int block_size = 2; + fp_depthtospace_p(input, output, in_shape, block_size, sizeof(float)); + return 0; + } + + diff --git a/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt b/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt index 6817511..b63f4c0 100644 --- a/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt +++ b/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt @@ -3,5 +3,50 @@ DSP Library C API Reference .. toctree:: :maxdepth: 1 - - equal \ No newline at end of file + + equal + activation + expfusion + reverse_sequence + reversev2 + fillv2 + squeeze + unsqueeze + expand_dims + leaky_relu + lstm + scatter_elements + reduce + resize + crop_and_resize + crop + conv2d + adder + conv2d_transpose + gru + assert + range + raggedrange + linspace + matmulfusion + attention + floor + floordiv + embeddinglookup + eltwise + adamweightdecay + applymomentum + avgpoolinggrad + batchtospace + batchtospacend + broadcastto + depthtospace + spacetodepth + spacetobatch + spacetobatchnd + fusedbatchnorm + groupnormfusion + conv2dbackpropinputfusion + conv2dbackpropfilterfusion + sgd + scalefusion diff --git a/master/html/_sources/functionlib/dsplib/eltwise.rst.txt b/master/html/_sources/functionlib/dsplib/eltwise.rst.txt new file mode 100644 index 0000000..dbae68d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/eltwise.rst.txt @@ -0,0 +1,99 @@ +Eltwise +================= + +输入两个等长数组以及控制参数,根据控制参数的值决定对两个数组做对位相加、对位相乘或取最大值操作。 + +.. math:: + + \mathbf{output_i} = + \begin{cases} + \mathbf{Input0_i} \cdot \mathbf{Input1_i}, & \text{if } \text{eltwise_mode} = \text{Eltwise_PROD} \\[6pt] + \mathbf{Input0_i} + \mathbf{Input1_i}, & \text{if } \text{eltwise_mode} = \text{Eltwise_SUM} \\[6pt] + \max(\mathbf{Input0_i}, \mathbf{Input1_i}), & \text{if } \text{eltwise_mode} = \text{Eltwise_MAXIMUM} + \end{cases} + +输入: + - **Input0** - 第一个输入数据地址。 + - **Input1** - 第二个输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64(除最大值), cplx128(除最大值) + - MT7004 支持fp16, fp32, int16, int32, cplx64(除最大值) + +**共享存储版本:** + +.. c:function:: void i8_eltwise_s(int8_t* Input0, int8_t* Input1, int8_t* output,int length, int eltwise_mode_, int core_mask) +.. c:function:: void i16_eltwise_s(int16_t* Input0, int16_t* Input1, int16_t* output,int length, int eltwise_mode_, int core_mask) +.. c:function:: void i32_eltwise_s(int* Input0, int* Input1, int* output,int length, int eltwise_mode_, int core_mask) +.. c:function:: void hp_eltwise_s(half* Input0, half* Input1, half* output,int length, int eltwise_mode_, int core_mask) +.. c:function:: void fp_eltwise_s(float* Input0, float* Input1, float* output,int length, int eltwise_mode_, int core_mask) +.. c:function:: void dp_eltwise_s(double* Input0, double* Input1, double* output,int length, int eltwise_mode_, int core_mask) +.. c:function:: void c64_eltwise_s(float* Input0, float* Input1, float* output,int length, int eltwise_mode_, int core_mask) +.. c:function:: void c128_eltwise_s(double* Input0, double* Input1, double* output, int length, int eltwise_mode_, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + #define Eltwise_PROD 0 + #define Eltwise_SUM 1 + #define Eltwise_MAXIMUM 2 + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *input1 = (float *)0xB0000000; + float *output = (float *)0xC0000000; + int length = 1000; + int eltwise_mode_ = Eltwise_SUM; + int core_mask = 0xff; + fp_eltwise_s(input0, input1, output, length, eltwise_mode_, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_eltwise_p(int8_t *Input0, int8_t *Input1, int8_t *output, int eltwise_mode_, int length) +.. c:function:: void i16_eltwise_p(int16_t *Input0, int16_t *Input1, int16_t *output, int eltwise_mode_, int length) +.. c:function:: void i32_eltwise_p(int32_t *Input0, int32_t *Input1, int32_t *output, int eltwise_mode_, int length) +.. c:function:: void hp_eltwise_p(half* Input0, half* Input1, bool* output, int eltwise_mode_,int length) +.. c:function:: void fp_eltwise_p(float* Input0, float* Input1, float* output, int eltwise_mode_,int length) +.. c:function:: void dp_eltwise_p(double* Input0, double* Input1, double* output, int eltwise_mode_,int length) +.. c:function:: void c64_eltwise_p(float *Input0, float *Input1, float *output, int eltwise_mode_, int length) +.. c:function:: void c128_eltwise_p(double *Input0, double *Input1, double *output, int eltwise_mode_, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + #define Eltwise_PROD 0 + #define Eltwise_SUM 1 + #define Eltwise_MAXIMUM 2 + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; //input在L2空间 + float *input1 = (float *)0x10820000; + float *output = (float *)0x10830000; + int length = 1000; + int eltwise_mode_ = Eltwise_SUM; + fp_eltwise_p(input0, input1, output, eltwise_mode_, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/embeddinglookup.rst.txt b/master/html/_sources/functionlib/dsplib/embeddinglookup.rst.txt new file mode 100644 index 0000000..858cc87 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/embeddinglookup.rst.txt @@ -0,0 +1,101 @@ +EmbeddingLookup +================= + +传入一个矩阵和一组索引,根据给定的索引提取对应的行(向量),若此行不曾被标记为“已正则化”,则对此行进行正则化处理,将结果拼接输出,若已经正则化过则直接输出。 + +.. math:: + + \forall k \in [1, ids\_size], \quad + \begin{cases} + \text{if } \textbf{is_regulated}[i_k] = 0, & + \begin{cases} + \displaystyle X_{i_k} \leftarrow + X_{i_k} \cdot \frac{\text{max_norm}} + {\sum_{j=1}^{layer\_size\_} X_{i_k, j}} \\[10pt] + \textbf{is_regulated}[i_k] \leftarrow 1 + \end{cases} \\[12pt] + \text{输出向量 } Y_k \leftarrow X_{i_k} + \end{cases} + +输入: + - **input_data** - 输入矩阵数据地址。 + - **ids** - 输入索引的存储地址。 + - **max_norm** - 最大范数约束。 + - **is_regulated** - 记录矩阵行是否被正则化的标志数组。 + - **ids_size_** - 输入索引个数。 + - **layer_size_** - 输入矩阵的列数。 + - **layer_num_** - 输入矩阵的行数。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 结果输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void hp_embeddinglookup_s(half* input_data,int* ids, half* output, half max_norm_, bool* is_regulated, int ids_size_, int layer_size_, int layer_num_ , int core_mask) +.. c:function:: void fp_embeddinglookup_s(float* input_data,int* ids, float* output, float max_norm_, bool* is_regulated, int ids_size_, int layer_size_, int layer_num_ , int core_mask) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input_data = (float *)0xA0000000; //input在DDR空间 + float *output_data = (float *)0xA0872c00; + int layer_size = 4;//列 + int layer_num = 5;//行 + float max_norm = 4.5; + int ids[]={0,1,3}; + int ids_size = 3;//提取三行 + bool *output = (bool *)0xC0000000; + bool is_regulated_[5] = {0};//layer_num + int core_mask = 0xff; + fp_embeddinglookup_s(input_data, ids, output_data, max_norm, is_regulated_, ids_size, layer_size, layer_num, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_embeddinglookup_p(half* Input0, half* Input1, bool* output,int length) +.. c:function:: void fp_embeddinglookup_p(float* Input0, float* Input1, bool* output,int length) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input_data = (float *)0x10810000; //input在DDR空间 + float *output_data = (float *)0x10820000; + int layer_size = 4;//列 + int layer_num = 5;//行 + float max_norm = 4.5; + int ids[]={0,1,3}; + int ids_size = 3;//提取三行 + bool *output = (bool *)0xC0000000; + bool is_regulated_[5] = {0};//layer_num + fp_embeddinglookup_s(input_data, ids, output_data, max_norm, is_regulated_, ids_size, layer_size, layer_num); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/equal.rst.txt b/master/html/_sources/functionlib/dsplib/equal.rst.txt index bd9ff0c..8045d6b 100644 --- a/master/html/_sources/functionlib/dsplib/equal.rst.txt +++ b/master/html/_sources/functionlib/dsplib/equal.rst.txt @@ -42,7 +42,7 @@ Equal .. code-block:: c :linenos: - :emphasize-lines: 10 + :emphasize-lines: 11 //FT78NE示例 #include @@ -74,16 +74,16 @@ Equal .. code-block:: c :linenos: - :emphasize-lines: 9 + :emphasize-lines: 10 //FT78NE示例 #include #include int main(int argc, char* argv[]) { - float *input0 = (float *)0x10000000; //input在L2空间 - float *input1 = (float *)0x10001000; - bool *output = (bool *)0xC0000000; + float *input0 = (float *)0x10810000; //input在L2空间 + float *input1 = (float *)0x10820000; + bool *output = (bool *)0x10830000; int length = 1000; fp_equal_p(input0, input1, output, length); return 0; diff --git a/master/html/_sources/functionlib/dsplib/expand_dims.rst.txt b/master/html/_sources/functionlib/dsplib/expand_dims.rst.txt new file mode 100644 index 0000000..14c4aa0 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/expand_dims.rst.txt @@ -0,0 +1,42 @@ +ExpandDims +================= + +对输入张量在给定的轴上添加额外维度。由于该算子仅改变张量形状,因此其DSP算子的作用是将数据从输入张量完整拷贝到输出张量。 + +输入: + - **src** - 输入地址 + - **total_copy_size** - 计算得到的总共需拷贝的数据量,单位为字节。 + - **core_mask** - 核掩码。 + +输出: + - **dst** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享/私有存储版本:** + +.. c:function:: void anytype_expand_dims_anycore(void* src, void* dst, int total_copy_size, int core_mask) + +各种数据类型、私有及共享空间版本均使用该函数。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 8 + + void main(){ + int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可 + int core_num = GetCoreNum(core_mask); + float* src = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可 + float* dst = (float*)0x98000000; + int shape[3] = {1, 10, 10}; + int total_copy_size = shape[0] * shape[1] * shape[2] * sizeof(float); + anytype_expand_dims_anycore(src, dst, total_copy_size, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/expfusion.rst.txt b/master/html/_sources/functionlib/dsplib/expfusion.rst.txt new file mode 100644 index 0000000..9c67ae1 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/expfusion.rst.txt @@ -0,0 +1,101 @@ +ExpFusion +================= + + + + 传入一个数组,逐元素计算其乘上输入因子(可选择)后的指数值,再将指数值乘上输出因子后输出。 + + .. math:: + + dst_i = \exp(src_i \cdot s_{in}) \cdot s_{out} + \quad \text{where} \quad + s_{in} = + \begin{cases} + 1, & scale = 1 \\ + in\_scale, & scale \neq 1 + \end{cases} + + \quad s_{out} = out\_scale + + 输入: + - **src_data** - 输入数据地址。 + - **length** - 计算长度。 + - **in_scale** - 输入缩放因子,当scale != 1时启用。 + - **out_scale** - 输出缩放因子。 + - **scale** - 输入缩放因子启用控制。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **dst_data** - 计算结果地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_expfusion_s(int8_t* src_data, int8_t* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) +.. c:function:: void i16_expfusion_s(int16_t* src_data, half* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) +.. c:function:: void i32_expfusion_s(int* src_data, float* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) +.. c:function:: void hp_expfusion_s(half* src_data, half* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) +.. c:function:: void fp_expfusion_s(float* src_data, float* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) +.. c:function:: void dp_expfusion_s(double* src_data, double* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) +.. c:function:: void c64_expfusion_s(float* src_data, float* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) +.. c:function:: void c128_expfusion_s(double* src_data, double* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1000; + float in_scale = 0.5, out_scale = 1.2; + int scale = 1; + int core_mask = 0xff; + fp_expfusion_s( input0, output, length, in_scale, out_scale, scale,core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_expfusion_p(int8_t* src_data, int8_t* dst_data, int length, float in_scale, float out_scale, int scale) +.. c:function:: void i16_expfusion_p(int16_t* src_data, half* dst_data, int length, float in_scale, float out_scale, int scale) +.. c:function:: void i32_expfusion_p(int* src_data, float* dst_data, int length, float in_scale, float out_scale, int scale) +.. c:function:: void hp_expfusion_p(half* src_data, half* dst_data, int length, float in_scale, float out_scale, int scale) +.. c:function:: void fp_expfusion_p(float* src_data, float* dst_data, int length, float in_scale, float out_scale, int scale) +.. c:function:: void dp_expfusion_p(double* src_data, double* dst_data, int length, float in_scale, float out_scale, int scale) +.. c:function:: void c64_expfusion_p(float* src_data, float* dst_data, int length, float in_scale, float out_scale, int scale) +.. c:function:: void c128_expfusion_p(double* src_data, double* dst_data, int length, float in_scale, float out_scale, int scale) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; //input在L2空间 + float *output = (float *)0x10820000; + int length = 1000; + float in_scale = 0.5, out_scale = 1.2; + int scale = 1; + fp_expfusion_p( input0, output, length, in_scale, out_scale, scale); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/fillv2.rst.txt b/master/html/_sources/functionlib/dsplib/fillv2.rst.txt new file mode 100644 index 0000000..0cac948 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fillv2.rst.txt @@ -0,0 +1,71 @@ +FillV2 +================= + +创建一个Tensor,根据指定的shape,将其值由value进行填充。 + +输入: + - **value** - 填充值的地址 + - **length** - 由指定shape计算出的张量总长度。 + - **type_size** - 填充值的数据类型的长度 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void anytype_fillv2_s(void* value, void* output, int length, int type_size, int core_mask) + +对于不同数据类型,改变type_size参数即可。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + void main() { + float* input = (float*)0xA0000000; + float value = 789.1; + float* output = (float*)0x88000000; + int i; + int length = 1000; + int core_mask = 0b1111; + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int core_num = GetCoreNum(core_mask); + if (logic_core_id == 0) { + *input = value; + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + anytype_fillv2_s(input, output, length, 4, core_mask); + } + +**私有存储版本:** + +.. c:function:: void anytype_fillv2_p(void* value, void* output, int length, int type_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + void main() { + float* input = (float*)0x10000000; + float value = 789.1; + float* output = (float*)0x10010000; + int i; + int length = 1000; + int core_mask = 0b0001; // 要启动哪一个核,就将哪一位设置为1,只允许存在一个核心启动 + *input = value; + anytype_fillv2_p(input, output, length, 4, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/floor.rst.txt b/master/html/_sources/functionlib/dsplib/floor.rst.txt new file mode 100644 index 0000000..aa5236e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/floor.rst.txt @@ -0,0 +1,78 @@ +Floor +================= + + + + 传入一个浮点型数组,对于数组中每一个元素执行向下取整操作。 + + .. math:: + + dst_i = floorf(src_i ) + + 输入: + - **input** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 计算结果地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持fp32, fp64 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + + +.. c:function:: void hp_floor_s(half* src_data, half* dst_data, int length, int core_mask) +.. c:function:: void fp_floor_s(float* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void dp_floor_s(double* src_data, double* dst_data, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + fp_floor_s( input0, output, length,core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_floor_p(half* src_data, half* dst_data, int length) +.. c:function:: void fp_floor_p(float* src_data, float* dst_data, int length) +.. c:function:: void dp_floor_p(double* src_data, double* dst_data, int length) + + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; //input在L2空间 + float *output = (float *)0xC0000000; + int length = 1000; + fp_floor_p( input0, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/floordiv.rst.txt b/master/html/_sources/functionlib/dsplib/floordiv.rst.txt new file mode 100644 index 0000000..1fdd465 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/floordiv.rst.txt @@ -0,0 +1,81 @@ +FloorDiv +================= + + + + 传入两个等长的浮点型数组,对于数组中对位元素做除法,再对除法所得结果执行向下取整操作。 + + .. math:: + + dst_i = floorf(\frac{input0_i}{input1_i}) + + 输入: + - **input0** - 被除数数据地址。 + - **input1** - 除数数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 计算结果地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持fp32, fp64 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + + +.. c:function:: void hp_floordiv_s(half* src_data0, half* src_data1, half* dst_data, int length, int core_mask) +.. c:function:: void fp_floordiv_s(float* src_data0, float* src_data1, float* dst_data, int length, int core_mask) +.. c:function:: void dp_floordiv_s(double* src_data0, double* src_data1, double* dst_data, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *input1 = (float *)0xB0000000; + float *output = (float *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + fp_floordiv_s( input0,input1, output, length,core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_floordiv_p(half* src_data0, half* src_data1, half* dst_data, int length) +.. c:function:: void fp_floordiv_p(float* src_data0, float* src_data1, float* dst_data, int length) +.. c:function:: void dp_floordiv_p(double* src_data0, double* src_data1, double* dst_data, int length) + + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; //input在L2空间 + float *input1 = (float *)0x10820000; + float *output = (float *)0x10830000; + int length = 100; + fp_floordiv_p( input0, input1, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/fusedbatchnorm.rst.txt b/master/html/_sources/functionlib/dsplib/fusedbatchnorm.rst.txt new file mode 100644 index 0000000..6095913 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fusedbatchnorm.rst.txt @@ -0,0 +1,91 @@ +FusedBatchNorm +================= +对输入张量执行融合批归一化(Fused Batch Normalization),在多核间拆分批量单元并行完成归一化与仿射变换。 + + .. math:: + + \hat{x}_{b,c} = \frac{x_{b,c} - mean_c}{\sqrt{variance_c + \epsilon}}, \quad y_{b,c} = scale_c \cdot \hat{x}_{b,c} + offset_c + + + 输入: + - **input** - 输入张量首地址,形状为 ``[unit, channel]``。 + - **scale** - 缩放系数数组首地址,长度为 ``channel``。 + - **offset** - 平移系数数组首地址,长度为 ``channel``。 + - **mean** - 归一化均值数组首地址,长度为 ``channel``。 + - **variance** - 归一化方差数组首地址,长度为 ``channel``。 + - **epsilon** - 数值稳定项。 + - **channel** - 通道数。 + - **unit** - 归一化单元数量(批量大小 × 高 × 宽)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 写回融合批归一化计算结果的张量首地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + +**共享存储版本:** + +.. c:function:: void hp_fusedbatchnorm_s(const half *input, const half *scale, const half *offset, const half *mean, const half *variance, float epsilon, int channel, int unit, int core_mask, half *output) +.. c:function:: void fp_fusedbatchnorm_s(const float *input, const float *scale, const float *offset, const float *mean, const float *variance, float epsilon, int channel, int unit, int core_mask, float *output) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 多核示例 + #include + + int main(void) { + const float *input = (const float *)0xA0000000; // DDR 存储 + const float *scale = (const float *)0xB0000000; + const float *offset = (const float *)0xB0001000; + const float *mean = (const float *)0xB0002000; + const float *variance = (const float *)0xB0003000; + float *output = (float *)0xC0000000; + int channel = 64; + int unit = 1024; + float epsilon = 1e-5f; + int core_mask = 0xff; + fp_fusedbatchnorm_s(input, scale, offset, mean, variance, + epsilon, channel, unit, core_mask, + output); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_fusedbatchnorm_p(const half *input, const half *scale, const half *offset, const half *mean, const half *variance, float epsilon, int channel, int unit, half *output) +.. c:function:: void fp_fusedbatchnorm_p(const float *input, const float *scale, const float *offset, const float *mean, const float *variance, float epsilon, int channel, int unit, float *output) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + // MT7004 单核示例 + #include + + int main(void) { + const half *input = (const half *)0x10000000; // L2 存储 + const half *scale = (const half *)0x10004000; + const half *offset = (const half *)0x10008000; + const half *mean = (const half *)0x1000C000; + const half *variance = (const half *)0x10010000; + half *output = (half *)0x10014000; + int channel = 32; + int unit = 512; + float epsilon = 1e-4f; + hp_fusedbatchnorm_p(input, scale, offset, mean, variance, + epsilon, channel, unit, output); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/groupnormfusion.rst.txt b/master/html/_sources/functionlib/dsplib/groupnormfusion.rst.txt new file mode 100644 index 0000000..0947ddc --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/groupnormfusion.rst.txt @@ -0,0 +1,100 @@ +GroupNormFusion +================= +对输入张量执行分组归一化融合操作(Group Normalization Fusion),在多核环境中按批次拆分并行完成均值、方差与归一化计算。 + + .. math:: + + \hat{x}_{(b,u,c)} = \frac{x_{(b,u,c)} - \mu_{(b,g)}}{\sqrt{\sigma^2_{(b,g)} + \epsilon}}, \quad y_{(b,u,c)} = \hat{x}_{(b,u,c)} \cdot scale_c + offset_c + + 其中 :math:`(b,u,c)` 表示批次、空间位置与通道索引,:math:`g` 为通道所属分组。 + + 输入: + - **input** - 输入张量首地址,形状 ``[batch, unit, channel]``。 + - **scale** - 通道缩放系数首地址,长度为 ``channel``。 + - **offset** - 通道偏移系数首地址,长度为 ``channel``。 + - **mean** - 批次 × 分组的均值缓冲区首地址,长度为 ``batch * num_groups``。 + - **variance** - 批次 × 分组的方差缓冲区首地址,长度为 ``batch * num_groups``。 + - **epsilon** - 数值稳定项。 + - **num_groups** - 分组数。 + - **channel** - 通道总数。 + - **unit** - 每批次内的归一化单元数(H×W)。 + - **batch** - 批次数。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 写回分组归一化结果的张量首地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + + +**共享存储版本:** + +.. c:function:: void hp_groupnormfusion_s(const half *input, const half *scale, const half *offset, half *mean, half *variance, float epsilon, int num_groups, int channel, int unit, int batch, int core_mask, half *output) +.. c:function:: void fp_groupnormfusion_s(const float *input, const float *scale, const float *offset, float *mean, float *variance, float epsilon, int num_groups, int channel, int unit, int batch, int core_mask, float *output) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 17 + + // FT78NE 多核示例 + #include + + int main(void) { + const float *input = (const float *)0xA0000000; // DDR 存储 + const float *scale = (const float *)0xB0000000; + const float *offset = (const float *)0xB0001000; + float *mean = (float *)0xB0002000; + float *variance = (float *)0xB0003000; + float *output = (float *)0xC0000000; + int num_groups = 8; + int channel = 64; + int unit = 49; + int batch = 32; + float epsilon = 1e-5f; + int core_mask = 0xff; + fp_groupnormfusion_s(input, scale, offset, mean, variance, + epsilon, num_groups, channel, unit, + batch, core_mask, output); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_groupnormfusion_p(const half *input, const half *scale, const half *offset, half *mean, half *variance, float epsilon, int num_groups, int channel, int unit, int batch, half *output) +.. c:function:: void fp_groupnormfusion_p(const float *input, const float *scale, const float *offset, float *mean, float *variance, float epsilon, int num_groups, int channel, int unit, int batch, float *output) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 16 + + // MT7004 单核示例 + #include + + int main(void) { + const half *input = (const half *)0x10000000; // L2 存储 + const half *scale = (const half *)0x10004000; + const half *offset = (const half *)0x10008000; + half *mean = (half *)0x1000C000; + half *variance = (half *)0x10010000; + half *output = (half *)0x10014000; + int num_groups = 4; + int channel = 32; + int unit = 36; + int batch = 16; + float epsilon = 1e-4f; + hp_groupnormfusion_p(input, scale, offset, mean, variance, + epsilon, num_groups, channel, unit, + batch, output); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/gru.rst.txt b/master/html/_sources/functionlib/dsplib/gru.rst.txt new file mode 100644 index 0000000..4441a91 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/gru.rst.txt @@ -0,0 +1,201 @@ +GRU +================= + +将多层门控循环单元 (GRU) RNN 应用于输入序列。 + +GRU 网络模型中有两个门:更新门和重置门。将两个连续的时间节点表示为 :math:`t - 1` 和 :math:`t`。给定一个在时刻 :math:`t` 的输入 :math:`x_t`,一个隐藏状态 :math:`h_{t-1}`,在时刻 :math:`t` 的更新门和重置门使用门控制机制计算。更新门 :math:`z_t` 用于控制前一时刻的状态信息被带入到当前状态中的程度,重置门 :math:`r_t` 控制前一状态有多少信息被写入到当前候选集 :math:`n_t` 上 + +对于输入序列中的每个元素,每一层计算以下函数: + +.. math:: + :nowrap: + + \begin{align*} + r_t &= \sigma(W_{ir}x_t + b_{ir} + W_{hr}h_{(t-1)} + b_{hr}) \\ + z_t &= \sigma(W_{iz}x_t + b_{iz} + W_{hz}h_{(t-1)} + b_{hz}) \\ + n_t &= \tanh(W_{in}x_t + b_{in} + r_t \odot (W_{hn}h_{(t-1)} + b_{hn})) \\ + h_t &= (1-z_t) \odot n_t + z_t \odot h_{(t-1)} + \end{align*} + +其中 :math:`\sigma` 是 sigmoid 激活函数,:math:`\odot` 是 Hadamard 积(逐元素乘积)。:math:`W, b` 是公式中输出和输入之间的可学习权重。例如,:math:`W_{ir}, b_{ir}` 是用于将输入 :math:`x_t` 转换为 :math:`r_t` 的权重和偏置。 + +注意,本算子中候选门 :math:`n_t` 的计算与原始论文和Mindspore框架略有不同。在原始实现中,:math:`r_t` 和上一隐藏状态 :math:`h_{(t-1)}` 之间的 Hadamard 积 (:math:`\odot`) 在与权重矩阵 :math:`W` 相乘和加上偏置之前进行: + +.. math:: + + n_t = \tanh(W_{in}x_t + b_{in} + W_{hn}(r_t \odot h_{(t-1)}) + b_{hn}) + +本算子采用 PyTorch 实现方式,是在 :math:`W_{hn}h_{(t-1)}` 之后完成的: + +.. math:: + + n_t = \tanh(W_{in}x_t + b_{in} + r_t \odot (W_{hn}h_{(t-1)} + b_{hn})) + +输入: + - **input** - 输入数据的地址。 + - **weight_g** - 可学习的输入-隐藏权重的地址。 + - **weight_r** - 可学习的隐藏-隐藏权重的地址。 + - **input_bias** - 可学习的输入-隐藏偏置的地址。 + - **state_bias** - 可学习的隐藏-隐藏偏置的地址。 + - **hidden_state** - 初始隐藏状态的地址。 + - **buffer** - 用于存储中间计算结果。 + - **gru_param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**GruParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct GruParameter { + int input_size_; // 输入input中预期特征的数量 + int hidden_size_; // 隐藏状态h中的特征数量 + int seq_len_; // 输入batch中每个序列的长度 + int batch_; // 总批次数 + int output_step_; // 每次循环中output步长 + int bidirectional_; // 是否为双向GRU + int input_row_align_; // 输入行对齐值 + int input_col_align_; // 输入列对齐值 + int state_row_align_; // 隐藏状态行对齐值 + int state_col_align_; // 隐藏状态列对齐值 + int check_seq_len_; // 进行计算的序列长度 + } GruParameter; + +输出: + - **output** - 输出地址。 + - **hidden_state** - 最终的隐藏状态。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_Gru_s(int8_t *output, int8_t *input, int8_t *weight_g, int8_t *weight_r, int8_t *input_bias, int8_t *state_bias, int8_t *hidden_state, int8_t *buffer[4], GruParameter *gru_param, int core_mask) +.. c:function:: void hp_Gru_s(half *output, half *input, half *weight_g, half *weight_r, half *input_bias, half *state_bias, half *hidden_state, half *buffer[4], GruParameter *gru_param, int core_mask); +.. c:function:: void fp_Gru_s(float *output, float *input, float *weight_g, float *weight_r, float *input_bias, float *state_bias, float *hidden_state, float *buffer[4], GruParameter *gru_param, int core_mask); + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 41 + + void TestGruSMCFp32(int check_seq_len, int seq_len, int batch_size, int input_size, int bidirectional, int hidden_size, int core_mask) { + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int core_num = GetCoreNum(core_mask); + float *output = (void*)0x88000000; + float *input = (void*)0x88100000; + float *weight_g = (void*)0x88200000; + float *weight_r = (void*)0x88300000; + float *input_bias = (void*)0x88400000; + float *state_bias = (void*)0x88500000; + float *hidden_state = (void*)0x88600000; + float** buffer = (float**)0x88700000; + float *output_hidden_state = (void*)0x88800000; + GruParameter* param = (GruParameter*)0x88900000; + int hidden_state_batch = 1; + int num_directions = 1; + if (bidirectional) { + hidden_state_batch = hidden_state_batch * 2; + num_directions = num_directions * 2; + } + int input_col_align = hidden_size; + int state_col_align = hidden_size; + if (logic_core_id == 0) { + memcpy(output_hidden_state, hidden_state, hidden_state_batch * batch_size * hidden_size * sizeof(float)); + memcpy(check_output_hidden_state, hidden_state, hidden_state_batch * batch_size * hidden_size * sizeof(float)); + buffer[0] = (void*)0x88A00000; + buffer[1] = (void*)0x88B00000; + buffer[2] = (void*)0x88C00000; + buffer[3] = (void*)0x88D00000; + param->batch_ = batch_size; + param->bidirectional_ = bidirectional; + param->hidden_size_ = hidden_size; + param->input_col_align_ = input_col_align; + param->input_size_ = input_size; + param->output_step_ = batch_size * hidden_size * num_directions; + param->seq_len_ = seq_len; + param->state_col_align_ = state_col_align; + param->check_seq_len_ = check_seq_len; + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_Gru_s(output, input, weight_g, weight_r, input_bias, state_bias, output_hidden_state, buffer, param, core_mask); + } + + void main() { + int check_seq_len = 2; + int seq_len = 2; + int batch_size = 2; + int input_size = 2; + int bidirectional = 0; + int hidden_size = 2; + int core_mask = 0b1111; + TestGruSMCFp32(check_seq_len, seq_len, batch_size, input_size, bidirectional, hidden_size, core_mask); + } + +**私有存储版本:** + +.. c:function:: void i8_Gru_p(int8_t *output, int8_t *input, int8_t *weight_g, int8_t *weight_r, int8_t *input_bias, int8_t *state_bias, int8_t *hidden_state, int8_t *buffer[4], GruParameter *gru_param, int core_mask) +.. c:function:: void hp_Gru_p(half *output, half *input, half *weight_g, half *weight_r, half *input_bias, half *state_bias, half *hidden_state, half *buffer[4], GruParameter *gru_param, int core_mask); +.. c:function:: void fp_Gru_p(float *output, float *input, float *weight_g, float *weight_r, float *input_bias, float *state_bias, float *hidden_state, float *buffer[4], GruParameter *gru_param, int core_mask); + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 35 + + void TestGruL2Fp32(int check_seq_len, int seq_len, int batch_size, int input_size, int bidirectional, int hidden_size, int core_mask) { + float *output = (void*)0x10000000; // 私有存储版本地址设置在AM内 + float *input = (void*)0x10004000; + float *weight_g = (void*)0x10008000; + float *weight_r = (void*)0x1000C000; + float *input_bias = (void*)0x10010000; + float *state_bias = (void*)0x10014000; + float *hidden_state = (void*)0x10018000; + float** buffer = (float**)0x1001C000; + float *output_hidden_state = (void*)0x10020000; + GruParameter* param = (GruParameter*)0x10024000; + int hidden_state_batch = 1; + int num_directions = 1; + if (bidirectional) { + hidden_state_batch = hidden_state_batch * 2; + num_directions = num_directions * 2; + } + int input_col_align = hidden_size; + int state_col_align = hidden_size; + memcpy(output_hidden_state, hidden_state, hidden_state_batch * batch_size * hidden_size * sizeof(float)); + memcpy(check_output_hidden_state, hidden_state, hidden_state_batch * batch_size * hidden_size * sizeof(float)); + buffer[0] = (void*)0x10030000; + buffer[1] = (void*)0x10034000; + buffer[2] = (void*)0x10038000; + buffer[3] = (void*)0x1003C000; + param->batch_ = batch_size; + param->bidirectional_ = bidirectional; + param->hidden_size_ = hidden_size; + param->input_col_align_ = input_col_align; + param->input_size_ = input_size; + param->output_step_ = batch_size * hidden_size * num_directions; + param->seq_len_ = seq_len; + param->state_col_align_ = state_col_align; + param->check_seq_len_ = check_seq_len; + fp_Gru_p(output, input, weight_g, weight_r, input_bias, state_bias, output_hidden_state, buffer, param, core_mask); + } + + void main() { + int check_seq_len = 2; + int seq_len = 2; + int batch_size = 2; + int input_size = 2; + int bidirectional = 0; + int hidden_size = 2; + int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动 + TestGruL2Fp32(check_seq_len, seq_len, batch_size, input_size, bidirectional, hidden_size, core_mask); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/leaky_relu.rst.txt b/master/html/_sources/functionlib/dsplib/leaky_relu.rst.txt new file mode 100644 index 0000000..769eca8 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/leaky_relu.rst.txt @@ -0,0 +1,75 @@ +LeakyReLu +================= + +Leaky ReLU激活函数。 + +该激活函数定义如下: + +.. math:: + leaky\_relu(x) = + \begin{cases} + x, & \text{if } x \geq 0; \\ + \alpha \cdot x, & \text{otherwise.} + \end{cases} + +其中, :math:`\alpha` 表示 alpha 参数。 + +输入: + - **input** - 输入数据的地址。 + - **elem_cnt** - 计算长度。 + - **alpha** - 公式中的alpha参数。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_leaky_relu_s(float* input, float* output, int elem_cnt, float alpha, int core_mask) +.. c:function:: void hp_leaky_relu_s(half* input, half* output, int elem_cnt, half alpha, int core_mask) +.. c:function:: void i8_leaky_relu_s(const int8_t* input, int8_t* output, int elem_cnt, float alpha, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 7 + + void main() { + float* input = (float*)0x82000000; + float* output = (float*)0x88000000; + int length = 1000; + float alpha = 0.9; + int core_mask = 0b1111; + fp_leaky_relu_s(input, output, length, alpha, core_mask); + } + +**私有存储版本:** + +.. c:function:: void fp_leaky_relu_p(float* input, float* output, int elem_cnt, float alpha, int core_mask) +.. c:function:: void hp_leaky_relu_p(half* input, half* output, int elem_cnt, half alpha, int core_mask) +.. c:function:: void i8_leaky_relu_p(const int8_t* input, int8_t* output, int elem_cnt, float alpha, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 7 + + void main() { + float* input = (float*)0x10000000; // 私有存储版本地址设置在AM内 + float* output = (float*)0x10010000; + int length = 1000; + float alpha = 0.9; + int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动 + fp_leaky_relu_p(input, output, length, alpha, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/linspace.rst.txt b/master/html/_sources/functionlib/dsplib/linspace.rst.txt new file mode 100644 index 0000000..ba87351 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/linspace.rst.txt @@ -0,0 +1,68 @@ +LinSpace +================= + +生成从 ``start`` 到 ``end`` 的等间距序列,线性插值 + +.. math:: + + \text{output}_i = \text{start} + i \cdot \frac{\text{end} - \text{start}}{\text{length} - 1},\quad i = 0, 1, \dots, \text{length} - 1 + +输入: + - **start** - 序列起始值。 + - **end** - 序列终止值(包含)。 + - **length** - 序列点数;要求 length ≥ 2。 + - **core_mask(可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 输出序列地址(float32)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 多核版本按行数均匀分割,各核处理自己的段落。 + - 步长计算为 (end - start) / (length - 1)。 + +**共享存储版本:** + +.. c:function:: void fp_linspace_s(float *output, float start, float end, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0xA0000000; // DDR + float start = 0.0f; + float end = 100.0f; + int length = 1001; + int core_mask = 0xff; + fp_linspace_s(output, start, end, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_linspace_p(float *output, float start, float step, int num) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0x10000000; // L2 + float start = 1.5f; + float step = 0.5f; + int num = 1000; + fp_linspace_p(output, start, step, num); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/lstm.rst.txt b/master/html/_sources/functionlib/dsplib/lstm.rst.txt new file mode 100644 index 0000000..f693278 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lstm.rst.txt @@ -0,0 +1,181 @@ +LSTM +================= +一种常用的 循环神经网络(RNN) 模块,用于处理具有时序依赖特征的数据(如语音、文本、时间序列等)。每个时间步的公式化描述如下。 + +.. math:: + + \begin{aligned} + i_t &= \sigma(W_{ii} x_t + W_{hi} h_{t-1} + b_i) && \text{(输入门)} \\[6pt] + f_t &= \sigma(W_{if} x_t + W_{hf} h_{t-1} + b_f) && \text{(遗忘门)} \\[6pt] + g_t &= \tanh(W_{ig} x_t + W_{hg} h_{t-1} + b_g) && \text{(候选状态)} \\[6pt] + o_t &= \sigma(W_{io} x_t + W_{ho} h_{t-1} + b_o) && \text{(输出门)} \\[6pt] + c_t &= f_t \odot c_{t-1} + i_t \odot g_t && \text{(细胞状态更新)} \\[6pt] + h_t &= o_t \odot \tanh(c_t) && \text{(隐藏状态更新)} + \end{aligned} + +- :math:`x_t` : 当前时间步输入向量 +- :math:`h_{t-1}` : 上一时间步的隐藏状态 +- :math:`c_{t-1}` : 上一时间步的细胞状态 +- :math:`i_t, f_t, g_t, o_t` : 四个门(输入门、遗忘门、候选门、输出门) +- :math:`W_*` : 对应的权重矩阵 +- :math:`b_*` : 偏置项 +- :math:`\sigma(\cdot)` : Sigmoid 函数 +- :math:`\odot` : 元素乘 + + + +输入: + - **input** - 输入序列数据,形状为 :math:`(seq_len, batch, input_size)`,即每个时间步的输入特征。 + - **weight_i** - 输入到各门 :math:`(input、forget、cell、output)` 的权重矩阵,大小为 4 * hidden_size * input_size。 + - **weight_h** - 上一隐藏状态到各门的权重矩阵,大小为 :math:`4 * hidden_size * hidden_size` + - **input_bias** - 输入部分的偏置项,对应 4 个门的偏置。 + - **state_bias** - 隐藏状态部分的偏置项(也是 :math:`4 * hidden_size`),与 input_bias 一起求和形成总偏置。 + - **hidden_state** - 当前批次初始隐藏状态输入( :math:`h₀` ),执行后更新为最后时刻的隐藏状态输出( :math:`hₜ`) + - **cell_state** - 当前批次初始细胞状态输入( :math:`c₀`),执行后更新为最后时刻的细胞状态输出( :math:`cₜ`)。 + - **buffer** - 临时工作区指针数组(中间计算缓存,如门值、激活结果、临时矩阵等,用于优化性能)。 + - **LstmParameter** - LSTM 配置参数结构体,包含输入大小、隐藏层维度、序列长度、是否双向等信息。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +**LstmParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct LstmParameter { + int input_size_;//每个时间步输入向量的维度(输入特征数)。 + int hidden_size_;//LSTM 隐藏状态的维度(每个门的内部计算大小)。 + int project_size_;//投影层输出维度(用于 LSTMP,有则在输出前线性压缩隐藏状态)。 + int output_size_;//实际输出维度,等于 hidden_size_ 或 project_size_(取决于是否使用投影层)。 + int seq_len_;//输入序列的时间步数(序列长度)。 + int batch_;//批次大小(一次处理的样本数量)。 + // other parameter + int output_step_;//指定输出第几个时间步的结果(通常为最后一步或每步)。 + bool bidirectional_;//是否为双向 LSTM(true 表示前向和后向各一层)。 + float zoneout_cell_;//单元状态的 Zoneout 比例(防止过拟合的正则化参数)。 + float zoneout_hidden_;//隐藏状态的 Zoneout 比例(防止过拟合)。 + int input_row_align_;//输入张量的行对齐参数(用于 DMA 或 SIMD 加速的内存对齐)。 + int input_col_align_;//输入张量的列对齐参数。 + int state_row_align_;//状态张量(hidden/cell)的行对齐参数。 + int state_col_align_;//状态张量的列对齐参数。 + int proj_col_align_;//投影层矩阵的列对齐参数。 + bool has_bias_;//是否包含偏置项(true 表示使用 bias)。 + } LstmParameter; + +输出: + - **output** - 计算结果地址,存放 LSTM 每个时间步输出结果的缓冲区,维度通常为 :math:`(seq\_len, batch, output\_size)` + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持fp32 + - MT7004 支持fp32 + +**共享存储版本:** + +.. c:function:: void fp_Lstm_s(float *output, const float *input, const float *weight_i, const float *weight_h, const float *input_bias,const float *state_bias, float *hidden_state, float *cell_state, float *buffer[9],const LstmParameter *lstm_param, int core_mask) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 40-42 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + LstmParameter *lstm_param = (LstmParameter *)0x90000000; + lstm_param->seq_len_ = 20; + lstm_param->batch_ = 1; + lstm_param->input_size_ = 2000; + lstm_param->hidden_size_ = 3; + lstm_param->bidirectional_ = false; + float * input = (float *)0xA0000000; //input在DDR空间 + float * weight_i = (float *)0xA1000000; + float * weight_h = (float *)0xA3000000; + float *input_bias_ =(float *) 0xB0900000; + float * state_bias_ =(float *) 0xB0B00000; + float * output_s = (float *)0xC0000000; + float *hidden_state_s = (float *)0xC0100000; + float *cell_state_s = (float *)0xC0200000; + float *buffer[9]; + float * packed_input_ = (float *)0xB0000000; + buffer[0] = packed_input_; + float * gate = (float *)0xB0100000; + buffer[1] = gate; + float * packed_state = (float *)0xB0200000; + buffer[2] = packed_state; + float * state_gate = (float *)0xB0300000; + buffer[3] = state_gate; + float * cell_buffer = (float *)0xB0400000; + buffer[4] = cell_buffer; + float * hidden_buffer = (float *)0xB0500000; + buffer[5] = hidden_buffer; + float * packed_output = (float *)0xB0600000; + buffer[6] = packed_output; + float * left_matrix = (float *)0xB0700000; + buffer[7] = left_matrix; + float * packed_ptr = (float *)0xB0800000; + buffer[8] = packed_ptr; + int core_mask = 0xff; + fp_Lstm_s(output_s, input, weight_i, weight_h, input_bias_, + state_bias, hidden_state_s, cell_state_s, buffer, + lstm_param, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_Lstm_p(float *output, const float *input, const float *weight_i, const float *weight_h, const float *input_bias, const float *state_bias, float *hidden_state, float *cell_state, float *buffer[9], const LstmParameter *lstm_param) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 38-40 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + LstmParameter *lstm_param = (LstmParameter *)0x10000000; + lstm_param->seq_len_ = 4; + lstm_param->batch_ = 1; + lstm_param->input_size_ = 2; + lstm_param->hidden_size_ = 3; + lstm_param->bidirectional_ = false; + float * input = (float *)0x10000200; //input在DDR空间 + float * weight_i = (float *)0x10000400; + float * weight_h = (float *)0x10000600; + float *input_bias_ =(float *) 0x10000800; + float * state_bias_ =(float *) 0x10000A00; + float * output_s = (float *)0x10000C00; + float *hidden_state_s = (float *)0x10000E00; + float *cell_state_s = (float *)0x10001000; + float *buffer[9]; + float * packed_input_ = (float *)0x10001200; + buffer[0] = packed_input_; + float * gate = (float *)0x10001400; + buffer[1] = gate; + float * packed_state = (float *)0x10001600; + buffer[2] = packed_state; + float * state_gate = (float *)0x10001800; + buffer[3] = state_gate; + float * cell_buffer = (float *)0x10001A00; + buffer[4] = cell_buffer; + float * hidden_buffer = (float *)0x10001C00; + buffer[5] = hidden_buffer; + float * packed_output = (float *)0x10001F00; + buffer[6] = packed_output; + float * left_matrix = (float *)0x10002000; + buffer[7] = left_matrix; + float * packed_ptr = (float *)0x10002200; + buffer[8] = packed_ptr; + fp_Lstm_p(output_s, input, weight_i, weight_h, input_bias_, + state_bias, hidden_state_s, cell_state_s, buffer, + lstm_param); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/matmulfusion.rst.txt b/master/html/_sources/functionlib/dsplib/matmulfusion.rst.txt new file mode 100644 index 0000000..bfbe446 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/matmulfusion.rst.txt @@ -0,0 +1,93 @@ +MatMulFusion +================= + +矩阵乘法融合(可选偏置和激活),计算: + +.. math:: + + C = \operatorname{act}(A \times B + \text{bias}) + +其中 ``A`` 形状为 :math:`M\times K`,``B`` 为 :math:`K\times N`,``C`` 与可选的 ``bias`` 为 :math:`M\times N`。 +激活 ``act`` 支持: + +- ``0``: 无激活(Identity) +- ``1``: ReLU +- ``2``: ReLU6 + +输入: + - **A** - 输入矩阵 A(行优先,连续存储)。大小 M×K。 + - **B** - 输入矩阵 B(行优先,连续存储)。大小 K×N。 + - **bias** - 偏置矩阵(可为 NULL,大小 M×N)。 + - **M, N, K** - 维度参数。 + - **activation_type** - 激活类型,取值 {0,1,2}。 + - **core_mask(可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **C** - 输出矩阵(行优先,大小 M×N)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 复数类型的激活逐分量应用于实部与虚部。 + - 请确保输入按行优先连续布局,且不发生类型范围溢出;int16/int32 计算未做饱和裁剪。 + - 共享存储版本要求地址位于 GSM/DDR/SMC 等共享可见空间。 + +**共享存储版本:** + +.. c:function:: void i16_matmulfusion_s(int16_t *A, int16_t *B, int16_t *C, int16_t *bias, int M, int N, int K, int activation_type, int core_mask) +.. c:function:: void i32_matmulfusion_s(int32_t *A, int32_t *B, int32_t *C, int32_t *bias, int M, int N, int K, int activation_type, int core_mask) +.. c:function:: void fp_matmulfusion_s(float *A, float *B, float *C, float *bias, int M, int N, int K, int activation_type, int core_mask) +.. c:function:: void dp_matmulfusion_s(double *A, double *B, double *C, double *bias, int M, int N, int K, int activation_type, int core_mask) +.. c:function:: void c64_matmulfusion_s(float complex *A, float complex *B, float complex *C, float complex *bias, int M, int N, int K, int activation_type, int core_mask) +.. c:function:: void c128_matmulfusion_s(double complex *A, double complex *B, double complex *C, double complex *bias, int M, int N, int K, int activation_type, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + #include + + int main(int argc, char* argv[]) { + float *A = (float *)0xA0000000; // DDR + float *B = (float *)0xA1000000; // DDR + float *C = (float *)0xA2000000; // DDR + float *bias = (float *)0xA3000000; // DDR,可为 NULL + int M = 512, N = 512, K = 512; + int activation_type = 1; // ReLU + int core_mask = 0xff; // 8 核 + fp_matmulfusion_s(A, B, C, bias, M, N, K, activation_type, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i16_matmulfusion_p(int16_t *A, int16_t *B, int16_t *C, int16_t *bias, int M, int N, int K, int activation_type) +.. c:function:: void i32_matmulfusion_p(int32_t *A, int32_t *B, int32_t *C, int32_t *bias, int M, int N, int K, int activation_type) +.. c:function:: void fp_matmulfusion_p(float *A, float *B, float *C, float *bias, int M, int N, int K, int activation_type) +.. c:function:: void dp_matmulfusion_p(double *A, double *B, double *C, double *bias, int M, int N, int K, int activation_type) +.. c:function:: void c64_matmulfusion_p(float complex *A, float complex *B, float complex *C, float complex *bias, int M, int N, int K, int activation_type) +.. c:function:: void c128_matmulfusion_p(double complex *A, double complex *B, double complex *C, double complex *bias, int M, int N, int K, int activation_type) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + + int main(int argc, char* argv[]) { + double *A = (double *)0x10000000; // L2 + double *B = (double *)0x10020000; // L2 + double *C = (double *)0x10040000; // L2/DDR + double *bias = NULL; // 可为 NULL + int M = 128, N = 128, K = 128; + int activation_type = 0; // None + dp_matmulfusion_p(A, B, C, bias, M, N, K, activation_type); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/raggedrange.rst.txt b/master/html/_sources/functionlib/dsplib/raggedrange.rst.txt new file mode 100644 index 0000000..e445805 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/raggedrange.rst.txt @@ -0,0 +1,88 @@ +RaggedRange +================= + +为每个区间生成等差序列并拼接到一个一维输出,同时返回每段的边界索引。 + +.. math:: + + \begin{aligned} + L_k &= \left\lceil \frac{\text{limit}_k - \text{start}_k}{\text{delta}_k} \right\rceil,\quad k=0,\dots,\text{range\_count}-1 \\ + \text{splits}[0] &= 0,\quad \text{splits}[k+1] = \text{splits}[k] + L_k \\ + \text{values}[\text{splits}[k] + j] &= \text{start}_k + j \cdot \text{delta}_k,\quad j=0,\dots,L_k-1 + \end{aligned} + +输入: + - **starts** - 各段起始值数组地址。 + - **limits** - 各段终止上界数组地址(半开区间 [start, limit) 语义)。 + - **deltas** - 各段步长数组地址,要求 ``delta_k != 0``,且方向与 ``limit_k - start_k`` 一致,否则该段长度视为 0。 + - **range_count** - 段数 K。 + - **core_mask(可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **values** - 扁平化拼接的结果数组地址,长度为 ``splits[range_count]``。 + - **splits** - 长度为 ``range_count + 1`` 的整型数组,记录各段边界,且 ``splits[0] = 0``。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 当前实现覆盖 int8、int16、int32、fp32、fp64 五种类型。 + - 不进行数值饱和裁剪;请确保序列不会在对应类型范围内溢出。 + - 共享存储版本要求输入/输出地址位于共享可见的存储空间(如 GSM/DDR/SMC 等)。 + +**共享存储版本:** + +.. c:function:: void i8_raggedrange_s(int8_t *starts, int8_t *limits, int8_t *deltas, int range_count, int8_t *values, int *splits, int core_mask) +.. c:function:: void i16_raggedrange_s(int16_t *starts, int16_t *limits, int16_t *deltas, int range_count, int16_t *values, int *splits, int core_mask) +.. c:function:: void i32_raggedrange_s(int32_t *starts, int32_t *limits, int32_t *deltas, int range_count, int32_t *values, int *splits, int core_mask) +.. c:function:: void fp_raggedrange_s(float *starts, float *limits, float *deltas, int range_count, float *values, int *splits, int core_mask) +.. c:function:: void dp_raggedrange_s(double *starts, double *limits, double *deltas, int range_count, double *values, int *splits, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + #include + + int main(int argc, char* argv[]) { + float *starts = (float *)0xA0000000; // DDR + float *limits = (float *)0xA0100000; // DDR + float *deltas = (float *)0xA0200000; // DDR + float *values = (float *)0xA0300000; // DDR + int *splits = (int *)0xA0400000; // DDR + int range_count = 3; + int core_mask = 0xff; + fp_raggedrange_s(starts, limits, deltas, range_count, values, splits, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_raggedrange_p(int8_t *starts, int8_t *limits, int8_t *deltas, int range_count, int8_t *values, int *splits) +.. c:function:: void i16_raggedrange_p(int16_t *starts, int16_t *limits, int16_t *deltas, int range_count, int16_t *values, int *splits) +.. c:function:: void i32_raggedrange_p(int32_t *starts, int32_t *limits, int32_t *deltas, int range_count, int32_t *values, int *splits) +.. c:function:: void fp_raggedrange_p(float *starts, float *limits, float *deltas, int range_count, float *values, int *splits) +.. c:function:: void dp_raggedrange_p(double *starts, double *limits, double *deltas, int range_count, double *values, int *splits) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + + int main(int argc, char* argv[]) { + int32_t *starts = (int32_t *)0x10000000; // L2 + int32_t *limits = (int32_t *)0x10001000; // L2 + int32_t *deltas = (int32_t *)0x10002000; // L2 + int32_t *values = (int32_t *)0x10003000; // L2/DDR + int *splits = (int *)0x10004000; // L2/DDR + int range_count = 2; + i32_raggedrange_p(starts, limits, deltas, range_count, values, splits); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/range.rst.txt b/master/html/_sources/functionlib/dsplib/range.rst.txt new file mode 100644 index 0000000..f5e7db7 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/range.rst.txt @@ -0,0 +1,78 @@ +Range +================= + +生成等差序列,逐元素写入输出: + +.. math:: + + output_i = start + i \times delta,\quad i = 0, 1, \dots, \text{length} - 1 + +输入: + - **start** - 序列起始值。 + - **delta** - 公差。 + - **length** - 序列长度。 + - **core_mask(可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 输出数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64。 + - MT7004 支持 fp16, fp32, int16, int32(如需 fp16,请使用相应 fp16 实现)。 + +**共享存储版本:** + +.. c:function:: void i8_range_s(int8_t* output, int8_t start, int8_t delta, int length, int core_mask) +.. c:function:: void i16_range_s(int16_t* output, int16_t start, int16_t delta, int length, int core_mask) +.. c:function:: void i32_range_s(int32_t* output, int32_t start, int32_t delta, int length, int core_mask) +.. c:function:: void fp_range_s(float* output, float start, float delta, int length, int core_mask) +.. c:function:: void dp_range_s(double* output, double start, double delta, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0xA0000000; // DDR + float start = 1.5f; + float delta = 0.5f; + int length = 1000; + int core_mask = 0xff; + fp_range_s(output, start, delta, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_range_p(int8_t* output, int8_t start, int8_t delta, int length) +.. c:function:: void i16_range_p(int16_t* output, int16_t start, int16_t delta, int length) +.. c:function:: void i32_range_p(int32_t* output, int32_t start, int32_t delta, int length) +.. c:function:: void fp_range_p(float* output, float start, float delta, int length) +.. c:function:: void dp_range_p(double* output, double start, double delta, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0x10000000; // L2 + float start = 1.5f; + float delta = 0.5f; + int length = 1000; + fp_range_p(output, start, delta, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/reduce.rst.txt b/master/html/_sources/functionlib/dsplib/reduce.rst.txt new file mode 100644 index 0000000..ad2646e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/reduce.rst.txt @@ -0,0 +1,211 @@ +Reduce +================= + +对指定维度进行归约。 + +输入: + - **src_data** - 输入数据的地址 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**ReduceParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct ReduceParameter { + void** data_buffers_; // 用于存储中间计算结果 + int* outer_sizes_; // 处理某个规约轴时,该轴之前所有轴的元素数 + int* inner_sizes_; // 某个规约轴之后的所有元素数 + int* axis_sizes_; // 规约轴的元素数 + int total_num_; // 输入张量的总元素数 + int num_axes_; // 待规约轴的数目 + int mode_; // 规约模式 + int output_num_; // 输出张量的总元素数 + /**该算子会根据ReduceParameter中的mode_参数选择实际规约所使用的方法。共有如下几种方法: + Reduce_Mean=0, + Reduce_Max=1, + Reduce_Min=2, + Reduce_Prod=3, + Reduce_Sum=4, + Reduce_SumSquare=5, + Reduce_ASum=6, + Reduce_L2Norm=7 + **/ + } ReduceParameter; + +输出: + - **dst_data** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64 + - MT7004 支持fp16, fp32, int16, int32 + +**共享存储版本:** + +.. c:function:: void i8_reduce_s(int8_t* src_data, int8_t* dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void i16_reduce_s(int16_t* src_data, half* dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void i32_reduce_s(int* src_data, float* dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void hp_reduce_s(half* src_data, half* dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void fp_reduce_s(float* src_data, float* dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void dp_reduce_s(double* src_data, double* dst_data, ReduceParameter* param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 50 + + void PackParam(ReduceParameter* param, int ndim, int* input_shape, int num_axes, int* axes) { + int tmp_input_shape[8]; + int total_num = 1; + int i, j, k; + for (i = 0; i < ndim; i++) { + tmp_input_shape[i] = input_shape[i]; + total_num *= input_shape[i]; + } + param->total_num_ = total_num; + int offset_size = 0; + for (i = 0; i < num_axes; ++i) { + int axis = axes[i]; + int outer_size = 1; + for (j = 0; j < axis; j++) { + outer_size *= tmp_input_shape[j]; + } + param->outer_sizes_[offset_size] = outer_size; + int inner_size = 1; + for (k = axis + 1; k < ndim; k++) { + inner_size *= tmp_input_shape[k]; + } + param->inner_sizes_[offset_size] = inner_size; + param->axis_sizes_[offset_size] = tmp_input_shape[axis]; + offset_size++; + tmp_input_shape[axis] = 1; + } + } + + void TestReduceSMCFp32(int* input_shape, int ndim, int* axes, int num_axes, int mode, int keep_dims, int core_mask) { + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int core_num = GetCoreNum(core_mask); + float* input = (float*)0x88000000; + float* output = (float*)0x98000000; + ReduceParameter* param = (ReduceParameter*)0xA8480000; + if (logic_core_id == 0) { + param->num_axes_ = num_axes; + param->mode_ = mode; + param->data_buffers_ = (void**)0xA8483000; + param->inner_sizes_ = (int*)0xA8484000; + param->outer_sizes_ = (int*)0xA8485000; + param->axis_sizes_ = (int*)0xA8486000; + int i; + for (i = 0; i < num_axes - 1; i++) { + param->data_buffers_[i] = (void*)(0xA8490000 + 0x1000000); + } + PackParam(param, ndim, input_shape, num_axes, axes); + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_reduce_s(input, check, param, core_mask); + } + + void main(){ + int input_shape[3] = {4, 5, 5}; + int ndim = 3; + int axes[1] = {1}; + int num_axes = 1; + int mode = 7; + int keep_dims = 1; + int core_mask = 0b1111; + TestReduceSMCFp32(input_shape, ndim, axes, num_axes, mode, keep_dims, core_mask); + } + +**私有存储版本:** + +.. c:function:: void i8_reduce_p(int8_t* src_data, int8_t* dst_data, void* tmp_src_data, void* tmp_dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void i16_reduce_p(int16_t* src_data, half* dst_data, void* tmp_src_data, void* tmp_dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void i32_reduce_p(int* src_data, float* dst_data, void* tmp_src_data, void* tmp_dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void hp_reduce_p(half* src_data, half* dst_data, void* tmp_src_data, void* tmp_dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void fp_reduce_p(float* src_data, float* dst_data, void* tmp_src_data, void* tmp_dst_data, ReduceParameter* param, int core_mask) +.. c:function:: void dp_reduce_p(double* src_data, double* dst_data, void* tmp_src_data, void* tmp_dst_data, ReduceParameter* param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 59 + + void PackParam(ReduceParameter* param, int ndim, int* input_shape, int num_axes, int* axes) { + int tmp_input_shape[8]; + int total_num = 1; + int i, j, k; + for (i = 0; i < ndim; i++) { + tmp_input_shape[i] = input_shape[i]; + total_num *= input_shape[i]; + } + param->total_num_ = total_num; + int offset_size = 0; + for (i = 0; i < num_axes; ++i) { + int axis = axes[i]; + int outer_size = 1; + for (j = 0; j < axis; j++) { + outer_size *= tmp_input_shape[j]; + } + param->outer_sizes_[offset_size] = outer_size; + int inner_size = 1; + for (k = axis + 1; k < ndim; k++) { + inner_size *= tmp_input_shape[k]; + } + param->inner_sizes_[offset_size] = inner_size; + param->axis_sizes_[offset_size] = tmp_input_shape[axis]; + offset_size++; + tmp_input_shape[axis] = 1; + } + } + + void TestReduceL2Fp32(int* input_shape, int ndim, int* axes, int num_axes, int mode, int keep_dims, int core_mask) { + float* input = (float*)0x10000000; // 原始输入输出数据需分配在AM中 + float* output = (float*)0x10010000; + float* tmp_input = (float*)0x88000000; // 临时输入输出空间需分配在DDR或SMC中 + float* tmp_output = (float*)0x98000000; + ReduceParameter* param = (ReduceParameter*)0x10020000; + param->num_axes_ = num_axes; + param->mode_ = mode; + param->data_buffers_ = (void**)0x10021000; + param->inner_sizes_ = (int*)0x10022000; + param->outer_sizes_ = (int*)0x10023000; + param->axis_sizes_ = (int*)0x10024000; + int i, j; + for (i = 0; i < ndim; i++) { + int reduce_axis = 0; + for (j = 0; j < num_axes; j++) { + if (axes[j] == i) { + reduce_axis = 1; + break; + } + } + if (!reduce_axis) { + length *= input_shape[i]; + } + } + for (i = 0; i < num_axes - 1; i++) { + param->data_buffers_[i] = (void*)(0xA8490000 + 0x1000000); // 每一个中间计算结果空间都需分配在DDR或SMC中 + } + param->output_num_ = length; + PackParam(param, ndim, input_shape, num_axes, axes); + fp_reduce_p(input, check, param, core_mask); + } + + void main() { + int input_shape[3] = {4, 5, 5}; + int ndim = 3; + int axes[1] = {1}; + int num_axes = 1; + int mode = 7; + int keep_dims = 1; + int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动 + TestReduceL2Fp32(input_shape, ndim, axes, num_axes, mode, keep_dims, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/resize.rst.txt b/master/html/_sources/functionlib/dsplib/resize.rst.txt new file mode 100644 index 0000000..16992d5 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/resize.rst.txt @@ -0,0 +1,96 @@ +Resize +================= + +对输入图像使用给定的插值方式去调整为给定的尺寸大小。 + +输入: + - **input** - 输入数据的地址 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**ResizeParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct ResizeParameter { + int* input_shape_; // 输入张量形状 + int* output_shape_; // 输出张量形状 + int* x_lefts_; // 用于存储预处理结果 + int* x_rights_; // 用于存储预处理结果 + int* y_tops_; // 用于存储预处理结果 + int* y_bottoms_; // 用于存储预处理结果 + void* x_weights_; // 用于存储预处理结果 + void* y_weights_; // 用于存储预处理结果 + void* line_buffers_; // 用于存储中间结果 + int method_; // 所用的插值方法,0:最邻近插值,1:双线性插值,2:双三次插值 + int coordinate_transform_mode_; // 像素点对齐方式,0:非对称,1:中心对齐,2:偏移半像素 + float cubic_coeff_; // 一个仅在双三次插值中使用到的系数 + } ResizeParameter; + +输出: + - **output** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享/私有存储版本:** + +.. c:function:: void i8_resize_anycore(int8_t* input, int8_t* output, ResizeParameter* param, int core_mask) +.. c:function:: void hp_resize_anycore(half* input, half* output, ResizeParameter* param, int core_mask) +.. c:function:: void fp_resize_anycore(float* input, float* output, ResizeParameter* param, int core_mask) + +私有及共享空间版本均使用这些函数。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 30 + + void TestResizeFp32SMC(int* input_shape, int* output_shape, ResizeMethod method, CoordinateTransformMode mode, float cubic_coeff, int core_mask) { + int core_id = get_core_id(); + int core_num = GetCoreNum(core_mask); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + float* input = (float*)0x84000000; // 测试私有空间时地址设置在私有空间内即可 + float* output = (float*)0x85000000; + ResizeParameter* param = (ResizeParameter*)0x86000000; + if (logic_core_id == 0) { + param->coordinate_transform_mode_ = mode; + param->method_ = method; + param->cubic_coeff_ = cubic_coeff; + param->input_shape_ = (int*)0x87000000; + memcpy(param->input_shape_, input_shape, sizeof(int) * 4); + param->output_shape_ = (int*)0x87100000; + memcpy(param->output_shape_, output_shape, sizeof(int) * 4); + param->line_buffers_ = (void*)0x88000000; + param->x_lefts_ = (int*)0x89000000; + param->x_rights_ = (int*)0x8A000000; + param->y_bottoms_ = (int*)0x8B000000; + param->y_tops_ = (int*)0x8C000000; + param->x_weights_ = (void*)0x8D000000; + param->y_weights_ = (void*)0x8E000000; + if (method == BILINEAR) { + PrepareResizeBilinear(param); // 做预处理 + } else if (method == CUBIC) { + PrepareResizeBicubic(param, cubic_coeff); // 做预处理 + } + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_resize_anycore(input, check, param, core_mask); + } + + void main(){ + int input_shape[4] = {2, 4, 4, 4}; + int output_shape[4] = {2, 8, 8, 4}; + ResizeMethod method = NEAREST; + CoordinateTransformMode mode = ALIGN_CORNERS; + float cubic_coeff = -0.75; + int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可 + TestResizeFp32SMC(input_shape, output_shape, method, mode, cubic_coeff, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/reverse_sequence.rst.txt b/master/html/_sources/functionlib/dsplib/reverse_sequence.rst.txt new file mode 100644 index 0000000..49c00ad --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/reverse_sequence.rst.txt @@ -0,0 +1,105 @@ +ReverseSequence +================= + +对输入序列进行部分反转。 + +输入: + - **src** - 需反转数据的地址 + - **seq_lengths** - 指定反转长度,为一维向量。 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**ReverseSequenceParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct ReverseSequenceParameter { + int* shape_; // 输入和输出张量的形状 + int* strides_; // 一个记录张量每一维步长的数组 + int seq_dim_; // 指定反转的维度 + int batch_dim_; // 指定切片维度 + int ndim_; // 输入和输出张量的维度 + int type_size_; // 输入和输出张量数据类型的长度 + int copy_elem_num_; // 在inner_count循环中一次应被拷贝的元素数目 + int outer_stride_; // 外层元素间的步长 + int outer_count_; // 外层元素数 + int inner_stride_; // 内层元素间的步长 + int inner_count_; // 内层元素数 + } ReverseSequenceParameter; + +输出: + - **dst** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享/私有存储版本:** + +.. c:function:: void anytype_reverse_sequence_anycore(void* src, void* dst, int* seq_lengths, ReverseSequenceParameter* param, int core_mask) + +各种数据类型、私有及共享空间版本均使用该函数。对于不同数据类型,改变param中的type_size_参数即可。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 41 + + void Resize(ReverseSequenceParameter* param) { + param->strides_ = (int*)0xA8020000; + ComputeStrides(param->shape_, param->strides_, param->ndim_); + int less_dim = param->batch_dim_ > param->seq_dim_ ? param->seq_dim_ : param->batch_dim_; + int greater_dim = param->batch_dim_ < param->seq_dim_ ? param->seq_dim_ : param->batch_dim_; + // calculate the size of elements should be copied at one time + param->copy_elem_num_ = CountElementAfterDim(param->shape_, greater_dim, param->ndim_); + // calculate the number of elements before the less axis and the stride + param->outer_count_ = CountElementBeforeDim(param->shape_, less_dim); + param->outer_stride_ = param->shape_[less_dim] * CountElementAfterDim(param->shape_, less_dim, param->ndim_); + // calculate the number of elements between the less axis and the greater axis and the stride + param->inner_count_ = 1; + int i; + for (i = less_dim + 1; i < greater_dim; i++) { + param->inner_count_ *= param->shape_[i]; + } + param->inner_stride_ = param->shape_[greater_dim] * CountElementAfterDim(param->shape_, greater_dim, param->ndim_); + } + + void TestReverseSequenceFp32(int* shape, int* orig_seq_lengths, int seq_dim, int batch_dim, int ndim, int core_mask) { + int core_id = get_core_id(); + int core_num = GetCoreNum(core_mask); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + float* input_data = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可 + float* output_data = (float*)0x98000000; + float* check = (float*)0xB8000000; + int* seq_lengths = (int*)0xC8000000; + ReverseSequenceParameter* param = (ReverseSequenceParameter*)0xA8000000; + int i; + if (logic_core_id == 0) { + param->shape_ = (int*)0xA8010000; + memcpy(param->shape_, shape, sizeof(int) * ndim); + memcpy(seq_lengths, orig_seq_lengths, sizeof(int) * shape[batch_dim]); + param->batch_dim_ = batch_dim; + param->seq_dim_ = seq_dim; + param->ndim_ = ndim; + param->type_size_ = sizeof(float); + Resize(param); + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + anytype_reverse_sequence_anycore(input_data, output_data, seq_lengths, param, core_mask); + } + + void main(){ + int shape[3] = {3, 4, 10}; + int seq_lengths[3] = {1, 2, 3}; + int ndim = 3; + int seq_dim = 1; + int batch_dim = 0; + int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可 + TestReverseSequenceFp32(shape, seq_lengths, seq_dim, batch_dim, ndim, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/reversev2.rst.txt b/master/html/_sources/functionlib/dsplib/reversev2.rst.txt new file mode 100644 index 0000000..3efed46 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/reversev2.rst.txt @@ -0,0 +1,109 @@ +ReverseV2 +================= + +对输入Tensor按指定维度反转。 + +输入: + - **src** - 需反转数据的地址 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**ReverseV2Parameter定义:** + +.. code-block:: c + :linenos: + + typedef struct ReverseV2Parameter { + int* axis_flag_; // 用于存储需反转的轴,需反转的轴对应于数组索引的元素被置为1,其余元素为0 + int* input_shape_; // 输入张量的形状 + int* input_strides_; // 一个记录输入张量每一维步长的数组 + int** cur_coord_; // 二维数组,每个元素存储对应核心所使用的cur_coord数组地址,这个数组用于记录当前循环到的元素坐标 + int ndim_; // 输入和输出张量的维度 + int axis_ndim_; // 需反转的轴的数目 + int num_elem_; // 元素总数 + int type_size_; // 输入和输出张量数据类型的长度 + } ReverseV2Parameter; + +输出: + - **dst** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享/私有存储版本:** + +.. c:function:: void anytype_reversev2_anycore(void* src, void* dst, ReverseV2Parameter* param, int core_mask) + +各种数据类型、私有及共享空间版本均使用该函数。对于不同数据类型,改变param中的type_size_参数即可。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 50 + + void Resize(ReverseV2Parameter* param, int* axis) { + int i; + param->input_strides_[param->ndim_ - 1] = 1; + for (i = param->ndim_ - 1; i > 0; i--) { + param->input_strides_[i - 1] = param->input_strides_[i] * param->input_shape_[i]; + } + for (i = 0; i < param->ndim_; i++) { + // initialize axis_flag array by 0 + param->axis_flag_[i] = 0; + } + for (i = 0; i < param->axis_ndim_; i++) { + param->axis_flag_[axis[i]] = 1; + } + param->num_elem_ = 1; + for (i = 0; i < param->ndim_; i++) { + param->num_elem_ *= param->input_shape_[i]; + } + } + + void TestReverseV2(int* shape, int ndim, int* axis, int axis_ndim, int core_mask) { + int type_size = 4; + int core_num = GetCoreNum(core_mask); + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + float* input_data = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可 + float* output_data = (float*)0xA8000000; + float* check = (float*)0xC8000000; + ReverseV2Parameter* param = (ReverseV2Parameter*)0x84000000; + if (logic_core_id == 0) { + int i, j; + param->axis_ndim_ = axis_ndim; + param->input_shape_ = (int*)0x84003000; + for (i = 0; i < ndim; i++) { + param->input_shape_[i] = shape[i]; + } + param->ndim_ = ndim; + param->type_size_ = type_size; + param->input_strides_ = (int*)0x84004000; + param->axis_flag_ = (int*)0x84005000; + param->cur_coord_ = (int**)0x84006000; + for (i = 0; i < 4; i++) { + param->cur_coord_[i] = (int*)(0x84007000 + i * 0x1000LL); + for (j = 0; j < ndim; j++) { + param->cur_coord_[i][j] = 0; + } + } + Resize(param, axis); + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + anytype_reversev2_anycore(input_data, output_data, param, core_mask); + } + + void main(){ + int shape[3] = {10, 10, 1000}; + int ndim = 3; + int axis[3] = {0, 1, 2}; + int axis_ndim = 3; + int core_mask = 0b1111; + TestReverseV2(shape, ndim, axis, axis_ndim, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/scalefusion.rst.txt b/master/html/_sources/functionlib/dsplib/scalefusion.rst.txt new file mode 100644 index 0000000..d472537 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/scalefusion.rst.txt @@ -0,0 +1,89 @@ +ScaleFusion +================= + + + + 传入一个数组,逐元素乘上因数,并加上偏置的值后输出。 + + .. math:: + + dst_i = src_i \cdot scale_i + bias_i + + 输入: + - **src_data** - 输入数据地址。 + - **length** - 计算长度。 + - **scale** - 缩放因子数组首地址。 + - **bias** - 偏置数组首地址。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **dst_data** - 计算结果地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64 + - MT7004 支持fp16, fp32, int16, int32 + +**共享存储版本:** + +.. c:function:: void i8_scalefusion_s(int8_t* src_data, int8_t* dst_data, int length, float* scale, float* bias, int core_mask) +.. c:function:: void i16_scalefusion_s(int16_t* src_data, int16_t* dst_data, int length, float* scale, float* bias, int core_mask) +.. c:function:: void i32_scalefusion_s(int* src_data, int* dst_data, int length, float* scale, float* bias, int core_mask) +.. c:function:: void hp_scalefusion_s(half* src_data, half* dst_data, int length, half* scale, half* bias, int core_mask) +.. c:function:: void fp_scalefusion_s(float* src_data, float* dst_data, int length, float* scale, float* bias, int core_mask) +.. c:function:: void dp_scalefusion_s(double* src_data, double* dst_data, int length, double* scale, double* bias, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + float *scale = (float *)0xB0000000; //scale在DDR空间 + float *bias = (float *)0xB1000000; //bias在DDR空间 + int length = 1000; + int core_mask = 0xff; + fp_scalefusion_s( input0, output, length, scale, bias, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_scalefusion_p(int8_t* src_data, int8_t* dst_data, int length, float* scale, float* bias) +.. c:function:: void i16_scalefusion_p(int16_t* src_data, int16_t* dst_data, int length, float* scale, float* bias) +.. c:function:: void i32_scalefusion_p(int* src_data, int* dst_data, int length, float* scale, float* bias) +.. c:function:: void hp_scalefusion_p(half* src_data, half* dst_data, int length, half* scale, half* bias) +.. c:function:: void fp_scalefusion_p(float* src_data, float* dst_data, int length, float* scale, float* bias) +.. c:function:: void dp_scalefusion_p(double* src_data, double* dst_data, int length, double* scale, double* bias) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; //input在L2空间 + float *output = (float *)0x10820000; + float *scale = (float *)0x10830000; //scale在L2空间 + float *bias = (float *)0x10840000; //bias在L2空间 + int length = 1000; + fp_scalefusion_p( input0, output, length, scale, bias); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/scatter_elements.rst.txt b/master/html/_sources/functionlib/dsplib/scatter_elements.rst.txt new file mode 100644 index 0000000..986ecae --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/scatter_elements.rst.txt @@ -0,0 +1,178 @@ +ScatterElements +================= + +返回一个新tensor,根据指定索引和更新值对input中的元素进行指定操作(替换或相加)。不支持隐式类型转换。举例:一个三维输入tensor的返回为: + +.. code-block:: python + + output[indices[i][j][k]][j][k] = updates[i][j][k] #if axis == 0, reduction == "none" + output[i][indices[i][j][k]][k] += updates[i][j][k] #if axis == 1, reduction == "add" + output[i][j][indices[i][j][k]] = updates[i][j][k] #if axis == 2, reduction == "none" + +输入: + - **input** - 输入数据的地址 + - **indices** - 指定索引。 + - **updates** - 更新值。 + - **param** - 算子计算所需参数的结构体。其各成员见下述。 + - **core_mask** - 核掩码。 + +**ScatterElementsParameter定义:** + +.. code-block:: c + :linenos: + + typedef struct ScatterElementsParameter { + int* indices_stride_; // 对应于indices数组每一维度的步长 + int* output_stride_; // 对应于output数组每一维度的步长 + int input_dims_; // 输入张量的维度数 + int axis_; // 指定索引所在的轴 + int input_axis_size_; // 索引所在轴的元素数 + int indices_total_num_; // indices数组的总元素数 + int input_total_num_; // input数组的总元素数 + int reduction_type_; // 规约类型,0代表none,1代表add + } ScatterElementsParameter; + +输出: + - **output** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + - 如果 indices 中有多个索引向量对应于同一位置,则输出中该位置值是不确定的。 + - 如果 indices 的值超出 input 索引上下界,则相应的 updates 不会更新到 input,也不会抛出索引错误。 + +**共享存储版本:** + +.. c:function:: void i8_scatter_elements_s(int8_t* input, int8_t* output, int* indices, int8_t* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void i16_scatter_elements_s(int16_t* input, int16_t* output, int* indices, int16_t* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void i32_scatter_elements_s(int* input, int* output, int* indices, int* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void hp_scatter_elements_s(half* input, half* output, int* indices, half* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void fp_scatter_elements_s(float* input, float* output, int* indices, float* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void dp_scatter_elements_s(double* input, double* output, int* indices, double* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void c64_scatter_elements_s(float* input, float* output, int* indices, float* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void c128_scatter_elements_s(double* input, double* output, int* indices, double* updates, ScatterElementsParameter* param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 40 + + void PackParam(ScatterElementsParameter* param, int* indices_shape, int* input_shape) { + param->indices_stride_[param->input_dims_ - 1] = 1; + int i; + for (i = param->input_dims_ - 1; i > 0; --i) { + param->indices_stride_[i - 1] = param->indices_stride_[i] * indices_shape[i]; + } + param->output_stride_[param->input_dims_ - 1] = 1; + for (i = param->input_dims_ - 1; i > 0; --i) { + param->output_stride_[i - 1] = param->output_stride_[i] * input_shape[i]; + } + param->indices_total_num_ = 1; + for (i = 0; i < param->input_dims_; i++) { + param->indices_total_num_ *= indices_shape[i]; + } + param->input_total_num_ = 1; + for (i = 0; i < param->input_dims_; i++) { + param->input_total_num_ *= input_shape[i]; + } + param->input_axis_size_ = input_shape[param->axis_]; + } + + void TestScatterElementsSMC(int* input_shape, int* indices_shape, int ndim, int axis, int reduction_type, int core_mask) { + int core_num = GetCoreNum(core_mask); + int core_id = get_core_id(); + int logic_core_id = GetLogicCoreId(core_mask, core_id); + void* input_data = (void*)0x88000000; + void* output_data = (void*)0x98000000; + int* indices_data = (int*)0xA8000000; + void* updates_data = (void*)0xB8000000; + ScatterElementsParameter* param = (ScatterElementsParameter*)0xC8000000; + if (logic_core_id == 0) { + param->axis_ = axis; + param->input_dims_ = ndim; + param->indices_stride_ = (int*)0xC8020000; + param->output_stride_ = (int*)0xC8040000; + param->reduction_type_ = reduction_type; + PackParam(param, indices_shape, input_shape); + } + sys_bar(0, core_num); // 初始化参数完成后进行同步 + fp_scatter_elements_s(input_data, output_data, indices_data, updates_data, param, core_mask); + } + + void main() { + int input_shape[2] = {8, 30}; + int indices_shape[2] = {3, 3}; + int ndim = 2; + int axis = 0; + int reduction_type = 0; + int core_mask = 0b1111; + TestScatterElementsSMC(input_shape, indices_shape, ndim, axis, reduction_type, core_mask); + } + +**私有存储版本:** + +.. c:function:: void i8_scatter_elements_p(int8_t* input, int8_t* output, int* indices, int8_t* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void i16_scatter_elements_p(int16_t* input, int16_t* output, int* indices, int16_t* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void i32_scatter_elements_p(int* input, int* output, int* indices, int* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void hp_scatter_elements_p(half* input, half* output, int* indices, half* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void fp_scatter_elements_p(float* input, float* output, int* indices, float* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void dp_scatter_elements_p(double* input, double* output, int* indices, double* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void c64_scatter_elements_p(float* input, float* output, int* indices, float* updates, ScatterElementsParameter* param, int core_mask) +.. c:function:: void c128_scatter_elements_p(double* input, double* output, int* indices, double* updates, ScatterElementsParameter* param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 34 + + void PackParam(ScatterElementsParameter* param, int* indices_shape, int* input_shape) { + param->indices_stride_[param->input_dims_ - 1] = 1; + int i; + for (i = param->input_dims_ - 1; i > 0; --i) { + param->indices_stride_[i - 1] = param->indices_stride_[i] * indices_shape[i]; + } + param->output_stride_[param->input_dims_ - 1] = 1; + for (i = param->input_dims_ - 1; i > 0; --i) { + param->output_stride_[i - 1] = param->output_stride_[i] * input_shape[i]; + } + param->indices_total_num_ = 1; + for (i = 0; i < param->input_dims_; i++) { + param->indices_total_num_ *= indices_shape[i]; + } + param->input_total_num_ = 1; + for (i = 0; i < param->input_dims_; i++) { + param->input_total_num_ *= input_shape[i]; + } + param->input_axis_size_ = input_shape[param->axis_]; + } + + void TestScatterElementsL2(int* input_shape, int* indices_shape, int ndim, int axis, int reduction_type, int core_mask) { + void* input_data = (void*)0x10000000; // 私有存储版本地址设置在AM内 + void* output_data = (void*)0x10001000; + int* indices_data = (int*)0x10002000; + void* updates_data = (void*)0x10003000; + ScatterElementsParameter* param = (ScatterElementsParameter*)0x10004000; + param->axis_ = axis; + param->input_dims_ = ndim; + param->indices_stride_ = (int*)0x10005000; + param->output_stride_ = (int*)0x10006000; + param->reduction_type_ = reduction_type; + PackParam(param, indices_shape, input_shape); + fp_scatter_elements_p(input_data, output_data, indices_data, updates_data, param, core_mask); + } + + void main() { + int input_shape[2] = {8, 30}; + int indices_shape[2] = {3, 3}; + int ndim = 2; + int axis = 0; + int reduction_type = 0; + int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动 + TestScatterElementsL2(input_shape, indices_shape, ndim, axis, reduction_type, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/sgd.rst.txt b/master/html/_sources/functionlib/dsplib/sgd.rst.txt new file mode 100644 index 0000000..b4bb87c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sgd.rst.txt @@ -0,0 +1,109 @@ +SGD +================= + 对权重张量执行带动量与权重衰减的随机梯度下降更新。 + + 输入: + - **weight** - 待更新权重张量首地址。 + - **accumulate** - 动量累积张量首地址。 + - **gradient** - 梯度张量首地址。 + - **learning_rate** - 学习率。 + - **dampening** - 动量阻尼系数。 + - **moment** - 动量系数。 + - **nesterov** - 是否启用 Nesterov 动量。 + - **weight_decay** - 权重衰减系数。 + - **start** - 参与计算的起始索引(闭区间)。 + - **end** - 参与计算的结束索引(开区间)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **weight** - 原地写回更新后的权重张量。 + - **accumulate** - 原地写回更新后的动量张量。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32 数据类型。 + - MT7004 支持 fp16、fp32 数据类型。 + + +**共享存储版本:** + +.. c:function:: void hp_sgd_s(half *weight, half *accumulate, const half *gradient, float learning_rate, float dampening, float moment, bool nesterov, float weight_decay, int start, int end, int core_mask) +.. c:function:: void fp_sgd_s(float *weight, float *accumulate, const float *gradient, float learning_rate, float dampening, float moment, bool nesterov, float weight_decay, int start, int end, int core_mask) + + + + .. math:: + + \begin{aligned} + g'_t &= g_t + weight\_decay \cdot w_{t-1} \\ + m_t &= moment \cdot m_{t-1} + (1 - dampening) \cdot g'_t \\ + u_t &= + \begin{cases} + m_t \cdot moment + g'_t, & \text{if nesterov = True} \\ + m_t, & \text{otherwise} + \end{cases} \\ + w_t &= w_{t-1} - learning\_rate \cdot u_t + \end{aligned} + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 17 + + // FT78NE 多核示例 + #include + #include + + int main(void) { + float *weight = (float *)0xA0000000; // DDR 存储 + float *accumulate = (float *)0xB0000000; + float *gradient = (float *)0xC0000000; + int start = 0; + int end = 4096; + int core_mask = 0xff; + float learning_rate = 1e-2f; + float dampening = 0.0f; + float moment = 0.9f; + bool nesterov = true; + float weight_decay = 1e-2f; + fp_sgd_s(weight, accumulate, gradient, learning_rate, + dampening, moment, nesterov, weight_decay, + start, end, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_sgd_p(half *weight, half *accumulate, const half *gradient, float learning_rate, float dampening, float moment, bool nesterov, float weight_decay, int length) +.. c:function:: void fp_sgd_p(float *weight, float *accumulate, const float *gradient, float learning_rate, float dampening, float moment, bool nesterov, float weight_decay, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // MT7004 单核示例 + #include + #include + + int main(void) { + half *weight = (half *)0x10000000; // L2 存储 + half *accumulate = (half *)0x10002000; + half *gradient = (half *)0x10004000; + int length = 2048; + float learning_rate = 5e-3f; + float dampening = 0.0f; + float moment = 0.9f; + bool nesterov = false; + float weight_decay = 5e-3f; + hp_sgd_p(weight, accumulate, gradient, learning_rate, + dampening, moment, nesterov, weight_decay, + length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/spacetobatch.rst.txt b/master/html/_sources/functionlib/dsplib/spacetobatch.rst.txt new file mode 100644 index 0000000..d8391ab --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/spacetobatch.rst.txt @@ -0,0 +1,86 @@ +SpaceToBatch +================= +将输入在空间维度按照 block_size 分块并重新排列到 batch 维度,同时根据 paddings 执行必要的零填充。 + + 输入: + - **input** - 输入数据地址。 + - **block_size** - 分块因子,格式为 ``[block_h, block_w]``。 + - **paddings** - 填充参数,格式为 ``[top, bottom, left, right]``。 + - **input_shape** - 输入形状,格式为 ``[batch, height, width, channel]``。 + - **data_size** - 单个元素字节数(例如 sizeof(float))。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 输出数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。 + - MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。 + + +**共享存储版本:** + +.. c:function:: void i8_spacetobatch_s(int8_t *input, int8_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void i16_spacetobatch_s(int16_t *input, int16_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void i32_spacetobatch_s(int32_t *input, int32_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void hp_spacetobatch_s(half *input, half *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void fp_spacetobatch_s(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void dp_spacetobatch_s(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void c64_spacetobatch_s(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void c128_spacetobatch_s(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) + + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 多核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0xA0000000; // 输入在 DDR 地址 0xA0000000 + float *output = (float *)0xB0000000; // 输出在 DDR 地址 0xB0000000 + int block_size[2] = {2, 2}; + int paddings[4] = {0, 0, 0, 0}; + int input_shape[4] = {1, 100, 10, 10}; + int core_mask = 0xff; + fp_spacetobatch_s(input, output, block_size, paddings, input_shape, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_spacetobatch_p(int8_t *input, int8_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void i16_spacetobatch_p(int16_t *input, int16_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void i32_spacetobatch_p(int32_t *input, int32_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void hp_spacetobatch_p(half *input, half *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void fp_spacetobatch_p(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void dp_spacetobatch_p(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void c64_spacetobatch_p(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void c128_spacetobatch_p(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 单核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0x10000000; // 单核版本:输入放在 L2 地址 0x10000000 + float *output = (float *)0x10040000; // 单核版本:输出放在 L2 地址 0x10040000 + int block_size[2] = {2, 2}; + int paddings[4] = {0, 0, 0, 0}; + int input_shape[4] = {1, 100, 10, 10}; + fp_spacetobatch_p(input, output, block_size, paddings, input_shape, sizeof(float)); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/spacetobatchnd.rst.txt b/master/html/_sources/functionlib/dsplib/spacetobatchnd.rst.txt new file mode 100644 index 0000000..3afde44 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/spacetobatchnd.rst.txt @@ -0,0 +1,87 @@ +SpaceToBatchND +================= +将输入在空间维度按照 block_size 分块并重新排列到 batch 维度,同时根据 paddings 执行必要的零填充。 + + 输入: + - **input** - 输入数据地址。 + - **block_size** - 分块因子,格式为 ``[block_h, block_w]``。 + - **paddings** - 填充参数,格式为 ``[top, bottom, left, right]``。 + - **input_shape** - 输入形状,格式为 ``[batch, height, width, channel]``。 + - **data_size** - 单个元素字节数。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 输出数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。 + - MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。 + + +**共享存储版本:** + +.. c:function:: void i8_spacetobatchnd_s(int8_t *input, int8_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void i16_spacetobatchnd_s(int16_t *input, int16_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void i32_spacetobatchnd_s(int32_t *input, int32_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void hp_spacetobatchnd_s(half *input, half *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void fp_spacetobatchnd_s(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void dp_spacetobatchnd_s(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void c64_spacetobatchnd_s(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) +.. c:function:: void c128_spacetobatchnd_s(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size, int core_mask) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 多核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0xA0000000; // 输入在 DDR 地址 0xA0000000 + float *output = (float *)0xB0000000; // 输出在 DDR 地址 0xB0000000 + int block_size[2] = {2, 2}; + int paddings[4] = {0, 0, 0, 0}; + int input_shape[4] = {1, 100, 10, 10}; + int core_mask = 0xff; + fp_spacetobatchnd_s(input, output, block_size, paddings, input_shape, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_spacetobatchnd_p(int8_t *input, int8_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void i16_spacetobatchnd_p(int16_t *input, int16_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void i32_spacetobatchnd_p(int32_t *input, int32_t *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void hp_spacetobatchnd_p(half *input, half *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void fp_spacetobatchnd_p(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void dp_spacetobatchnd_p(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void c64_spacetobatchnd_p(float *input, float *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) +.. c:function:: void c128_spacetobatchnd_p(double *input, double *output, const int *block_size, const int *paddings, const int *input_shape, int data_size) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 单核示例 + #include + + int main(int argc, char *argv[]) { + float *input = (float *)0x10000000; // 单核版本:输入放在 L2 地址 0x10000000 + float *output = (float *)0x10040000; // 单核版本:输出放在 L2 地址 0x10040000 + int block_size[2] = {2, 2}; + int paddings[4] = {0, 0, 0, 0}; + int input_shape[4] = {1, 100, 10, 10}; + fp_spacetobatchnd_p(input, output, block_size, paddings, input_shape, sizeof(float)); + return 0; + } + + diff --git a/master/html/_sources/functionlib/dsplib/spacetodepth.rst.txt b/master/html/_sources/functionlib/dsplib/spacetodepth.rst.txt new file mode 100644 index 0000000..ea98858 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/spacetodepth.rst.txt @@ -0,0 +1,91 @@ +SpaceToDepth +================= +将输入张量的空间维按块(block)重组到深度通道(channel)中。 + + 假设输入形状为: \[N, H, W, C\],块大小为 B,则输出形状为: + + .. math:: + + \text{out\_shape} = [N, H/ B, W / B, C * B * B] + + 对应元素映射为: + + .. math:: + + output[n, h_{out}, w_{out}, c * B * B + (l * B + m)] = input[n, h_{out} * B + l, w_{out} * B + m, c] + + 输入: + - **input** - 输入数据地址,按 NHWC 存储。 + - **in_shape** - 指向长度为4的数组,表示输入维度 \[N, H, W, C\]。 + - **block** - block 大小。 + - **data_size** - 每个元素的字节大小。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 输出数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 fp32, fp64, cplx64, cplx128, int16, int8, int32 + - MT7004 支持 fp32, fp16, cplx64, int16, int32 + +**共享存储版本:** + +.. c:function:: void i8_spacetodepth_s(int8_t* input, int8_t* output, const int* in_shape, int block, int data_size, int core_mask) +.. c:function:: void i16_spacetodepth_s(int16_t* input, int16_t* output, const int* in_shape, int block, int data_size, int core_mask) +.. c:function:: void i32_spacetodepth_s(int32_t* input, int32_t* output, const int* in_shape, int block, int data_size, int core_mask) +.. c:function:: void hp_spacetodepth_s(half* input, half* output, const int* in_shape, int block, int data_size, int core_mask) +.. c:function:: void fp_spacetodepth_s(float* input, float* output, const int* in_shape, int block, int data_size, int core_mask) +.. c:function:: void dp_spacetodepth_s(double* input, double* output, const int* in_shape, int block, int data_size, int core_mask) +.. c:function:: void c64_spacetodepth_s(float* input, float* output, const int* in_shape, int block, int data_size, int core_mask) +.. c:function:: void c128_spacetodepth_s(double* input, double* output, const int* in_shape, int block, int data_size, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // 多核版:输入在 DDR 区域 + float *output = (float *)0xB0000000; // 多核版:输出在 DDR 区域 + int in_shape[4] = {1, 8, 8, 1}; // N,H,W,C + int block = 2; + int core_mask = 0xff; + fp_spacetodepth_s(input, output, in_shape, block, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_spacetodepth_p(int8_t* input, int8_t* output, const int* in_shape, int block, int data_size) +.. c:function:: void i16_spacetodepth_p(int16_t* input, int16_t* output, const int* in_shape, int block, int data_size) +.. c:function:: void i32_spacetodepth_p(int32_t* input, int32_t* output, const int* in_shape, int block, int data_size) +.. c:function:: void hp_spacetodepth_p(half* input, half* output, const int* in_shape, int block, int data_size) +.. c:function:: void fp_spacetodepth_p(float* input, float* output, const int* in_shape, int block, int data_size) +.. c:function:: void dp_spacetodepth_p(double* input, double* output, const int* in_shape, int block, int data_size) +.. c:function:: void c64_spacetodepth_p(float* input, float* output, const int* in_shape, int block, int data_size) +.. c:function:: void c128_spacetodepth_p(double* input, double* output, const int* in_shape, int block, int data_size) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // 单核版:输入在 L2 + float *output = (float *)0x10010000; // 单核版:输出在 L2 + int in_shape[4] = {1, 8, 8, 1}; + int block = 2; + fp_spacetodepth_p(input, output, in_shape, block, sizeof(float)); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/squeeze.rst.txt b/master/html/_sources/functionlib/dsplib/squeeze.rst.txt new file mode 100644 index 0000000..b8c064e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/squeeze.rst.txt @@ -0,0 +1,42 @@ +Squeeze +================= + +返回删除指定axis中大小为1的维度后的Tensor。由于该算子仅改变张量形状,因此其DSP算子的作用是将数据从输入张量完整拷贝到输出张量。 + +输入: + - **src** - 输入地址 + - **total_copy_size** - 计算得到的总共需拷贝的数据量,单位为字节。 + - **core_mask** - 核掩码。 + +输出: + - **dst** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享/私有存储版本:** + +.. c:function:: void anytype_squeeze_anycore(void* src, void* dst, int total_copy_size, int core_mask) + +各种数据类型、私有及共享空间版本均使用该函数。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 8 + + void main(){ + int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可 + int core_num = GetCoreNum(core_mask); + float* src = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可 + float* dst = (float*)0x98000000; + int shape[3] = {1, 10, 10}; + int total_copy_size = shape[0] * shape[1] * shape[2] * sizeof(float); + anytype_squeeze_anycore(src, dst, total_copy_size, core_mask); + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/unsqueeze.rst.txt b/master/html/_sources/functionlib/dsplib/unsqueeze.rst.txt new file mode 100644 index 0000000..7786d94 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/unsqueeze.rst.txt @@ -0,0 +1,42 @@ +UnSqueeze +================= + +对输入张量在给定的轴上添加额外维度。由于该算子仅改变张量形状,因此其DSP算子的作用是将数据从输入张量完整拷贝到输出张量。 + +输入: + - **src** - 输入地址 + - **total_copy_size** - 计算得到的总共需拷贝的数据量,单位为字节。 + - **core_mask** - 核掩码。 + +输出: + - **dst** - 输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享/私有存储版本:** + +.. c:function:: void anytype_unsqueeze_anycore(void* src, void* dst, int total_copy_size, int core_mask) + +各种数据类型、私有及共享空间版本均使用该函数。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 8 + + void main(){ + int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可 + int core_num = GetCoreNum(core_mask); + float* src = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可 + float* dst = (float*)0x98000000; + int shape[3] = {1, 10, 10}; + int total_copy_size = shape[0] * shape[1] * shape[2] * sizeof(float); + anytype_unsqueeze_anycore(src, dst, total_copy_size, core_mask); + } \ No newline at end of file diff --git a/master/html/functionlib/dsplib/activation.html b/master/html/functionlib/dsplib/activation.html new file mode 100644 index 0000000..f72f5e0 --- /dev/null +++ b/master/html/functionlib/dsplib/activation.html @@ -0,0 +1,887 @@ + + + + + + + + + Activation — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

Activation

+

对输入的数组中每一个元素执行激活函数计算,激活函数可选,具体函数见以下说明。

+
    +
  • Relu - 标准Relu函数。

    +
    +
    +\[output_i = \max(0, input_i)\]
    +
    +
  • +
  • Relu6 - 在标准Relu函数的基础上进行输出上限限制。

    +
    +
    +\[output_i = \min(\max(0, input_i),6)\]
    +
    +
  • +
  • Clip - 将输入裁剪到区间 [min_val, max_val]

    +
    +
    +\[output_i = \min(\max(input_i, \text{min_val}), \text{max_val})\]
    +
    +
  • +
  • LRelu - 带泄露的线性整流单元(Leaky Rectified Linear Unit),它在输入为正时保持线性,在输入为负时也保留一个很小的斜率,以避免标准 ReLU 中的“死亡神经元”问题。

    +
    +
    +\[\begin{split}output_i = +\begin{cases} +input_i, & input_i \ge 0 \\ +\alpha \cdot input_i, & input_i < 0 +\end{cases}\end{split}\]
    +
    +
  • +
  • Sigmoid - 常用的平滑非线性激活函数(又称逻辑函数),可以将任意实数映射到区间 \((0, 1)\),常用于二分类问题的输出层,表示概率意义的结果。

    +
    +
    +\[output_i = \frac{1}{1 + e^{-input_i}}\]
    +
    +
  • +
  • Tanh - 双曲正切激活函数(Hyperbolic Tangent),其输出范围为 \((-1, 1)\)

    +
    +
    +\[output_i = \tanh(input_i) = \frac{e^{input_i} - e^{-input_i}}{e^{input_i} + e^{-input_i}}\]
    +
    +
  • +
  • HSigmoid - 硬 Sigmoid 激活函数(Hard Sigmoid),是 Sigmoid 函数的近似形式,计算简单、效率更高。

    +
    +
    +\[output_i = \text{clip}\left(\frac{input_i + 3}{6}, 0, 1\right)\]
    +

    其中 clip(a, 0, 1) 表示将 a 限制在区间 \([0, 1]\) 内。

    +
    +
  • +
  • Swish - 自门控(Self-Gated)激活函数,由 Google 提出,结合了 Sigmoid 与线性特性,具有平滑且非单调的特点。

    +
    +
    +\[output_i = input_i \cdot \sigma(input_i) = \frac{input_i}{1 + e^{-input_i}}\]
    +

    其中 \(\sigma(x)\) 为标准 Sigmoid 函数。Swish 在深层网络中通常表现优于 ReLU

    +
    +
  • +
  • HSwish - 硬 Swish 激活函数(Hard Swish),是 Swish 函数的近似形式,计算简单且在移动端模型(如 MobileNetV3)中被广泛采用。

    +
    +
    +\[output_i = input_i \cdot \text{clip}\left(\frac{input_i + 3}{6}, 0, 1\right)\]
    +

    其中 clip(a, 0, 1) 表示将 a 限制在区间 \([0, 1]\) 内。

    +
    +
  • +
  • HardTanh - 硬双曲正切激活函数(Hard Tanh),是 Tanh 函数的分段线性近似形式,计算简单、梯度稳定,常用于量化或轻量网络中。

    +
    +
    +\[output_i = \text{clip}(input_i, min\_val, max\_val)\]
    +

    其中 clip(x, min_val, max_val) 表示当 \(x < min\_val\) 时输出 min_val,当 \(x > max\_val\) 时输出 max_val,否则输出 \(x\) 本身。

    +
    +
  • +
  • Gelu - 高斯误差线性单元(Gaussian Error Linear Unit),是一种平滑的非线性激活函数,结合了 ReLU 与概率特性。 该函数支持精确计算及非近似计算模式,近似算法由 Hendrycks & Gimpel (2016) 提出,用以替代精确形式 \(output_i=x\Phi(x)\) ,计算速度更快且精度损失极小。

    +
    +
    +\[\begin{split}\begin{aligned} +output_i = +\begin{cases} +0.5\,input_i \Bigl[ 1 + \tanh\!\Bigl( + \sqrt{\frac{2}{\pi}}\,(input_i + 0.044715\,input_i^3) +\Bigr) \Bigr], & flag = true, \\[6pt] +input_i \,\Phi(input_i) += \tfrac{1}{2}x \Bigl[ + 1 + \mathrm{erf}\!\Bigl(\tfrac{input_i}{\sqrt{2}}\Bigr) +\Bigr], & flag = false. +\end{cases} +\end{aligned}\end{split}\]
    +

    其中 \(\Phi(x)\) 为标准正态分布的累积分布函数。

    +
    +
  • +
  • Softplus - ReLU 的平滑近似形式,能在零点处保持可导性。

    +
    +
    +\[\begin{split}output_i = + \begin{cases} + input_i, & input_i \gt 88.0 \\ + \ln(1 + e^{input_i}), & \text{otherwise} + \end{cases}\end{split}\]
    +
    +
  • +
  • Elu - 在输入为正时保持线性,在输入为负时呈指数衰减,可缓解 ReLU 的“死亡神经元”问题。

    +
    +
    +\[\begin{split}output_i = + \begin{cases} + input_i, & input_i \ge 0 \\ + \alpha (e^{input_i} - 1), & input_i < 0 + \end{cases}\end{split}\]
    +
    +

    其中 \(\alpha\) 为超参数,通常取 \(\alpha = 1.0\)

    +
  • +
  • Celu - 连续指数线性单元(Continuously Differentiable ELU),是 ELU 的改进版本,保证在零点处连续可导。

    +
    +
    +\[\begin{split}output_i = +\begin{cases} +input_i, & input_i \ge 0 \\ +\alpha (e^{\frac{input_i}{alpha}} - 1), & input_i < 0 +\end{cases}\end{split}\]
    +

    其中 \(\alpha\) 为可调超参数,用于控制负区间的平滑程度。

    +
    +
  • +
  • HardShrink - 硬收缩激活函数(Hard Shrinkage),用于稀疏化输出。

    +
    +
    +\[\begin{split}output_i = +\begin{cases} + input_i, & \text{if } |input_i| > \lambda \\ + 0, & \text{otherwise} +\end{cases}\end{split}\]
    +
    +

    其中 \(\lambda\) 为阈值常数。

    +
  • +
  • SoftShrink - 软收缩激活函数(Soft Shrinkage),与 HardShrink 类似,但收缩过程更加平滑。

    +
    +
    +\[\begin{split}output_i = +\begin{cases} +input_i - \lambda, & \text{if } input_i > \lambda \\ +input_i + \lambda, & \text{if } input_i < -\lambda \\ +0, & \text{otherwise} +\end{cases}\end{split}\]
    +
    +
  • +
  • SoftsignOpt - 优化的软符号函数(Optimized Softsign),是一种平滑的压缩函数,用于将输入映射到有限区间。

    +
    +
    +\[output_i = \frac{input_i}{1 + |input_i|}\]
    +
    +
  • +
+
+
输入:
    +
  • Input0 - 输入数据地址。

  • +
  • length - 数组长度。

  • +
  • args(部分激活函数) - 激活函数计算参数(仅适用于部分函数)。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • output - 计算结果地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持int8, fp32

  • +
  • MT7004 支持fp16, fp32

  • +
+
+

共享存储版本:

+
+
+void i8_relu_s(int8_t *Input0, int8_t *output, int length, int core_mask)
+
+ +
+
+void fp_relu_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_relu_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_relu6_s(int8_t *Input0, int8_t *output, int length, int core_mask)
+
+ +
+
+void fp_relu6_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_relu6_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_clip_s(int8_t *Input0, int8_t *output, int length, int8_t min_val, int8_t max_val, int core_mask)
+
+ +
+
+void fp_clip_s(float *Input0, float *output, int length, float min_val, float max_val, int core_mask)
+
+ +
+
+void hp_clip_s(half *Input0, half *output, int length, half min_val, half max_val, int core_mask)
+
+ +
+
+void i8_lrelu_s(int8_t *Input0, int8_t *output, int length, float alpha, int core_mask)
+
+ +
+
+void fp_lrelu_s(float *Input0, float *output, int length, float alpha, int core_mask)
+
+ +
+
+void hp_lrelu_s(half *Input0, half *output, int length, half alpha, int core_mask)
+
+ +
+
+void i8_sigmoid_s(int8_t *Input0, float *output, int length, int core_mask)
+
+ +
+
+void fp_sigmoid_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_sigmoid_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_tanh_s(int8_t *Input0, float *output, int length, int core_mask)
+
+ +
+
+void fp_tanh_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_tanh_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_hsigmoid_s(int8_t *Input0, float *output, int length, int core_mask)
+
+ +
+
+void fp_hsigmoid_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_hsigmoid_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_swish_s(int8_t *Input0, float *output, int length, int core_mask)
+
+ +
+
+void fp_swish_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_swish_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_hswish_s(int8_t *Input0, float *output, int length, int core_mask)
+
+ +
+
+void fp_hswish_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_hswish_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_hardtanh_s(int8_t *Input0, int8_t *output, int length, int8_t min_val, int8_t max_val, int core_mask)
+
+ +
+
+void fp_hardtanh_s(float *Input0, float *output, int length, float min_val, float max_val, int core_mask)
+
+ +
+
+void hp_hardtanh_s(half *Input0, half *output, int length, half min_val, half max_val, int core_mask)
+
+ +
+
+void i8_gelu_s(int8_t *Input0, float *output, int length, int approximate, int core_mask)
+
+ +
+
+void fp_gelu_s(float *Input0, float *output, int length, int approximate, int core_mask)
+
+ +
+
+void hp_gelu_s(half *Input0, half *output, int length, int approximate, int core_mask)
+
+ +
+
+void i8_softplus_s(int8_t *Input0, float *output, int length, int core_mask)
+
+ +
+
+void fp_softplus_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_softplus_s(half *Input0, half *output, int length, int core_mask)
+
+ +
+
+void i8_elu_s(int8_t *Input0, float *output, int length, float alpha, int core_mask)
+
+ +
+
+void fp_elu_s(float *Input0, float *output, int length, float alpha, int core_mask)
+
+ +
+
+void hp_elu_s(half *Input0, half *output, int length, half alpha, int core_mask)
+
+ +
+
+void i8_celu_s(int8_t *Input0, float *output, int length, float alpha, int core_mask)
+
+ +
+
+void fp_celu_s(float *Input0, float *output, int length, float alpha, int core_mask)
+
+ +
+
+void hp_celu_s(half *Input0, half *output, int length, half alpha, int core_mask)
+
+ +
+
+void i8_hardshrink_s(int8_t *Input0, int8_t *output, int length, int8_t lambd, int core_mask)
+
+ +
+
+void fp_hardshrink_s(float *Input0, float *output, int length, float lambd, int core_mask)
+
+ +
+
+void hp_hardshrink_s(half *Input0, half *output, int length, half lambd, int core_mask)
+
+ +
+
+void i8_softshrink_s(int8_t *Input0, int8_t *output, int length, int8_t lambd, int core_mask)
+
+ +
+
+void fp_softshrink_s(float *Input0, float *output, int length, float lambd, int core_mask)
+
+ +
+
+void hp_softshrink_s(half *Input0, half *output, int length, half lambd, int core_mask)
+
+ +
+
+void i8_softsignopt_s(int8_t *Input0, float *output, int length, int core_mask)
+
+ +
+
+void fp_softsignopt_s(float *Input0, float *output, int length, int core_mask)
+
+ +
+
+void hp_softsignopt_s(half *Input0, half *output, int length, int core_mask)
+
+ +

C调用示例:

+
+
 1//FT78NE示例
+ 2#include <stdio.h>
+ 3#include <activation.h>
+ 4
+ 5int main(int argc, char* argv[]) {
+ 6    float *input0 = (float *)0xA0000000;   //input在DDR空间
+ 7    float *output = (float *)0xC0000000;
+ 8    int length = 1000;
+ 9    int core_mask = 0xff;
+10    fp_tanh_s(input0, output, length, core_mask);
+11    return 0;
+12}
+
+
+
+

私有存储版本:

+
+
+void i8_relu_p(int8_t *Input0, int8_t *output, int length)
+
+ +
+
+void fp_relu_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_relu_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_relu6_p(int8_t *Input0, int8_t *output, int length)
+
+ +
+
+void fp_relu6_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_relu6_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_clip_p(int8_t *Input0, int8_t *output, int length, int8_t min_val, int8_t max_val)
+
+ +
+
+void fp_clip_p(float *Input0, float *output, int length, float min_val, float max_val)
+
+ +
+
+void hp_clip_p(half *Input0, half *output, int length, half min_val, half max_val)
+
+ +
+
+void i8_lrelu_p(int8_t *Input0, int8_t *output, int length, float alpha)
+
+ +
+
+void fp_lrelu_p(float *Input0, float *output, int length, float alpha)
+
+ +
+
+void hp_lrelu_p(half *Input0, half *output, int length, half alpha)
+
+ +
+
+void i8_sigmoid_p(int8_t *Input0, float *output, int length)
+
+ +
+
+void fp_sigmoid_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_sigmoid_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_tanh_p(int8_t *Input0, float *output, int length)
+
+ +
+
+void fp_tanh_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_tanh_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_hsigmoid_p(int8_t *Input0, float *output, int length)
+
+ +
+
+void fp_hsigmoid_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_hsigmoid_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_swish_p(int8_t *Input0, float *output, int length)
+
+ +
+
+void fp_swish_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_swish_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_hswish_p(int8_t *Input0, float *output, int length)
+
+ +
+
+void fp_hswish_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_hswish_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_hardtanh_p(int8_t *Input0, int8_t *output, int length, int8_t min_val, int8_t max_val)
+
+ +
+
+void fp_hardtanh_p(float *Input0, float *output, int length, float min_val, float max_val)
+
+ +
+
+void hp_hardtanh_p(half *Input0, half *output, int length, half min_val, half max_val)
+
+ +
+
+void i8_gelu_p(int8_t *Input0, float *output, int length, int approximate)
+
+ +
+
+void fp_gelu_p(float *Input0, float *output, int length, int approximate)
+
+ +
+
+void hp_gelu_p(half *Input0, half *output, int length, int approximate)
+
+ +
+
+void i8_softplus_p(int8_t *Input0, float *output, int length)
+
+ +
+
+void fp_softplus_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_softplus_p(half *Input0, half *output, int length)
+
+ +
+
+void i8_elu_p(int8_t *Input0, float *output, int length, float alpha)
+
+ +
+
+void fp_elu_p(float *Input0, float *output, int length, float alpha)
+
+ +
+
+void hp_elu_p(half *Input0, half *output, int length, half alpha)
+
+ +
+
+void i8_celu_p(int8_t *Input0, float *output, int length, float alpha)
+
+ +
+
+void fp_celu_p(float *Input0, float *output, int length, float alpha)
+
+ +
+
+void hp_celu_p(half *Input0, half *output, int length, half alpha)
+
+ +
+
+void i8_hardshrink_p(int8_t *Input0, int8_t *output, int length, int8_t lambd)
+
+ +
+
+void fp_hardshrink_p(float *Input0, float *output, int length, float lambd)
+
+ +
+
+void hp_hardshrink_p(half *Input0, half *output, int length, half lambd)
+
+ +
+
+void i8_softshrink_p(int8_t *Input0, int8_t *output, int length, int8_t lambd)
+
+ +
+
+void fp_softshrink_p(float *Input0, float *output, int length, float lambd)
+
+ +
+
+void hp_softshrink_p(half *Input0, half *output, int length, half lambd)
+
+ +
+
+void i8_softsignopt_p(int8_t *Input0, float *output, int length)
+
+ +
+
+void fp_softsignopt_p(float *Input0, float *output, int length)
+
+ +
+
+void hp_softsignopt_p(half *Input0, half *output, int length)
+
+ +

C调用示例:

+
+
 1//FT78NE示例
+ 2#include <stdio.h>
+ 3#include <activation.h>
+ 4
+ 5int main(int argc, char* argv[]) {
+ 6    float *input0 = (float *)0x10000000;   //input在DDR空间
+ 7    float *output = (float *)0x10004000;
+ 8    int length = 1000;
+ 9    fp_tanh_p(input0, output, length);
+10    return 0;
+11}
+
+
+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/adamweightdecay.html b/master/html/functionlib/dsplib/adamweightdecay.html new file mode 100644 index 0000000..2bbd947 --- /dev/null +++ b/master/html/functionlib/dsplib/adamweightdecay.html @@ -0,0 +1,281 @@ + + + + + + + + + AdamWeightDecay — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

AdamWeightDecay

+

对权重张量执行 Adam Weight Decay 优化更新。

+
+
+\[\begin{split}\begin{aligned} +m_t &= \beta_1 \cdot m_{t-1} + (1 - \beta_1) \cdot g_t \\ +v_t &= \beta_2 \cdot v_{t-1} + (1 - \beta_2) \cdot g_t^2 \\ +\hat{m}_t &= \frac{m_t}{\sqrt{v_t} + \epsilon} \\ +var_t &= var_{t-1} - lr \cdot (\hat{m}_t + decay \cdot var_{t-1}) +\end{aligned}\end{split}\]
+
+
输入:
    +
  • var - 待更新权重张量首地址。

  • +
  • m - 一阶动量张量首地址。

  • +
  • v - 二阶动量张量首地址。

  • +
  • gradient - 梯度张量首地址。

  • +
  • lr - 学习率。

  • +
  • beta1 - 一阶动量衰减系数。

  • +
  • beta2 - 二阶动量衰减系数。

  • +
  • epsilon - 数值稳定项。

  • +
  • decay - 权重衰减系数。

  • +
  • start - 参与计算的起始索引(闭区间)。

  • +
  • end - 参与计算的结束索引(开区间)。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • var - 原地写回更新后的权重张量。

  • +
  • m - 原地写回更新后的一阶动量张量。

  • +
  • v - 原地写回更新后的二阶动量张量。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持 fp32 数据类型。

  • +
  • MT7004 支持 fp16、fp32 数据类型。

  • +
+
+
+

共享存储版本:

+
+
+void hp_adamweightdecay_s(half *var, half *m, half *v, const half *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int start, int end, int core_mask)
+
+ +
+
+void fp_adamweightdecay_s(float *var, float *m, float *v, const float *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int start, int end, int core_mask)
+

C调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(void) {
+ 5        float *var = (float *)0xA0000000;        // DDR 存储
+ 6        float *m = (float *)0xB0000000;
+ 7        float *v = (float *)0xC0000000;
+ 8        float *gradient = (float *)0xD0000000;
+ 9        int start = 0;
+10        int end = 4096;
+11        int core_mask = 0xff;
+12        float lr = 1e-3f;
+13        float beta1 = 0.9f;
+14        float beta2 = 0.999f;
+15        float epsilon = 1e-8f;
+16        float decay = 1e-2f;
+17        fp_adamweightdecay_s(var, m, v, gradient, lr,
+18                                                 beta1, beta2, epsilon, decay,
+19                                                 start, end, core_mask);
+20        return 0;
+21}
+
+
+
+ +

私有存储版本:

+
+
+void hp_adamweightdecay_p(half *var, half *m, half *v, const half *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int length)
+
+ +
+
+void fp_adamweightdecay_p(float *var, float *m, float *v, const float *gradient, float lr, float beta1, float beta2, float epsilon, float decay, int length)
+

C调用示例:

+
 1// MT7004 单核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(void) {
+ 5        half *var = (half *)0x10000000;        // L2 存储
+ 6        half *m = (half *)0x10002000;
+ 7        half *v = (half *)0x10004000;
+ 8        half *gradient = (half *)0x10006000;
+ 9        int length = 2048;
+10        float lr = 5e-4f;
+11        float beta1 = 0.9f;
+12        float beta2 = 0.999f;
+13        float epsilon = 1e-6f;
+14        float decay = 5e-3f;
+15        hp_adamweightdecay_p(var, m, v, gradient, lr,
+16                                                 beta1, beta2, epsilon, decay,
+17                                                 length);
+18        return 0;
+19}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/adder.html b/master/html/functionlib/dsplib/adder.html new file mode 100644 index 0000000..54116a2 --- /dev/null +++ b/master/html/functionlib/dsplib/adder.html @@ -0,0 +1,352 @@ + + + + + + + + + Adder — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

Adder

+

Adder是一种卷积替代算子,它使用L1距离度量(绝对差的和)代替传统卷积中的点积操作。与标准卷积不同,Adder通过计算特征与卷积核之间的绝对差的负和来进行特征提取。假定输入X,filter表示为F,它按以下公式计算:

+
+\[Y(m,n,t) = - \sum_{i=0}^{d} \sum_{j=0}^{d} \sum_{k=0}^{C_{in}} |X(m+i, n+j, k) - F(i,j,k,t)|\]
+
+
输入:
    +
  • input_x - 输入数据的地址

  • +
  • input_w - 输入卷积核权重的地址

  • +
  • bias - 输入偏置的地址

  • +
  • param - 算子计算所需参数的结构体。其各成员见下述。

  • +
  • core_mask - 核掩码。

  • +
+
+
+

AdderParameter定义:

+
 1typedef struct AdderParameter {
+ 2    void* workspace_; // 用于存放中间计算结果
+ 3    int output_batch_; // 输出数据总批次
+ 4    int input_batch_; // 输入数据总批次
+ 5    int input_h_; // 输入数据h维度大小
+ 6    int input_w_; // 输入数据w维度大小
+ 7    int output_h_; // 输出数据h维度大小
+ 8    int output_w_; // 输出数据w维度大小
+ 9    int input_channel_; // 输入数据通道数
+10    int output_channel_; // 输出数据通道数
+11    int kernel_h_; // 卷积核h维度大小
+12    int kernel_w_; // 卷积核w维度大小
+13    int group_; // 组数
+14    int pad_l_; // 左填充大小
+15    int pad_u_; // 上填充大小
+16    int dilation_h_; // 卷积核h维度膨胀尺寸大小
+17    int dilation_w_; // 卷积核w维度膨胀尺寸大小
+18    int stride_h_; // 卷积核h维度步长
+19    int stride_w_; // 卷积核w维度步长
+20    int buffer_size_; // 为分块计算所分配的缓存大小
+21} AdderParameter;
+
+
+
+
输出:
    +
  • out_y - 输出地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持int8, fp32

  • +
  • MT7004 支持fp16, fp32

  • +
+
+

共享存储版本:

+
+
+void i8_adder_s(int8_t *input_x, int8_t *input_w, int8_t *out_y, int *bias, AdderParameter *param, int core_mask)
+
+ +
+
+void hp_adder_s(half *input_x, half *input_w, half *out_y, half *bias, AdderParameter *param, int core_mask)
+
+ +
+
+void fp_adder_s(float *input_x, float *input_w, float *out_y, float *bias, AdderParameter *param, int core_mask)
+
+ +

C调用示例:

+
 1void TestAdderSMCFp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) {
+ 2    int core_id = get_core_id();
+ 3    int logic_core_id = GetLogicCoreId(core_mask, core_id);
+ 4    int core_num = GetCoreNum(core_mask);
+ 5    float* input_data = (float*)0x88000000;
+ 6    float* weight = (float*)0x89000000;
+ 7    float* output_data = (float*)0x90000000;
+ 8    float* bias_data = (float*)0x91000000;
+ 9    AdderParameter* param = (AdderParameter*)0x92000000;
+10    if (logic_core_id == 0) {
+11        memcpy(bias_data, bias, sizeof(float) * output_shape[3]);
+12        param->dilation_h_ = dilation[0];
+13        param->dilation_w_ = dilation[1];
+14        param->group_ = groups;
+15        param->input_batch_ = input_shape[0];
+16        param->input_h_ = input_shape[1];
+17        param->input_w_ = input_shape[2];
+18        param->input_channel_ = input_shape[3];
+19        param->kernel_h_ = weight_shape[1];
+20        param->kernel_w_ = weight_shape[2];
+21        param->output_batch_ = output_shape[0];
+22        param->output_h_ = output_shape[1];
+23        param->output_w_ = output_shape[2];
+24        param->output_channel_ = output_shape[3];
+25        param->stride_h_ = stride[0];
+26        param->stride_w_ = stride[0];
+27        param->pad_u_ = padding[0];
+28        param->pad_l_ = padding[2];
+29        param->workspace_ = (float*)0x10000000; // workspace空间需分配在AM内,计算过程中会将数据搬运到workspace空间内进行计算
+30    }
+31    sys_bar(0, core_num); // 初始化参数完成后进行同步
+32    fp_adder_s(input_data, weight, output_data, bias_data, param, core_mask);
+33}
+34
+35void main(){
+36    int in_channel = 4;
+37    int out_channel = 4;
+38    int groups = 4;
+39    int input_shape[4] = {1, 30, 30, in_channel}; // NHWC
+40    int weight_shape[4] = {out_channel, 3, 3, in_channel / groups};
+41    int output_shape[4] = {1, 10, 10, out_channel}; // NHWC
+42    int stride[2] = {2, 2};
+43    int padding[4] = {1, 1, 1, 1};
+44    int dilation[2]= {2, 2};
+45    float bias[4] = {0, 0, 0, 0};
+46    int core_mask = 0b1111;
+47    TestAdderSMCFp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask);
+48}
+
+
+

私有存储版本:

+
+
+void i8_adder_p(int8_t *input_x, int8_t *input_w, int8_t *out_y, int *bias, ConvParameter *conv_param, ConvQuantParameter quant_param, int core_mask)
+
+ +
+
+void hp_adder_p(half *input_x, half *input_w, half *out_y, half *bias, ConvParameter *conv_param, int core_mask)
+
+ +
+
+void fp_adder_p(float *input_x, float *input_w, float *out_y, float *bias, ConvParameter *conv_param, int core_mask)
+
+ +

C调用示例:

+
 1void TestAdderL2Fp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) {
+ 2    float* input_data = (float*)0x10010000; // 私有存储版本地址设置在AM内
+ 3    float* weight = (float*)0x10020000;
+ 4    float* output_data = (float*)0x10030000;
+ 5    float* bias_data = (float*)0x10040000;
+ 6    AdderParameter* param = (AdderParameter*)0x10060000;
+ 7    memcpy(bias_data, bias, sizeof(float) * output_shape[3]);
+ 8    param->dilation_h_ = dilation[0];
+ 9    param->dilation_w_ = dilation[1];
+10    param->group_ = groups;
+11    param->input_batch_ = input_shape[0];
+12    param->input_h_ = input_shape[1];
+13    param->input_w_ = input_shape[2];
+14    param->input_channel_ = input_shape[3];
+15    param->kernel_h_ = weight_shape[1];
+16    param->kernel_w_ = weight_shape[2];
+17    param->output_batch_ = output_shape[0];
+18    param->output_h_ = output_shape[1];
+19    param->output_w_ = output_shape[2];
+20    param->output_channel_ = output_shape[3];
+21    param->stride_h_ = stride[0];
+22    param->stride_w_ = stride[0];
+23    param->pad_u_ = padding[0];
+24    param->pad_l_ = padding[2];
+25    param->workspace_ = (float*)0x10070000;
+26    param->buffer_size_ = 2048; // 私有存储版本中,必须设置该参数,用于确定分块计算的大小
+27    fp_adder_p(input_data, weight, output_data, bias_data, param, core_mask);
+28}
+29
+30void main(){
+31    int in_channel = 4;
+32    int out_channel = 4;
+33    int groups = 4;
+34    int input_shape[4] = {1, 30, 30, in_channel}; // NHWC
+35    int weight_shape[4] = {out_channel, 3, 3, in_channel / groups};
+36    int output_shape[4] = {1, 10, 10, out_channel}; // NHWC
+37    int stride[2] = {2, 2};
+38    int padding[4] = {1, 1, 1, 1};
+39    int dilation[2]= {2, 2};
+40    float bias[4] = {0, 0, 0, 0};
+41    int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动
+42    TestAdderL2Fp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask);
+43}
+
+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/applymomentum.html b/master/html/functionlib/dsplib/applymomentum.html new file mode 100644 index 0000000..203a98f --- /dev/null +++ b/master/html/functionlib/dsplib/applymomentum.html @@ -0,0 +1,275 @@ + + + + + + + + + ApplyMomentum — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

ApplyMomentum

+

对权重张量执行 Momentum/改进动量优化更新。

+
+
+\[\begin{split}\begin{aligned} +accu_t &= moment \cdot accu_{t-1} + g_t \\ +update_t &= \begin{cases} + (accu_t \cdot moment + g_t), & \text{if nesterov = True} \\ + accu_t, & \text{otherwise} +\end{cases} \\ +weight_t &= weight_{t-1} - learning\_rate \cdot update_t +\end{aligned}\end{split}\]
+
+
输入:
    +
  • weight - 待更新权重张量首地址。

  • +
  • accumulate - 动量累积张量首地址。

  • +
  • gradient - 梯度张量首地址。

  • +
  • learning_rate - 学习率。

  • +
  • moment - 动量系数。

  • +
  • nesterov - 是否启用 Nesterov 动量。

  • +
  • start - 参与计算的起始索引(闭区间)。

  • +
  • end - 参与计算的结束索引(开区间)。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • weight - 原地写回更新后的权重张量。

  • +
  • accumulate - 原地写回更新后的动量张量。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持 fp32 数据类型。

  • +
  • MT7004 支持 fp16、fp32 数据类型。

  • +
+
+
+

共享存储版本:

+
+
+void hp_applymomentum_s(half *weight, half *accumulate, const half *gradient, float learning_rate, float moment, bool nesterov, int start, int end, int core_mask)
+
+ +
+
+void fp_applymomentum_s(float *weight, float *accumulate, const float *gradient, float learning_rate, float moment, bool nesterov, int start, int end, int core_mask)
+

C调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3#include <stdbool.h>
+ 4
+ 5int main(void) {
+ 6    float *weight = (float *)0xA0000000;      // DDR 存储
+ 7    float *accumulate = (float *)0xB0000000;
+ 8    float *gradient = (float *)0xC0000000;
+ 9    int start = 0;
+10    int end = 4096;
+11    int core_mask = 0xff;
+12    float learning_rate = 1e-2f;
+13    float moment = 0.99f;
+14    bool nesterov = false;
+15    fp_applymomentum_s(weight, accumulate, gradient,
+16                        learning_rate, moment, nesterov,
+17                        start, end, core_mask);
+18    return 0;
+19}
+
+
+
+ +

私有存储版本:

+
+
+void hp_applymomentum_p(half *weight, half *accumulate, const half *gradient, float learning_rate, float moment, bool nesterov, int length)
+
+ +
+
+void fp_applymomentum_p(float *weight, float *accumulate, const float *gradient, float learning_rate, float moment, bool nesterov, int length)
+

C调用示例:

+
 1// MT7004 单核示例
+ 2#include <stdio.h>
+ 3#include <stdbool.h>
+ 4
+ 5int main(void) {
+ 6    half *weight = (half *)0x10000000;       // L2 存储
+ 7    half *accumulate = (half *)0x10002000;
+ 8    half *gradient = (half *)0x10004000;
+ 9    int length = 2048;
+10    float learning_rate = 5e-3f;
+11    float moment = 0.9f;
+12    bool nesterov = true;
+13    hp_applymomentum_p(weight, accumulate, gradient,
+14                       learning_rate, moment, nesterov,
+15                       length);
+16    return 0;
+17}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/assert.html b/master/html/functionlib/dsplib/assert.html new file mode 100644 index 0000000..6f2d34a --- /dev/null +++ b/master/html/functionlib/dsplib/assert.html @@ -0,0 +1,213 @@ + + + + + + + + + Assert — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

Assert

+
+
+void assert(bool *Input, bool *output)
+
+ +

判断输入是否为 True。

+
+\[\begin{split}output_i = \begin{cases} + \text{True}, & \text{if } Input_i = \text{True} \\ + \text{False}, & \text{if } Input_i = \text{False} +\end{cases}\end{split}\]
+
+
输入:
    +
  • Input - 输入数据地址(布尔类型)。

  • +
+
+
输出:
    +
  • output - 计算结果地址(布尔类型)。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • 本算子只有一个版本

  • +
+
+

C调用示例:

+
 1// FT78NE/MT7004 示例(共享存储)
+ 2#include <stdio.h>
+ 3#include <stdbool.h>
+ 4
+ 5int main(int argc, char* argv[]) {
+ 6    bool *input  = (bool *)0xA0000000;
+ 7    bool *output = (bool *)0xC0000000;
+ 8    assert_s(input, output);
+ 9    return 0;
+10}
+
+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/attention.html b/master/html/functionlib/dsplib/attention.html new file mode 100644 index 0000000..c54bf9c --- /dev/null +++ b/master/html/functionlib/dsplib/attention.html @@ -0,0 +1,246 @@ + + + + + + + + + Attention — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

Attention

+

多头缩放点积注意力机制(Scaled Dot-Product Attention)

+
+\[\text{Attention}(Q, K, V) = \operatorname{softmax}\left(\frac{Q K^\top}{\sqrt{d_k}}\right) V\]
+
+
输入:
    +
  • Q - 查询矩阵地址(行优先),形状 \([B, H, L, D]\) 展平。

  • +
  • K - 键矩阵地址(行优先),形状 \([B, H, L, D]\) 展平。

  • +
  • V - 值矩阵地址(行优先),形状 \([B, H, L, D]\) 展平。

  • +
  • batch_size (B) - 批大小。

  • +
  • seq_len (L) - 序列长度。

  • +
  • head_num (H) - 多头数量。

  • +
  • head_dim (D) - 每头通道维数。

  • +
  • QK, softmax_out - 中间缓冲区地址,容量不小于 \(B\times H\times L\times L\)

  • +
  • core_mask(可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • output - 输出地址(行优先),形状 \([B, H, L, D]\) 展平。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • 当前实现基于 fp32;输入/中间/输出缓冲区不应重叠。

  • +
  • 内存布局为行优先(row-major)。

  • +
+
+

共享存储版本:

+
+
+void fp_attention_s(float *Q, float *K, float *V, float *output, int batch_size, int seq_len, int head_num, int head_dim, float *QK, float *softmax_out, int core_mask)
+

C调用示例:

+
 1#include <stdio.h>
+ 2
+ 3int main(int argc, char* argv[]) {
+ 4    int B = 2, L = 128, H = 8, D = 64;
+ 5    float *Q = (float *)0xA0000000;      // DDR
+ 6    float *K = (float *)0xA1000000;      // DDR
+ 7    float *V = (float *)0xA2000000;      // DDR
+ 8    float *O = (float *)0xA3000000;      // DDR
+ 9    float *QK = (float *)0xA4000000;     // DDR
+10    float *SM = (float *)0xA5000000;     // DDR
+11    int core_mask = 0xff;
+12    fp_attention_s(Q, K, V, O, B, L, H, D, QK, SM, core_mask);
+13    return 0;
+14}
+
+
+
+ +

私有存储版本:

+
+
+void fp_attention_p(float *Q, float *K, float *V, float *output, int batch_size, int seq_len, int head_num, int head_dim, float *QK, float *softmax_out)
+

C调用示例:

+
 1#include <stdio.h>
+ 2
+ 3int main(int argc, char* argv[]) {
+ 4    int B = 1, L = 64, H = 4, D = 32;
+ 5    float *Q = (float *)0x10000000;   // L2
+ 6    float *K = (float *)0x10040000;   // L2
+ 7    float *V = (float *)0x10080000;   // L2
+ 8    float *O = (float *)0x100C0000;   // L2
+ 9    float *QK = (float *)0x10100000;  // L2
+10    float *SM = (float *)0x10200000;  // L2
+11    fp_attention_p(Q, K, V, O, B, L, H, D, QK, SM);
+12    return 0;
+13}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/avgpoolinggrad.html b/master/html/functionlib/dsplib/avgpoolinggrad.html new file mode 100644 index 0000000..77ed85f --- /dev/null +++ b/master/html/functionlib/dsplib/avgpoolinggrad.html @@ -0,0 +1,293 @@ + + + + + + + + + AvgPoolingGrad — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

AvgPoolingGrad

+

根据输入梯度对平均池化前的特征图计算反向传播梯度。

+
+
+\[output_{b, x_h, x_w, c} += \frac{input_{b, y_h, y_w, c}}{\text{window}_h \cdot \text{window}_w}\]
+

其中 \((y_h, y_w)\)\((x_h, x_w)\) 之间满足窗口与步长的映射关系。

+
+
输入:
    +
  • input - 反向传播输入梯度张量首地址。

  • +
  • output - 反向传播输出梯度张量首地址。

  • +
  • batch - 批大小。

  • +
  • output_h - 池化输出高度。

  • +
  • output_w - 池化输出宽度。

  • +
  • channel - 通道数。

  • +
  • input_w - 池化输入宽度。

  • +
  • input_h - 池化输入高度。

  • +
  • stride_w - 水平步长。

  • +
  • stride_h - 垂直步长。

  • +
  • pad_l - 左侧填充大小。

  • +
  • pad_u - 上侧填充大小。

  • +
  • window_w - 池化窗口宽度。

  • +
  • window_h - 池化窗口高度。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • output - 原地累加平均池化反向传播梯度。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持 fp32 数据类型。

  • +
  • MT7004 支持 fp16、fp32 数据类型。

  • +
+
+
+

共享存储版本:

+
+
+void hp_avgpoolinggrad_s(const half *input, half *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int core_mask)
+
+ +
+
+void fp_avgpoolinggrad_s(const float *input, float *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int core_mask)
+

C调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(void) {
+ 5    const float *input = (const float *)0xA0000000;  // DDR 存储
+ 6    float *output = (float *)0xB0000000;
+ 7    int batch = 16;
+ 8    int output_h = 7;
+ 9    int output_w = 7;
+10    int channel = 64;
+11    int input_w = 14;
+12    int input_h = 14;
+13    int stride_w = 2;
+14    int stride_h = 2;
+15    int pad_l = 0;
+16    int pad_u = 0;
+17    int window_w = 2;
+18    int window_h = 2;
+19    int core_mask = 0xff;
+20    fp_avgpoolinggrad_s(input, output, batch, output_h, output_w,
+21                         channel, input_w, input_h,
+22                         stride_w, stride_h,
+23                         pad_l, pad_u,
+24                         window_w, window_h,
+25                         core_mask);
+26    return 0;
+27}
+
+
+
+ +

私有存储版本:

+
+
+void hp_avgpoolinggrad_p(const half *input, half *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int start_idx, int end_idx)
+
+ +
+
+void fp_avgpoolinggrad_p(const float *input, float *output, int batch, int output_h, int output_w, int channel, int input_w, int input_h, int stride_w, int stride_h, int pad_l, int pad_u, int window_w, int window_h, int start_idx, int end_idx)
+

C调用示例:

+
 1// MT7004 单核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(void) {
+ 5    const half *input = (const half *)0x10000000;    // L2 存储
+ 6    half *output = (half *)0x10004000;
+ 7    int batch = 4;
+ 8    int output_h = 4;
+ 9    int output_w = 4;
+10    int channel = 128;
+11    int input_w = 8;
+12    int input_h = 8;
+13    int stride_w = 2;
+14    int stride_h = 2;
+15    int pad_l = 0;
+16    int pad_u = 0;
+17    int window_w = 2;
+18    int window_h = 2;
+19    int start_idx = 0;
+20    int end_idx = batch * output_h * output_w * channel;
+21    hp_avgpoolinggrad_p(input, output, batch, output_h, output_w,
+22                        channel, input_w, input_h,
+23                        stride_w, stride_h,
+24                        pad_l, pad_u,
+25                        window_w, window_h,
+26                        start_idx, end_idx);
+27    return 0;
+28}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/batchtospace.html b/master/html/functionlib/dsplib/batchtospace.html new file mode 100644 index 0000000..1734c10 --- /dev/null +++ b/master/html/functionlib/dsplib/batchtospace.html @@ -0,0 +1,318 @@ + + + + + + + + + BatchToSpace — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

BatchToSpace

+

将输入张量在批维度上分块并重新分布到空间维度,同时按照 crops 对输出空间范围进行裁剪。

+
+
+\[\begin{split}\begin{aligned} +N_{\text{out}} &= \frac{N}{b_h \times b_w}, \\ +H_{\text{out}} &= b_h \times H - c_{\text{top}} - c_{\text{bottom}}, \\ +W_{\text{out}} &= b_w \times W - c_{\text{left}} - c_{\text{right}}, \\ + ext{output}[n, h, w, c] &= \text{input}[n', h', w', c] +\end{aligned}\end{split}\]
+

其中 \(N, H, W, C\) 分别表示输入的 batch、高度、宽度和通道数;\(b_h, b_w\)block_size\(c_{*}\) 来源于 crops\(n', h', w'\)BatchToSpace 映射关系确定。

+
+
输入:
    +
  • input - 输入数据地址。

  • +
  • input_shape - 输入形状,格式为 [batch, height, width, channel]

  • +
  • block_size - 分块因子,格式为 [block_h, block_w]

  • +
  • crops - 裁剪参数,格式为 [top, bottom, left, right]

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • output - 输出数据地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持 fp32、fp64、cplx64、cplx128、int16、int8、int32 数据类型。

  • +
  • MT7004 支持 fp32、fp16、cplx64、int16、int32 数据类型。

  • +
+
+
+

共享存储版本:

+
+
+void i8_batchtospace_s(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void i16_batchtospace_s(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void i32_batchtospace_s(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void hp_batchtospace_s(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void fp_batchtospace_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void dp_batchtospace_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void c64_batchtospace_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void c128_batchtospace_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+

C 调用示例:

+
 1// 多核(共享存储)示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5        float *input = (float *)0xA0000000;   // 输入在 DDR 空间
+ 6        float *output = (float *)0xB0000000;
+ 7        int input_shape[4] = {400, 2, 2, 3};
+ 8        int block_size[2] = {2, 2};
+ 9        int crops[4] = {0, 0, 0, 0};
+10        int core_mask = 0xff;
+11        fp_batchtospace_s(input, output, input_shape, block_size, crops, sizeof(float), core_mask);
+12        return 0;
+13}
+
+
+
+ +

私有存储版本:

+
+
+void i8_batchtospace_p(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void i16_batchtospace_p(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void i32_batchtospace_p(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void hp_batchtospace_p(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void fp_batchtospace_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void dp_batchtospace_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void c64_batchtospace_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void c128_batchtospace_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+

C 调用示例:

+
 1// 单核(私有存储)示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5        float *input = (float *)0x10000000;   // 输入在 L2 空间
+ 6        float *output = (float *)0x10010000;
+ 7        int input_shape[4] = {400, 2, 2, 3};
+ 8        int block_size[2] = {2, 2};
+ 9        int crops[4] = {0, 0, 0, 0};
+10        fp_batchtospace_p(input, output, input_shape, block_size, crops, sizeof(float));
+11        return 0;
+12}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/batchtospacend.html b/master/html/functionlib/dsplib/batchtospacend.html new file mode 100644 index 0000000..3998a3d --- /dev/null +++ b/master/html/functionlib/dsplib/batchtospacend.html @@ -0,0 +1,314 @@ + + + + + + + + + BatchToSpaceND — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

BatchToSpaceND

+

将输入张量的 batch 维度按 block 因子分解并重排到空间维度,随后根据 crops 参数对输出空间范围进行裁剪。

+
+
    +
  • 输入形状: [batch, height, width, channel]

  • +
  • block_size[block_h, block_w]

  • +
  • crops[top, bottom, left, right]

  • +
+
+
输入:
    +
  • input - 输入数据地址。

  • +
  • input_shape - 输入形状,格式为 [batch, height, width, channel]

  • +
  • block_size - 分块因子,格式为 [block_h, block_w]

  • +
  • crops - 裁剪参数,格式为 [top, bottom, left, right]

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • output - 输出数据地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。

  • +
  • MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。

  • +
+
+
+

共享存储版本:

+
+
+void i8_batchtospacend_s(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void i16_batchtospacend_s(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void i32_batchtospacend_s(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void hp_batchtospacend_s(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void fp_batchtospacend_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void dp_batchtospacend_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void c64_batchtospacend_s(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+
+ +
+
+void c128_batchtospacend_s(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size, int core_mask)
+

C 调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5    float *input = (float *)0xA0000000;   // 多核版本:输入放在 DDR 地址 0xA0000000
+ 6    float *output = (float *)0xB0000000;  // 多核版本:输出放在 DDR 地址 0xB0000000
+ 7    int input_shape[4] = {400, 2, 2, 3};
+ 8    int block_size[2] = {2, 2};
+ 9    int crops[4] = {0, 0, 0, 0};
+10    int core_mask = 0xff;
+11    fp_batchtospacend_s(input, output, input_shape, block_size, crops, sizeof(float), core_mask);
+12    return 0;
+13}
+
+
+
+ +

私有存储版本:

+
+
+void i8_batchtospacend_p(int8_t *input, int8_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void i16_batchtospacend_p(int16_t *input, int16_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void i32_batchtospacend_p(int32_t *input, int32_t *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void hp_batchtospacend_p(half *input, half *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void fp_batchtospacend_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void dp_batchtospacend_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void c64_batchtospacend_p(float *input, float *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+
+ +
+
+void c128_batchtospacend_p(double *input, double *output, const int *input_shape, const int *block_size, const int *crops, int data_size)
+

C 调用示例:

+
 1// FT78NE 单核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5    float *input = (float *)0x10000000;   // 单核版本:输入放在 L2 地址 0x10000000
+ 6    float *output = (float *)0x10010000;  // 单核版本:输出放在 L2 地址 0x10010000
+ 7    int input_shape[4] = {400, 2, 2, 3};
+ 8    int block_size[2] = {2, 2};
+ 9    int crops[4] = {0, 0, 0, 0};
+10    fp_batchtospacend_p(input, output, input_shape, block_size, crops, sizeof(float));
+11    return 0;
+12}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/broadcastto.html b/master/html/functionlib/dsplib/broadcastto.html new file mode 100644 index 0000000..dd0650d --- /dev/null +++ b/master/html/functionlib/dsplib/broadcastto.html @@ -0,0 +1,309 @@ + + + + + + + + + BroadcastTo — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

BroadcastTo

+

将较小张量按 N-D 规则广播到目标形状并写入输出。

+
+
+
输入:
    +
  • input - 输入数据地址。

  • +
  • input_shape - 输入形状数组。

  • +
  • input_shape_size - 输入形状长度。

  • +
  • output_shape - 目标输出形状数组。

  • +
  • output_shape_size - 输出形状长度。

  • +
  • data_size - 单个元素字节数(例如 sizeof(float))。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • output - 输出数据地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。

  • +
  • MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。

  • +
+
+
+

共享存储版本:

+
+
+void i8_broadcastto_s(int8_t *input, int8_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+
+ +
+
+void i16_broadcastto_s(int16_t *input, int16_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+
+ +
+
+void i32_broadcastto_s(int32_t *input, int32_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+
+ +
+
+void hp_broadcastto_s(half *input, half *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+
+ +
+
+void fp_broadcastto_s(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+
+ +
+
+void dp_broadcastto_s(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+
+ +
+
+void c64_broadcastto_s(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+
+ +
+
+void c128_broadcastto_s(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size, int core_mask)
+

C 调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5    float *input = (float *)0xA0000000;    // 输入在 DDR 地址 0xA0000000
+ 6    float *output = (float *)0xB0000000;   // 输出在 DDR 地址 0xB0000000
+ 7    int input_shape[4] = {1, 50, 1, 1};
+ 8    int output_shape[4] = {10, 50, 20, 1};
+ 9    int core_mask = 0xff;
+10    fp_broadcastto_s(input, output, input_shape, 4, output_shape, 4, sizeof(float), core_mask);
+11    return 0;
+12}
+
+
+
+ +

私有存储版本:

+
+
+void i8_broadcastto_p(int8_t *input, int8_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+
+ +
+
+void i16_broadcastto_p(int16_t *input, int16_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+
+ +
+
+void i32_broadcastto_p(int32_t *input, int32_t *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+
+ +
+
+void hp_broadcastto_p(half *input, half *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+
+ +
+
+void fp_broadcastto_p(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+
+ +
+
+void dp_broadcastto_p(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+
+ +
+
+void c64_broadcastto_p(float *input, float *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+
+ +
+
+void c128_broadcastto_p(double *input, double *output, const int *input_shape, int input_shape_size, const int *output_shape, int output_shape_size, int data_size)
+

C 调用示例:

+
 1// FT78NE 单核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5    float *input = (float *)0x10000000;   // 单核版本:输入放在 L2 地址 0x10000000
+ 6    float *output = (float *)0x10020000;  // 单核版本:输出放在 L2 地址 0x10020000
+ 7    int input_shape[4] = {1, 50, 1, 1};
+ 8    int output_shape[4] = {10, 50, 20, 1};
+ 9    fp_broadcastto_p(input, output, input_shape, 4, output_shape, 4, sizeof(float));
+10    return 0;
+11}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/conv2d.html b/master/html/functionlib/dsplib/conv2d.html new file mode 100644 index 0000000..0813ec4 --- /dev/null +++ b/master/html/functionlib/dsplib/conv2d.html @@ -0,0 +1,372 @@ + + + + + + + + + Conv2d — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

Conv2d

+

对输入 Tensor 计算二维卷积,输入的 shape 为 \((N, H_{in}, W_{in}, C_{in})\),其中 \(N\) 为 batch size,\(C\) 为通道数,\(H\) 为特征图的高度,\(W\) 为特征图的宽度。

+

根据以下公式计算输出:

+
+\[out(N_i, C_{out_j}) = bias(C_{out_j}) + \sum_{k=0}^{C_{in}-1} \text{ccor}(\text{weight}(C_{out_j}, k), X(N_i, k))\]
+

其中,\(bias\) 为输出偏置,\(\text{ccor}\) 为 cross-correlation 操作,\(weight\) 为卷积核的值,\(X\) 为输入的特征图。

+
    +
  • \(i\) 对应 batch 数,其范围为 \([0, N-1]\),其中 \(N\) 为输入 batch。

  • +
  • \(j\) 对应输出通道,其范围为 \([0, C_{out}-1]\),其中 \(C_{out}\) 为输出通道数,该值也等于卷积核的个数。

  • +
  • \(k\) 对应输入通道数,其范围为 \([0, C_{in}-1]\),其中 \(C_{in}\) 为输入通道数,该值也等于卷积核的通道数。

  • +
+

因此,上面的公式中,\(bias(C_{out_j})\) 为第 \(j\) 个输出通道的偏置,\(weight(C_{out_j}, k)\) 表示第 \(j\) 个卷积核在第 \(k\) 个输入通道的卷积核切片,\(X(N_i, k)\) 为特征图第 \(i\) 个 batch 第 \(k\) 个输入通道的切片。卷积核 shape 为 \((\text{kernel_size}[0], \text{kernel_size}[1])\),其中 kernel_size[0] 和 kernel_size[1] 是卷积核的高度和宽度。若考虑到输入输出通道以及 group,则完整卷积核的 shape 为 \((C_{out}, \text{kernel_size}[0], \text{kernel_size}[1], C_{in}/\text{group})\),其中 group 是分组卷积时在通道上分割输入 \(x\) 的组数。

+
+
输入:
    +
  • input_x - 输入数据的地址

  • +
  • input_w - 输入卷积核权重的地址

  • +
  • bias - 输入偏置的地址

  • +
  • conv_param - 算子计算所需参数的结构体。其各成员见下述。

  • +
  • quant_param - 对int8类型进行量化计算所需参数的结构体。其各成员见下述。

  • +
  • core_mask - 核掩码。

  • +
+
+
+

ConvParameter及ConvQuantParameter定义:

+
 1typedef struct ConvParameter {
+ 2    void* workspace_; // 用于存放中间计算结果
+ 3    int output_batch_; // 输出数据总批次
+ 4    int input_batch_; // 输入数据总批次
+ 5    int input_h_; // 输入数据h维度大小
+ 6    int input_w_; // 输入数据w维度大小
+ 7    int output_h_; // 输出数据h维度大小
+ 8    int output_w_; // 输出数据w维度大小
+ 9    int input_channel_; // 输入数据通道数
+10    int output_channel_; // 输出数据通道数
+11    int kernel_h_; // 卷积核h维度大小
+12    int kernel_w_; // 卷积核w维度大小
+13    int group_; // 组数
+14    int pad_l_; // 左填充大小
+15    int pad_u_; // 上填充大小
+16    int dilation_h_; // 卷积核h维度膨胀尺寸大小
+17    int dilation_w_; // 卷积核w维度膨胀尺寸大小
+18    int stride_h_; // 卷积核h维度步长
+19    int stride_w_; // 卷积核w维度步长
+20    int buffer_size_; // 为分块计算所分配的缓存大小
+21} ConvParameter;
+22
+23typedef struct ConvQuantParameter {
+24    int32_t* left_shift_;
+25    int32_t* right_shift_;
+26    int32_t* multiplier_;
+27    int32_t* filter_zp_ptr_;
+28    int32_t output_zp_;
+29    int32_t mini_;
+30    int32_t maxi_;
+31    int per_channel_;
+32} ConvQuantParameter;
+
+
+
+
输出:
    +
  • out_y - 输出地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持int8, fp32

  • +
  • MT7004 支持fp16, fp32

  • +
+
+

共享存储版本:

+
+
+void i8_conv2d_s(int8_t *input_x, int8_t *input_w, int8_t *out_y, int *bias, ConvParameter *conv_param, ConvQuantParameter quant_param, int core_mask)
+
+ +
+
+void hp_conv2d_s(half *input_x, half *input_w, half *out_y, half *bias, ConvParameter *conv_param, int core_mask)
+
+ +
+
+void fp_conv2d_s(float *input_x, float *input_w, float *out_y, float *bias, ConvParameter *conv_param, int core_mask)
+
+ +

C调用示例:

+
 1void TestConvSMCFp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) {
+ 2    int core_id = get_core_id();
+ 3    int logic_core_id = GetLogicCoreId(core_mask, core_id);
+ 4    int core_num = GetCoreNum(core_mask);
+ 5    float* input_data = (float*)0x88000000;
+ 6    float* weight = (float*)0x89000000;
+ 7    float* output_data = (float*)0x90000000;
+ 8    float* bias_data = (float*)0x91000000;
+ 9    ConvParameter* param = (ConvParameter*)0x92000000;
+10    if (logic_core_id == 0) {
+11        memcpy(bias_data, bias, sizeof(float) * output_shape[3]);
+12        param->dilation_h_ = dilation[0];
+13        param->dilation_w_ = dilation[1];
+14        param->group_ = groups;
+15        param->input_batch_ = input_shape[0];
+16        param->input_h_ = input_shape[1];
+17        param->input_w_ = input_shape[2];
+18        param->input_channel_ = input_shape[3];
+19        param->kernel_h_ = weight_shape[1];
+20        param->kernel_w_ = weight_shape[2];
+21        param->output_batch_ = output_shape[0];
+22        param->output_h_ = output_shape[1];
+23        param->output_w_ = output_shape[2];
+24        param->output_channel_ = output_shape[3];
+25        param->stride_h_ = stride[0];
+26        param->stride_w_ = stride[0];
+27        param->pad_u_ = padding[0];
+28        param->pad_l_ = padding[2];
+29        param->workspace_ = (float*)0x10000000; // workspace空间需分配在AM内,计算过程中会将数据搬运到workspace空间内进行计算
+30    }
+31    sys_bar(0, core_num); // 初始化参数完成后进行同步
+32    fp_conv2d_s(input_data, weight, output_data, bias_data, param, core_mask);
+33}
+34
+35void main(){
+36    int in_channel = 4;
+37    int out_channel = 4;
+38    int groups = 4;
+39    int input_shape[4] = {1, 30, 30, in_channel}; // NHWC
+40    int weight_shape[4] = {out_channel, 3, 3, in_channel / groups};
+41    int output_shape[4] = {1, 10, 10, out_channel}; // NHWC
+42    int stride[2] = {2, 2};
+43    int padding[4] = {1, 1, 1, 1};
+44    int dilation[2]= {2, 2};
+45    float bias[4] = {0, 0, 0, 0};
+46    int core_mask = 0b1111;
+47    TestConvSMCFp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask);
+48}
+
+
+

私有存储版本:

+
+
+void i8_conv2d_p(int8_t *input_x, int8_t *input_w, int8_t *out_y, int *bias, ConvParameter *conv_param, ConvQuantParameter quant_param, int core_mask)
+
+ +
+
+void hp_conv2d_p(half *input_x, half *input_w, half *out_y, half *bias, ConvParameter *conv_param, int core_mask)
+
+ +
+
+void fp_conv2d_p(float *input_x, float *input_w, float *out_y, float *bias, ConvParameter *conv_param, int core_mask)
+
+ +

C调用示例:

+
 1void TestConvL2Fp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) {
+ 2    float* input_data = (float*)0x10010000; // 私有存储版本地址设置在AM内
+ 3    float* weight = (float*)0x10020000;
+ 4    float* output_data = (float*)0x10030000;
+ 5    float* bias_data = (float*)0x10040000;
+ 6    ConvParameter* param = (ConvParameter*)0x10060000;
+ 7    memcpy(bias_data, bias, sizeof(float) * output_shape[3]);
+ 8    param->dilation_h_ = dilation[0];
+ 9    param->dilation_w_ = dilation[1];
+10    param->group_ = groups;
+11    param->input_batch_ = input_shape[0];
+12    param->input_h_ = input_shape[1];
+13    param->input_w_ = input_shape[2];
+14    param->input_channel_ = input_shape[3];
+15    param->kernel_h_ = weight_shape[1];
+16    param->kernel_w_ = weight_shape[2];
+17    param->output_batch_ = output_shape[0];
+18    param->output_h_ = output_shape[1];
+19    param->output_w_ = output_shape[2];
+20    param->output_channel_ = output_shape[3];
+21    param->stride_h_ = stride[0];
+22    param->stride_w_ = stride[0];
+23    param->pad_u_ = padding[0];
+24    param->pad_l_ = padding[2];
+25    param->workspace_ = (float*)0x10070000;
+26    param->buffer_size_ = 2048; // 私有存储版本中,必须设置该参数,用于确定分块计算的大小
+27    fp_conv2d_p(input_data, weight, output_data, bias_data, param, core_mask);
+28}
+29
+30void main(){
+31    int in_channel = 4;
+32    int out_channel = 4;
+33    int groups = 4;
+34    int input_shape[4] = {1, 30, 30, in_channel}; // NHWC
+35    int weight_shape[4] = {out_channel, 3, 3, in_channel / groups};
+36    int output_shape[4] = {1, 10, 10, out_channel}; // NHWC
+37    int stride[2] = {2, 2};
+38    int padding[4] = {1, 1, 1, 1};
+39    int dilation[2]= {2, 2};
+40    float bias[4] = {0, 0, 0, 0};
+41    int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动
+42    TestConvL2Fp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask);
+43}
+
+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/conv2d_transpose.html b/master/html/functionlib/dsplib/conv2d_transpose.html new file mode 100644 index 0000000..b1b4313 --- /dev/null +++ b/master/html/functionlib/dsplib/conv2d_transpose.html @@ -0,0 +1,364 @@ + + + + + + + + + Conv2dTranspose — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

Conv2dTranspose

+

计算二维转置卷积,可以视为 Conv2d 对输入求梯度,也称为反卷积(实际不是真正的反卷积)。

+

输入的 shape 通常为 \((N, H_{in}, W_{in}, C_{in})\),其中:

+
+
    +
  • \(N\) 是 batch size

  • +
  • \(C_{in}\) 是空间维度

  • +
  • \(H_{in}, W_{in}\) 分别为特征层的高度和宽度

  • +
+
+
+
输入:
    +
  • input_x - 输入数据的地址

  • +
  • input_w - 输入卷积核权重的地址

  • +
  • bias - 输入偏置的地址

  • +
  • param - 算子计算所需参数的结构体。其各成员见下述。

  • +
  • core_mask - 核掩码。

  • +
+
+
+

ConvTransposeParameter定义:

+
 1typedef struct ConvTransposeParameter {
+ 2    void* workspace_; // 用于存放中间计算结果
+ 3    int output_batch_; // 输出数据总批次
+ 4    int input_batch_; // 输入数据总批次
+ 5    int input_h_; // 输入数据h维度大小
+ 6    int input_w_; // 输入数据w维度大小
+ 7    int output_h_; // 输出数据h维度大小
+ 8    int output_w_; // 输出数据w维度大小
+ 9    int input_channel_; // 输入数据通道数
+10    int output_channel_; // 输出数据通道数
+11    int kernel_h_; // 卷积核h维度大小
+12    int kernel_w_; // 卷积核w维度大小
+13    int group_; // 组数
+14    int pad_l_; // 左填充大小
+15    int pad_u_; // 上填充大小
+16    int dilation_h_; // 卷积核h维度膨胀尺寸大小
+17    int dilation_w_; // 卷积核w维度膨胀尺寸大小
+18    int stride_h_; // 卷积核h维度步长
+19    int stride_w_; // 卷积核w维度步长
+20    int buffer_size_; // 为分块计算所分配的缓存大小
+21} ConvTransposeParameter;
+
+
+
+
输出:
    +
  • out_y - 输出地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持int8, fp32

  • +
  • MT7004 支持fp16, fp32

  • +
+
+

共享存储版本:

+
+
+void i8_convtranspose_s(int8_t *input_x, int8_t *input_w, int8_t *out_y, int *bias, ConvTransposeParameter *conv_param, int core_mask)
+
+ +
+
+void hp_convtranspose_s(half *input_x, half *input_w, half *out_y, half *bias, ConvTransposeParameter *conv_param, int core_mask)
+
+ +
+
+void fp_convtranspose_s(float *input_x, float *input_w, float *out_y, float *bias, ConvTransposeParameter *conv_param, int core_mask)
+
+ +

C调用示例:

+
 1void TestConvTransposeSMCFp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) {
+ 2    int core_id = get_core_id();
+ 3    int logic_core_id = GetLogicCoreId(core_mask, core_id);
+ 4    int core_num = GetCoreNum(core_mask);
+ 5    float* input_data = (float*)0x88000000;
+ 6    float* weight = (float*)0x89000000;
+ 7    float* output_data = (float*)0x90000000;
+ 8    float* bias_data = (float*)0x91000000;
+ 9    float* check = (float*)0x94000000;
+10    ConvTransposeParameter* param = (ConvTransposeParameter*)0x92000000;
+11    if (logic_core_id == 0) {
+12        memcpy(bias_data, bias, sizeof(float) * output_shape[3]);
+13        memset(output_data, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float));
+14        memset(check, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float));
+15        param->dilation_h_ = dilation[0];
+16        param->dilation_w_ = dilation[1];
+17        param->group_ = groups;
+18        param->input_batch_ = input_shape[0];
+19        param->input_h_ = input_shape[1];
+20        param->input_w_ = input_shape[2];
+21        param->input_channel_ = input_shape[3];
+22        param->kernel_h_ = weight_shape[1];
+23        param->kernel_w_ = weight_shape[2];
+24        param->output_batch_ = output_shape[0];
+25        param->output_h_ = output_shape[1];
+26        param->output_w_ = output_shape[2];
+27        param->output_channel_ = output_shape[3];
+28        param->stride_h_ = stride[0];
+29        param->stride_w_ = stride[0];
+30        param->pad_u_ = padding[0];
+31        param->pad_l_ = padding[2];
+32        param->workspace_ = (float*)0xA0000000;
+33    }
+34    sys_bar(0, core_num); // 初始化参数完成后进行同步
+35    fp_convtranspose_s(input_data, weight, output_data, bias_data, param, core_mask);
+36}
+37
+38void main(){
+39    int in_channel = 6;
+40    int out_channel = 6;
+41    int groups = 6;
+42    int input_shape[4] = {2, 5, 7, in_channel}; // NHWC
+43    int weight_shape[4] = {in_channel, 3, 3, out_channel / groups};
+44    int output_shape[4] = {2, 7, 9, out_channel}; // NHWC
+45    int stride[2] = {1, 1};
+46    int padding[4] = {0, 0, 0, 0};
+47    int dilation[2]= {1, 1};
+48    float bias[] = {0, 0, 0, 0, 0, 0};
+49    int core_mask = 0b1111;
+50    TestConvTransposeSMCFp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask);
+51}
+
+
+

私有存储版本:

+
+
+void i8_convtranspose_p(int8_t *input_x, int8_t *input_w, int8_t *out_y, int *bias, ConvTransposeParameter *conv_param, int core_mask)
+
+ +
+
+void hp_convtranspose_p(half *input_x, half *input_w, half *out_y, half *bias, ConvTransposeParameter *conv_param, int core_mask)
+
+ +
+
+void fp_convtranspose_p(float *input_x, float *input_w, float *out_y, float *bias, ConvTransposeParameter *conv_param, int core_mask)
+
+ +

C调用示例:

+
 1void TestConvTransposeL2Fp32(int* input_shape, int* weight_shape, int* output_shape, int* stride, int* padding, int* dilation, int groups, float* bias, int core_mask) {
+ 2    float* input_data = (float*)0x10000000; // 私有存储版本地址设置在AM内
+ 3    float* weight = (float*)0x10001000;
+ 4    float* output_data = (float*)0x10002000;
+ 5    float* bias_data = (float*)0x10003000;
+ 6    float* check = (float*)0x10004000;
+ 7    ConvTransposeParameter* param = (ConvTransposeParameter*)0x10005000;
+ 8    memcpy(bias_data, bias, sizeof(float) * output_shape[3]);
+ 9    memset(output_data, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float));
+10    memset(check, 0, output_shape[0] * output_shape[1] * output_shape[2] * output_shape[3] * sizeof(float));
+11    param->dilation_h_ = dilation[0];
+12    param->dilation_w_ = dilation[1];
+13    param->group_ = groups;
+14    param->input_batch_ = input_shape[0];
+15    param->input_h_ = input_shape[1];
+16    param->input_w_ = input_shape[2];
+17    param->input_channel_ = input_shape[3];
+18    param->kernel_h_ = weight_shape[1];
+19    param->kernel_w_ = weight_shape[2];
+20    param->output_batch_ = output_shape[0];
+21    param->output_h_ = output_shape[1];
+22    param->output_w_ = output_shape[2];
+23    param->output_channel_ = output_shape[3];
+24    param->stride_h_ = stride[0];
+25    param->stride_w_ = stride[0];
+26    param->pad_u_ = padding[0];
+27    param->pad_l_ = padding[2];
+28    param->workspace_ = (float*)0x10006000;
+29    param->buffer_size_ = 1024; // 私有存储版本中,必须设置该参数,用于确定分块计算的大小
+30    fp_convtranspose_p(input_data, weight, output_data, bias_data, param, core_mask);
+31}
+32
+33void main(){
+34    int in_channel = 6;
+35    int out_channel = 6;
+36    int groups = 6;
+37    int input_shape[4] = {2, 5, 7, in_channel}; // NHWC
+38    int weight_shape[4] = {in_channel, 3, 3, out_channel / groups};
+39    int output_shape[4] = {2, 7, 9, out_channel}; // NHWC
+40    int stride[2] = {1, 1};
+41    int padding[4] = {0, 0, 0, 0};
+42    int dilation[2]= {1, 1};
+43    float bias[] = {0, 0, 0, 0, 0, 0};
+44    int core_mask = 0b0001; // 私有存储版本只能设置为一个核心启动
+45    TestConvTransposeL2Fp32(input_shape, weight_shape, output_shape, stride, padding, dilation, groups, bias, core_mask);
+46}
+
+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/conv2dbackpropfilterfusion.html b/master/html/functionlib/dsplib/conv2dbackpropfilterfusion.html new file mode 100644 index 0000000..94ed8a0 --- /dev/null +++ b/master/html/functionlib/dsplib/conv2dbackpropfilterfusion.html @@ -0,0 +1,314 @@ + + + + + + + + + Conv2DBackpropFilterFusion — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

Conv2DBackpropFilterFusion

+

计算二维卷积反向传播的权重梯度(Conv2D backprop filter fusion),支持常规卷积、Depthwise 卷积以及 1x1 优化路径,多核按批次与空间维度分块协同完成。

+
+
+\[dw = \text{Conv2DGradFilter}(x, dy)\]
+
+
输入:
    +
  • dy - 输出梯度张量首地址,形状 [batch, out_h, out_w, out_channel]

  • +
  • x - 正向输入张量首地址,形状 [batch, in_h, in_w, in_channel]

  • +
  • conv_param - 卷积参数结构体地址,包含 stridepaddilationgroup、输入输出维度及共享工作空间指针等信息。

  • +
+

ConvParameter 字段说明:

+
    +
  • workspace_ - 指向算子运行时使用的临时工作空间,需满足对齐与容量要求。

  • +
  • output_batch_ - 输出梯度 dy 的批次数(通常等于输入批次数)。

  • +
  • input_batch_ - 正向输入 x 的批次数,用于与 output_batch_ 校验。

  • +
  • input_h_ / input_w_ - 正向输入特征图的高度与宽度。

  • +
  • output_h_ / output_w_ - 输出梯度特征图的高度与宽度。

  • +
  • input_channel_ / output_channel_ - 输入与输出通道数,需与 group_ 配合满足整除关系。

  • +
  • kernel_h_ / kernel_w_ - 卷积核的高与宽。

  • +
  • group_ - 组卷积数量,group_ = 1 表示普通卷积。

  • +
  • pad_l_ / pad_r_ / pad_u_ / pad_d_ - 分别表示左右上下方向的填充大小。

  • +
  • dilation_h_ / dilation_w_ - 核心采样间隔(膨胀系数)。

  • +
  • stride_h_ / stride_w_ - 滑动窗口在高、宽方向的步长。

  • +
  • buffer_size_ - 分配给 workspace_ 的缓冲区字节数,在运行前需要正确设置。

  • +
  • nweights_ - 卷积权重 w 的元素总数,用于内部分块和校验。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • dw - 卷积核梯度张量首地址,形状 [out_channel, in_channel/group, kernel_h, kernel_w]

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持 fp32 数据类型。

  • +
  • MT7004 支持 fp16、fp32 数据类型。

  • +
  • 需在 conv_param->workspace_ 中预先分配共享工作空间,长度不少于 conv_param->buffer_size_

  • +
+
+
+

共享存储版本:

+
+
+void hp_conv2dbackpropfilterfusion_s(const half *dy, const half *x, half *dw, ConvParameter *conv_param, int core_mask)
+
+ +
+
+void fp_conv2dbackpropfilterfusion_s(const float *dy, const float *x, float *dw, ConvParameter *conv_param, int core_mask)
+

C调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3#include "conv_parameter.h"
+ 4
+ 5int main(void) {
+ 6    const float *dy = (const float *)0xA0000000;   // DDR 存储
+ 7    const float *x = (const float *)0xB0000000;
+ 8    float *dw = (float *)0xC0000000;
+ 9    ConvParameter *param = (ConvParameter *)0xB0001000;
+10    // 设置 ConvParameter 字段
+11    param->workspace_ = (void *)0xB0002000;
+12    param->buffer_size_ = 0x20000;
+13    param->input_batch_ = 1;
+14    param->input_h_ = 3;
+15    param->input_w_ = 3;
+16    param->input_channel_ = 4;
+17    param->output_batch_ = 1;
+18    param->output_h_ = 3;
+19    param->output_w_ = 3;
+20    param->output_channel_ = 4;
+21    param->kernel_h_ = 2;
+22    param->kernel_w_ = 2;
+23    param->group_ = 2;
+24    param->pad_u_ = 1;
+25    param->pad_d_ = 0;
+26    param->pad_l_ = 1;
+27    param->pad_r_ = 0;
+28    param->dilation_h_ = 1;
+29    param->dilation_w_ = 1;
+30    param->stride_h_ = 1;
+31    param->stride_w_ = 1;
+32    param->nweights_ = 4 * 2 * 2 * 2;  // 示例值
+33    int core_mask = 0xff;
+34    fp_conv2dbackpropfilterfusion_s(dy, x, dw, param, core_mask);
+35    return 0;
+36}
+
+
+
+ +

私有存储版本:

+
+
+void hp_conv2dbackpropfilterfusion_p(const half *dy, const half *x, half *dw, ConvParameter *conv_param)
+
+ +
+
+void fp_conv2dbackpropfilterfusion_p(const float *dy, const float *x, float *dw, ConvParameter *conv_param)
+

C调用示例:

+
 1// MT7004 单核示例
+ 2#include <stdio.h>
+ 3#include "conv_parameter.h"
+ 4
+ 5int main(void) {
+ 6    const half *dy = (const half *)0x10000000;   // L2 存储
+ 7    const half *x = (const half *)0x10020000;
+ 8    half *dw = (half *)0x10040000;
+ 9    ConvParameter *param = (ConvParameter *)0x10060000;
+10    // 设置 ConvParameter 字段
+11    param->workspace_ = (void *)0x10070000;
+12    param->buffer_size_ = 0x10000;
+13    param->input_batch_ = 1;
+14    param->input_h_ = 3;
+15    param->input_w_ = 3;
+16    param->input_channel_ = 4;
+17    param->output_batch_ = 1;
+18    param->output_h_ = 3;
+19    param->output_w_ = 3;
+20    param->output_channel_ = 4;
+21    param->kernel_h_ = 2;
+22    param->kernel_w_ = 2;
+23    param->group_ = 2;
+24    param->pad_u_ = 1;
+25    param->pad_d_ = 0;
+26    param->pad_l_ = 1;
+27    param->pad_r_ = 0;
+28    param->dilation_h_ = 1;
+29    param->dilation_w_ = 1;
+30    param->stride_h_ = 1;
+31    param->stride_w_ = 1;
+32    param->nweights_ = 4 * 2 * 2 * 2;  // 示例值
+33    hp_conv2dbackpropfilterfusion_p(dy, x, dw, param);
+34    return 0;
+35}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/conv2dbackpropinputfusion.html b/master/html/functionlib/dsplib/conv2dbackpropinputfusion.html new file mode 100644 index 0000000..15757f4 --- /dev/null +++ b/master/html/functionlib/dsplib/conv2dbackpropinputfusion.html @@ -0,0 +1,314 @@ + + + + + + + + + Conv2DBackpropInputFusion — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

Conv2DBackpropInputFusion

+

计算二维卷积反向传播的输入梯度(Conv2D backprop input fusion),支持普通卷积、Depthwise 卷积以及 1x1 优化路径,多个核心通过核掩码协同完成批次并行。

+
+
+\[dx = \text{Conv2D}^\top(dy, w)\]
+
+
输入:
    +
  • dy - 输出梯度张量首地址,形状 [batch, out_h, out_w, out_channel]

  • +
  • w - 卷积权重张量首地址,形状 [out_channel, kernel_h, kernel_w, in_channel/group]

  • +
  • conv_param - 卷积参数结构体地址,包含 stridepaddilationgroup、输入输出维度、批次数及共享工作空间指针等信息。

  • +
+

ConvParameter 字段说明:

+
    +
  • workspace_ - 指向算子运行时使用的临时工作空间,需满足对齐与容量要求。

  • +
  • output_batch_ - 输出梯度 dy 的批次数(通常等于输入批次数)。

  • +
  • input_batch_ - 正向输入 x 的批次数,用于与 output_batch_ 校验。

  • +
  • input_h_ / input_w_ - 正向输入特征图的高度与宽度。

  • +
  • output_h_ / output_w_ - 输出梯度特征图的高度与宽度。

  • +
  • input_channel_ / output_channel_ - 输入与输出通道数,需与 group_ 配合满足整除关系。

  • +
  • kernel_h_ / kernel_w_ - 卷积核的高与宽。

  • +
  • group_ - 组卷积数量,group_ = 1 表示普通卷积。

  • +
  • pad_l_ / pad_r_ / pad_u_ / pad_d_ - 分别表示左右上下方向的填充大小。

  • +
  • dilation_h_ / dilation_w_ - 核心采样间隔(膨胀系数)。

  • +
  • stride_h_ / stride_w_ - 滑动窗口在高、宽方向的步长。

  • +
  • buffer_size_ - 分配给 workspace_ 的缓冲区字节数,在运行前需要正确设置。

  • +
  • nweights_ - 卷积权重 w 的元素总数,用于内部分块和校验。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • dx - 输入梯度张量首地址,形状 [batch, in_h, in_w, in_channel]

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持 fp32 数据类型。

  • +
  • MT7004 支持 fp16、fp32 数据类型。

  • +
  • 需在 conv_param->workspace_ 中预先分配共享工作空间,并设置 conv_param->buffer_size_

  • +
+
+
+

共享存储版本:

+
+
+void hp_conv2dbackpropinputfusion_s(const half *dy, const half *w, half *dx, ConvParameter *conv_param, int core_mask)
+
+ +
+
+void fp_conv2dbackpropinputfusion_s(const float *dy, const float *w, float *dx, ConvParameter *conv_param, int core_mask)
+

C调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3#include "conv_parameter.h"
+ 4
+ 5int main(void) {
+ 6    const float *dy = (const float *)0xA0000000;   // DDR 存储
+ 7    const float *w = (const float *)0xB0000000;
+ 8    float *dx = (const float *)0xC0000000;
+ 9    ConvParameter *param = (ConvParameter *)0xB0001000;  // 卷积参数共享区域
+10    // 设置 ConvParameter 字段
+11    param->workspace_ = (void *)0xB0002000;               // 共享工作空间
+12    param->buffer_size_ = 0x20000;
+13    param->input_batch_ = 1;
+14    param->input_h_ = 5;
+15    param->input_w_ = 5;
+16    param->input_channel_ = 4;
+17    param->output_batch_ = 1;
+18    param->output_h_ = 3;
+19    param->output_w_ = 3;
+20    param->output_channel_ = 8;
+21    param->kernel_h_ = 3;
+22    param->kernel_w_ = 3;
+23    param->group_ = 1;
+24    param->pad_u_ = 1;
+25    param->pad_d_ = 1;
+26    param->pad_l_ = 1;
+27    param->pad_r_ = 1;
+28    param->dilation_h_ = 1;
+29    param->dilation_w_ = 1;
+30    param->stride_h_ = 2;
+31    param->stride_w_ = 2;
+32    param->nweights_ = 8 * 3 * 3 * 4;  // 示例值
+33    int core_mask = 0xff;
+34    fp_conv2dbackpropinputfusion_s(dy, w, dx, param, core_mask);
+35    return 0;
+36}
+
+
+
+ +

私有存储版本:

+
+
+void hp_conv2dbackpropinputfusion_p(const half *dy, const half *w, half *dx, ConvParameter *conv_param)
+
+ +
+
+void fp_conv2dbackpropinputfusion_p(const float *dy, const float *w, float *dx, ConvParameter *conv_param)
+

C调用示例:

+
 1// MT7004 单核示例
+ 2#include <stdio.h>
+ 3#include "conv_parameter.h"
+ 4
+ 5int main(void) {
+ 6    const half *dy = (const half *)0x10000000;   // L2 存储
+ 7    const half *w = (const half *)0x10020000;
+ 8    half *dx = (half *)0x10040000;
+ 9    ConvParameter *param = (ConvParameter *)0x10060000;
+10    // 设置 ConvParameter 字段
+11    param->workspace_ = (void *)0x10070000;
+12    param->buffer_size_ = 0x10000;
+13    param->input_batch_ = 1;
+14    param->input_h_ = 5;
+15    param->input_w_ = 5;
+16    param->input_channel_ = 4;
+17    param->output_batch_ = 1;
+18    param->output_h_ = 3;
+19    param->output_w_ = 3;
+20    param->output_channel_ = 8;
+21    param->kernel_h_ = 3;
+22    param->kernel_w_ = 3;
+23    param->group_ = 1;
+24    param->pad_u_ = 1;
+25    param->pad_d_ = 1;
+26    param->pad_l_ = 1;
+27    param->pad_r_ = 1;
+28    param->dilation_h_ = 1;
+29    param->dilation_w_ = 1;
+30    param->stride_h_ = 2;
+31    param->stride_w_ = 2;
+32    param->nweights_ = 8 * 3 * 3 * 4;  // 示例值
+33    hp_conv2dbackpropinputfusion_p(dy, w, dx, param);
+34    return 0;
+35}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/crop.html b/master/html/functionlib/dsplib/crop.html new file mode 100644 index 0000000..4190997 --- /dev/null +++ b/master/html/functionlib/dsplib/crop.html @@ -0,0 +1,233 @@ + + + + + + + + + Crop — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

Crop

+

类似于slice(张量切片)。不过只支持四维。

+
+
输入:
    +
  • input - 输入数据地址。

  • +
  • in_shape - 输入张量形状。

  • +
  • out_shape - 输出张量形状。

  • +
  • type_size - 输入和输出张量数据类型的长度。

  • +
  • offset - 每一维度裁剪开始的偏移量

  • +
  • axis - 裁剪开始的维度

  • +
  • core_mask - 核掩码。

  • +
+
+
输出:
    +
  • output - 输出地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持int8, fp32

  • +
  • MT7004 支持fp16, fp32

  • +
+
+

共享/私有存储版本:

+
+
+void anytype_crop_anycore(void *input, void *output, int *in_shape, int *out_shape, int type_size, int *offset, int axis, int core_mask)
+
+ +

各种数据类型、私有及共享空间版本均使用该函数。对于不同数据类型,改变type_size参数即可。

+

C调用示例:

+
 1void TestCropSMCFp32(int* in_shape, int* out_shape, int axis, int* offset_, int core_mask) {
+ 2    int core_id = get_core_id();
+ 3    int logic_core_id = GetLogicCoreId(core_mask, core_id);
+ 4    int core_num = GetCoreNum(core_mask);
+ 5    float* input = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可
+ 6    float* output = (float*)0x98000000;
+ 7    int* input_shape = (int*)0xA8000000;
+ 8    int* output_shape = (int*)0xA8200000;
+ 9    int type_size = sizeof(float);
+10    int* offset = (int*)0xA8410000;
+11    if (logic_core_id == 0) {
+12        memcpy(offset, offset_, sizeof(int) * (4 - axis));
+13        memcpy(input_shape, in_shape, sizeof(int) * 4);
+14        memcpy(output_shape, out_shape, sizeof(int) * 4);
+15    }
+16    sys_bar(0, core_num); // 初始化参数完成后进行同步
+17    anytype_crop_anycore(input, output, in_shape, out_shape, type_size, offset, axis, core_mask);
+18}
+19
+20void main(){
+21    int in_shape[4] = {2, 3, 3, 5};
+22    int out_shape[4] = {2, 2, 2, 5};
+23    int axis = 1;
+24    int offset[3] = {1, 1, 0};
+25    int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可
+26    TestCropSMCFp32(in_shape, out_shape, axis, offset, core_mask);
+27}
+
+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/crop_and_resize.html b/master/html/functionlib/dsplib/crop_and_resize.html new file mode 100644 index 0000000..ea9bc2c --- /dev/null +++ b/master/html/functionlib/dsplib/crop_and_resize.html @@ -0,0 +1,270 @@ + + + + + + + + + CropAndResize — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

CropAndResize

+

从输入图像Tensor中提取切片并调整其大小。仅支持双线性插值方法。

+
+
输入:
    +
  • src - 输入数据的地址。

  • +
  • box_idx - boxes的索引,box_idx[i]的值表示第i个框的图像的值。

  • +
  • boxes - 第i行表示box_index[i]图像区域的坐标,并且坐标[y1,x1,y2,x2]是归一化后的值。归一化后的坐标值y,映射到图像y*(image_height-1)处,因此归一化后的图像高度范围为[0,1],映射到实际图像高度范围为[0,image_height-1]。我们允许y1>y2,在这种情况下,视为原始图像的上下翻转变换。宽度尺寸的处理类似。坐标取值允许在[0,1]范围之外,在这种情况下,我们使用extrapolation_value外插值进行补齐。

  • +
  • param - 算子计算所需参数的结构体。其各成员见下述。

  • +
  • extrapolation_value - 外插值。

  • +
  • core_mask - 核掩码。

  • +
+
+
+

CropAndResizeParameter定义:

+
 1typedef struct CropAndResizeParameter {
+ 2    int* input_shape_; // 输入张量形状
+ 3    int* output_shape_; // 输出张量形状
+ 4    int* x_lefts_; // 用于存储预处理结果
+ 5    int* x_rights_; // 用于存储预处理结果
+ 6    int* y_tops_; // 用于存储预处理结果
+ 7    int* y_bottoms_; // 用于存储预处理结果
+ 8    void* x_weights_; // 用于存储预处理结果
+ 9    void* y_weights_; // 用于存储预处理结果
+10    void* line_buffers_; // 用于存储中间结果
+11} CropAndResizeParameter;
+
+
+
+
输出:
    +
  • output - 输出地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持int8, fp32

  • +
  • MT7004 支持fp16, fp32

  • +
+
+

共享/私有存储版本:

+
+
+void i8_crop_and_resize_anycore(int8_t *src, int8_t *dst, int *box_idx, float *boxes, CropAndResizeParameter *param, float extrapolation_value, int core_mask)
+
+ +
+
+void hp_crop_and_resize_anycore(half *src, half *dst, int *box_idx, float *boxes, CropAndResizeParameter *param, float extrapolation_value, int core_mask)
+
+ +
+
+void fp_crop_and_resize_anycore(float *src, float *dst, int *box_idx, float *boxes, CropAndResizeParameter *param, half extrapolation_value, int core_mask)
+
+ +

私有及共享空间版本均使用这些函数。

+

C调用示例:

+
 1void TestCropAndResizeSMCFp32(int* input_shape, int* output_shape, float* inp_boxes, int32_t* inp_box_idx, float extrapolation_value, int core_mask) {
+ 2    int core_id = get_core_id();
+ 3    int core_num = GetCoreNum(core_mask);
+ 4    int logic_core_id = GetLogicCoreId(core_mask, core_id);
+ 5    float* input = (float*)0x88000000; // 测试私有空间时地址设置在私有空间内即可
+ 6    float* output = (float*)0x89000000;
+ 7    float* boxes = (float*)0x8A000000;
+ 8    int* box_idx = (int*)0x8B000000;
+ 9    CropAndResizeParameter* param = (CropAndResizeParameter*)0x8C000000;
+10    if (logic_core_id == 0) {
+11        memcpy(boxes, inp_boxes, sizeof(float) * output_shape[0] * 4);
+12        memcpy(box_idx, inp_box_idx, sizeof(int) * output_shape[0]);
+13        param->input_shape_ = (int*)0x8D000000;
+14        memcpy(param->input_shape_, input_shape, sizeof(int) * 4);
+15        param->output_shape_ = (int*)0x8E000000;
+16        memcpy(param->output_shape_, output_shape, sizeof(int) * 4);
+17        param->line_buffers_ = (void*)0x8F000000;
+18        param->x_lefts_ = (int*)0x90000000;
+19        param->x_rights_ = (int*)0x91000000;
+20        param->y_bottoms_ = (int*)0x92000000;
+21        param->y_tops_ = (int*)0x93000000;
+22        param->x_weights_ = (void*)0x94000000;
+23        param->y_weights_ = (void*)0x95000000;
+24        PrepareCropAndResizeBilinear(param->input_shape_, boxes, param->output_shape_, param->y_bottoms_, param->y_tops_,
+25                                        param->x_lefts_, param->x_rights_, param->y_weights_, param->x_weights_); // 做预处理
+26    }
+27    sys_bar(0, core_num); // 初始化参数完成后进行同步
+28    fp_crop_and_resize_anycore(input, output, box_idx, boxes, param, extrapolation_value, core_mask);
+29}
+30
+31void main(){
+32    int input_shape[4] = {1, 4, 4, 4};
+33    int output_shape[4] = {1, 8, 8, 4};
+34    float boxes[4] = {0, 0, 0.5, 0.5};
+35    int box_idx[1] = {0};
+36    int core_mask = 0b1111; // 测试单核时核掩码设置为0b0001即可
+37    float extrapolation_value = 0.5;
+38    TestCropAndResizeSMCFp32(input_shape, output_shape, boxes, box_idx, extrapolation_value, core_mask);
+39}
+
+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/depthtospace.html b/master/html/functionlib/dsplib/depthtospace.html new file mode 100644 index 0000000..5b77d14 --- /dev/null +++ b/master/html/functionlib/dsplib/depthtospace.html @@ -0,0 +1,307 @@ + + + + + + + + + DepthToSpace — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+ +
+
+ +
+

DepthToSpace

+

将输入张量的深度通道按 block_size 分解并重排到空间维度(Depth -> Space)。

+
+
+
输入:
    +
  • input - 输入数据地址。

  • +
  • in_shape - 输入形状,格式为 [batch, height, width, channel]

  • +
  • block_size - block 因子(单个整数)。

  • +
  • data_size - 单个元素字节数(例如 sizeof(float))。

  • +
  • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

  • +
+
+
输出:
    +
  • output - 输出数据地址。

  • +
+
+
支持平台:

FT78NE +MT7004

+
+
+
+

备注

+
    +
  • FT78NE 支持的数据类型: fp32、fp64、cplx64、cplx128、int16、int8、int32。

  • +
  • MT7004 支持的数据类型: fp32、fp16、cplx64、int16、int32。

  • +
+
+
+

共享存储版本:

+
+
+void i8_depthtospace_s(int8_t *input, int8_t *output, const int *in_shape, int block_size, int data_size, int core_mask)
+
+ +
+
+void i16_depthtospace_s(int16_t *input, int16_t *output, const int *in_shape, int block_size, int data_size, int core_mask)
+
+ +
+
+void i32_depthtospace_s(int32_t *input, int32_t *output, const int *in_shape, int block_size, int data_size, int core_mask)
+
+ +
+
+void hp_depthtospace_s(half *input, half *output, const int *in_shape, int block_size, int data_size, int core_mask)
+
+ +
+
+void fp_depthtospace_s(float *input, float *output, const int *in_shape, int block_size, int data_size, int core_mask)
+
+ +
+
+void dp_depthtospace_s(double *input, double *output, const int *in_shape, int block_size, int data_size, int core_mask)
+
+ +
+
+void c64_depthtospace_s(float *input, float *output, const int *in_shape, int block_size, int data_size, int core_mask)
+
+ +
+
+void c128_depthtospace_s(double *input, double *output, const int *in_shape, int block_size, int data_size, int core_mask)
+

C 调用示例:

+
 1// FT78NE 多核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5    float *input = (float *)0xA0000000;   // 多核版本:输入放在 DDR 地址 0xA0000000
+ 6    float *output = (float *)0xB0000000;  // 多核版本:输出放在 DDR 地址 0xB0000000
+ 7    int in_shape[4] = {10, 16, 16, 4};
+ 8    int block_size = 2;
+ 9    int core_mask = 0xff;
+10    fp_depthtospace_s(input, output, in_shape, block_size, sizeof(float), core_mask);
+11    return 0;
+12}
+
+
+
+ +

私有存储版本:

+
+
+void i8_depthtospace_p(int8_t *input, int8_t *output, const int *in_shape, int block_size, int data_size)
+
+ +
+
+void i16_depthtospace_p(int16_t *input, int16_t *output, const int *in_shape, int block_size, int data_size)
+
+ +
+
+void i32_depthtospace_p(int32_t *input, int32_t *output, const int *in_shape, int block_size, int data_size)
+
+ +
+
+void hp_depthtospace_p(half *input, half *output, const int *in_shape, int block_size, int data_size)
+
+ +
+
+void fp_depthtospace_p(float *input, float *output, const int *in_shape, int block_size, int data_size)
+
+ +
+
+void dp_depthtospace_p(double *input, double *output, const int *in_shape, int block_size, int data_size)
+
+ +
+
+void c64_depthtospace_p(float *input, float *output, const int *in_shape, int block_size, int data_size)
+
+ +
+
+void c128_depthtospace_p(double *input, double *output, const int *in_shape, int block_size, int data_size)
+

C 调用示例:

+
 1// FT78NE 单核示例
+ 2#include <stdio.h>
+ 3
+ 4int main(int argc, char *argv[]) {
+ 5    float *input = (float *)0x10000000;   // 单核版本:输入放在 L2 地址 0x10000000
+ 6    float *output = (float *)0x10040000;  // 单核版本:输出放在 L2 地址 0x10040000
+ 7    int in_shape[4] = {10, 16, 16, 4};
+ 8    int block_size = 2;
+ 9    fp_depthtospace_p(input, output, in_shape, block_size, sizeof(float));
+10    return 0;
+11}
+
+
+
+ +
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/dsplib_index.html b/master/html/functionlib/dsplib/dsplib_index.html index 26a3d86..d974f35 100644 --- a/master/html/functionlib/dsplib/dsplib_index.html +++ b/master/html/functionlib/dsplib/dsplib_index.html @@ -54,6 +54,51 @@
  • 自定义算子列表
  • DSP Library C API Reference
  • @@ -91,6 +136,51 @@ diff --git a/master/html/functionlib/dsplib/eltwise.html b/master/html/functionlib/dsplib/eltwise.html new file mode 100644 index 0000000..dc7145b --- /dev/null +++ b/master/html/functionlib/dsplib/eltwise.html @@ -0,0 +1,321 @@ + + + + + + + + + Eltwise — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Eltwise

    +

    输入两个等长数组以及控制参数,根据控制参数的值决定对两个数组做对位相加、对位相乘或取最大值操作。

    +
    +\[\begin{split}\mathbf{output_i} = +\begin{cases} +\mathbf{Input0_i} \cdot \mathbf{Input1_i}, & \text{if } \text{eltwise_mode} = \text{Eltwise_PROD} \\[6pt] +\mathbf{Input0_i} + \mathbf{Input1_i}, & \text{if } \text{eltwise_mode} = \text{Eltwise_SUM} \\[6pt] +\max(\mathbf{Input0_i}, \mathbf{Input1_i}), & \text{if } \text{eltwise_mode} = \text{Eltwise_MAXIMUM} +\end{cases}\end{split}\]
    +
    +
    输入:
      +
    • Input0 - 第一个输入数据地址。

    • +
    • Input1 - 第二个输入数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持int8, int16, int32, fp32, fp64, cplx64(除最大值), cplx128(除最大值)

    • +
    • MT7004 支持fp16, fp32, int16, int32, cplx64(除最大值)

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_eltwise_s(int8_t *Input0, int8_t *Input1, int8_t *output, int length, int eltwise_mode_, int core_mask)
    +
    + +
    +
    +void i16_eltwise_s(int16_t *Input0, int16_t *Input1, int16_t *output, int length, int eltwise_mode_, int core_mask)
    +
    + +
    +
    +void i32_eltwise_s(int *Input0, int *Input1, int *output, int length, int eltwise_mode_, int core_mask)
    +
    + +
    +
    +void hp_eltwise_s(half *Input0, half *Input1, half *output, int length, int eltwise_mode_, int core_mask)
    +
    + +
    +
    +void fp_eltwise_s(float *Input0, float *Input1, float *output, int length, int eltwise_mode_, int core_mask)
    +
    + +
    +
    +void dp_eltwise_s(double *Input0, double *Input1, double *output, int length, int eltwise_mode_, int core_mask)
    +
    + +
    +
    +void c64_eltwise_s(float *Input0, float *Input1, float *output, int length, int eltwise_mode_, int core_mask)
    +
    + +
    +
    +void c128_eltwise_s(double *Input0, double *Input1, double *output, int length, int eltwise_mode_, int core_mask)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <eltwise.h>
    + 4#define Eltwise_PROD 0
    + 5#define Eltwise_SUM 1
    + 6#define Eltwise_MAXIMUM 2
    + 7int main(int argc, char* argv[]) {
    + 8    float *input0 = (float *)0xA0000000;   //input在DDR空间
    + 9    float *input1 = (float *)0xB0000000;
    +10    float *output = (float *)0xC0000000;
    +11    int length = 1000;
    +12    int eltwise_mode_ = Eltwise_SUM;
    +13    int core_mask = 0xff;
    +14    fp_eltwise_s(input0, input1, output, length, eltwise_mode_, core_mask);
    +15    return 0;
    +16}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_eltwise_p(int8_t *Input0, int8_t *Input1, int8_t *output, int eltwise_mode_, int length)
    +
    + +
    +
    +void i16_eltwise_p(int16_t *Input0, int16_t *Input1, int16_t *output, int eltwise_mode_, int length)
    +
    + +
    +
    +void i32_eltwise_p(int32_t *Input0, int32_t *Input1, int32_t *output, int eltwise_mode_, int length)
    +
    + +
    +
    +void hp_eltwise_p(half *Input0, half *Input1, bool *output, int eltwise_mode_, int length)
    +
    + +
    +
    +void fp_eltwise_p(float *Input0, float *Input1, float *output, int eltwise_mode_, int length)
    +
    + +
    +
    +void dp_eltwise_p(double *Input0, double *Input1, double *output, int eltwise_mode_, int length)
    +
    + +
    +
    +void c64_eltwise_p(float *Input0, float *Input1, float *output, int eltwise_mode_, int length)
    +
    + +
    +
    +void c128_eltwise_p(double *Input0, double *Input1, double *output, int eltwise_mode_, int length)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <eltwise.h>
    + 4#define Eltwise_PROD 0
    + 5#define Eltwise_SUM 1
    + 6#define Eltwise_MAXIMUM 2
    + 7
    + 8int main(int argc, char* argv[]) {
    + 9    float *input0 = (float *)0x10810000;   //input在L2空间
    +10    float *input1 = (float *)0x10820000;
    +11    float *output = (float *)0x10830000;
    +12    int length = 1000;
    +13    int eltwise_mode_ = Eltwise_SUM;
    +14    fp_eltwise_p(input0, input1, output, eltwise_mode_, length);
    +15    return 0;
    +16}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/embeddinglookup.html b/master/html/functionlib/dsplib/embeddinglookup.html new file mode 100644 index 0000000..9986e55 --- /dev/null +++ b/master/html/functionlib/dsplib/embeddinglookup.html @@ -0,0 +1,273 @@ + + + + + + + + + EmbeddingLookup — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    EmbeddingLookup

    +

    传入一个矩阵和一组索引,根据给定的索引提取对应的行(向量),若此行不曾被标记为“已正则化”,则对此行进行正则化处理,将结果拼接输出,若已经正则化过则直接输出。

    +
    +\[\begin{split}\forall k \in [1, ids\_size], \quad +\begin{cases} +\text{if } \textbf{is_regulated}[i_k] = 0, & + \begin{cases} + \displaystyle X_{i_k} \leftarrow + X_{i_k} \cdot \frac{\text{max_norm}} + {\sum_{j=1}^{layer\_size\_} X_{i_k, j}} \\[10pt] + \textbf{is_regulated}[i_k] \leftarrow 1 + \end{cases} \\[12pt] +\text{输出向量 } Y_k \leftarrow X_{i_k} +\end{cases}\end{split}\]
    +
    +
    输入:
      +
    • input_data - 输入矩阵数据地址。

    • +
    • ids - 输入索引的存储地址。

    • +
    • max_norm - 最大范数约束。

    • +
    • is_regulated - 记录矩阵行是否被正则化的标志数组。

    • +
    • ids_size_ - 输入索引个数。

    • +
    • layer_size_ - 输入矩阵的列数。

    • +
    • layer_num_ - 输入矩阵的行数。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 结果输出地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32

    • +
    • MT7004 支持fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void hp_embeddinglookup_s(half *input_data, int *ids, half *output, half max_norm_, bool *is_regulated, int ids_size_, int layer_size_, int layer_num_, int core_mask)
    +
    + +
    +
    +void fp_embeddinglookup_s(float *input_data, int *ids, float *output, float max_norm_, bool *is_regulated, int ids_size_, int layer_size_, int layer_num_, int core_mask)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <embeddinglookup.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input_data = (float *)0xA0000000;   //input在DDR空间
    + 7    float *output_data = (float *)0xA0872c00;
    + 8    int layer_size = 4;//列
    + 9    int layer_num = 5;//行
    +10    float max_norm = 4.5;
    +11    int ids[]={0,1,3};
    +12    int ids_size  = 3;//提取三行
    +13    bool *output = (bool *)0xC0000000;
    +14    bool is_regulated_[5] = {0};//layer_num
    +15    int core_mask = 0xff;
    +16    fp_embeddinglookup_s(input_data, ids, output_data, max_norm, is_regulated_, ids_size, layer_size, layer_num, core_mask);
    +17    return 0;
    +18}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void hp_embeddinglookup_p(half *Input0, half *Input1, bool *output, int length)
    +
    + +
    +
    +void fp_embeddinglookup_p(float *Input0, float *Input1, bool *output, int length)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <embeddinglookup.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input_data = (float *)0x10810000;   //input在DDR空间
    + 7    float *output_data = (float *)0x10820000;
    + 8    int layer_size = 4;//列
    + 9    int layer_num = 5;//行
    +10    float max_norm = 4.5;
    +11    int ids[]={0,1,3};
    +12    int ids_size  = 3;//提取三行
    +13    bool *output = (bool *)0xC0000000;
    +14    bool is_regulated_[5] = {0};//layer_num
    +15    fp_embeddinglookup_s(input_data, ids, output_data, max_norm, is_regulated_, ids_size, layer_size, layer_num);
    +16    return 0;
    +17}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/equal.html b/master/html/functionlib/dsplib/equal.html index 8613b6c..8ce0def 100644 --- a/master/html/functionlib/dsplib/equal.html +++ b/master/html/functionlib/dsplib/equal.html @@ -23,7 +23,7 @@ - + @@ -54,6 +54,51 @@
  • 自定义算子列表
  • DSP Library C API Reference
  • @@ -167,9 +212,9 @@ 7 float *input1 = (float *)0xB0000000; 8 bool *output = (bool *)0xC0000000; 9 int length = 1000; -10 int core_mask = 0xff; -11 fp_equal_p(input0, input1, output, length, core_mask); -12 return 0; +10 int core_mask = 0xff; +11 fp_equal_p(input0, input1, output, length, core_mask); +12 return 0; 13} @@ -220,12 +265,12 @@ 3#include <equal.h> 4 5int main(int argc, char* argv[]) { - 6 float *input0 = (float *)0x10000000; //input在L2空间 - 7 float *input1 = (float *)0x10001000; - 8 bool *output = (bool *)0xC0000000; - 9 int length = 1000; -10 fp_equal_p(input0, input1, output, length); -11 return 0; + 6 float *input0 = (float *)0x10810000; //input在L2空间 + 7 float *input1 = (float *)0x10820000; + 8 bool *output = (bool *)0x10830000; + 9 int length = 1000; +10 fp_equal_p(input0, input1, output, length); +11 return 0; 12} @@ -238,7 +283,7 @@