diff --git a/master/doctrees/environment.pickle b/master/doctrees/environment.pickle index ed85ef8..4ff7eba 100644 Binary files a/master/doctrees/environment.pickle and b/master/doctrees/environment.pickle differ diff --git a/master/doctrees/functionlib/dsplib/abs.doctree b/master/doctrees/functionlib/dsplib/abs.doctree new file mode 100644 index 0000000..3dff2d3 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/abs.doctree differ diff --git a/master/doctrees/functionlib/dsplib/absgrad.doctree b/master/doctrees/functionlib/dsplib/absgrad.doctree new file mode 100644 index 0000000..c70bb91 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/absgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/activation.doctree b/master/doctrees/functionlib/dsplib/activation.doctree index 80c1aff..afedab4 100644 Binary files a/master/doctrees/functionlib/dsplib/activation.doctree and b/master/doctrees/functionlib/dsplib/activation.doctree differ diff --git a/master/doctrees/functionlib/dsplib/activation_grad.doctree b/master/doctrees/functionlib/dsplib/activation_grad.doctree new file mode 100644 index 0000000..3caaa9f Binary files /dev/null and b/master/doctrees/functionlib/dsplib/activation_grad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/adam.doctree b/master/doctrees/functionlib/dsplib/adam.doctree new file mode 100644 index 0000000..29f1f51 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/adam.doctree differ diff --git a/master/doctrees/functionlib/dsplib/adamweightdecay.doctree b/master/doctrees/functionlib/dsplib/adamweightdecay.doctree index 0ecedd4..2b9ddba 100644 Binary files a/master/doctrees/functionlib/dsplib/adamweightdecay.doctree and b/master/doctrees/functionlib/dsplib/adamweightdecay.doctree differ diff --git a/master/doctrees/functionlib/dsplib/addfusion.doctree b/master/doctrees/functionlib/dsplib/addfusion.doctree new file mode 100644 index 0000000..17493b6 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/addfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/addgrad.doctree b/master/doctrees/functionlib/dsplib/addgrad.doctree new file mode 100644 index 0000000..49fc487 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/addgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/addn.doctree b/master/doctrees/functionlib/dsplib/addn.doctree new file mode 100644 index 0000000..81375bb Binary files /dev/null and b/master/doctrees/functionlib/dsplib/addn.doctree differ diff --git a/master/doctrees/functionlib/dsplib/affine.doctree b/master/doctrees/functionlib/dsplib/affine.doctree new file mode 100644 index 0000000..ce66fab Binary files /dev/null and b/master/doctrees/functionlib/dsplib/affine.doctree differ diff --git a/master/doctrees/functionlib/dsplib/all.doctree b/master/doctrees/functionlib/dsplib/all.doctree new file mode 100644 index 0000000..5b5841d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/all.doctree differ diff --git a/master/doctrees/functionlib/dsplib/allgather.doctree b/master/doctrees/functionlib/dsplib/allgather.doctree new file mode 100644 index 0000000..913a20e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/allgather.doctree differ diff --git a/master/doctrees/functionlib/dsplib/applymomentum.doctree b/master/doctrees/functionlib/dsplib/applymomentum.doctree index 9439431..797e015 100644 Binary files a/master/doctrees/functionlib/dsplib/applymomentum.doctree and b/master/doctrees/functionlib/dsplib/applymomentum.doctree differ diff --git a/master/doctrees/functionlib/dsplib/argmax.doctree b/master/doctrees/functionlib/dsplib/argmax.doctree new file mode 100644 index 0000000..88bf702 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/argmax.doctree differ diff --git a/master/doctrees/functionlib/dsplib/argmin.doctree b/master/doctrees/functionlib/dsplib/argmin.doctree new file mode 100644 index 0000000..8781de0 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/argmin.doctree differ diff --git a/master/doctrees/functionlib/dsplib/assign.doctree b/master/doctrees/functionlib/dsplib/assign.doctree new file mode 100644 index 0000000..be5051f Binary files /dev/null and b/master/doctrees/functionlib/dsplib/assign.doctree differ diff --git a/master/doctrees/functionlib/dsplib/assignadd.doctree b/master/doctrees/functionlib/dsplib/assignadd.doctree new file mode 100644 index 0000000..2e9bc46 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/assignadd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/audio_spectrogram.doctree b/master/doctrees/functionlib/dsplib/audio_spectrogram.doctree new file mode 100644 index 0000000..d2b7355 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/audio_spectrogram.doctree differ diff --git a/master/doctrees/functionlib/dsplib/avgpooling.doctree b/master/doctrees/functionlib/dsplib/avgpooling.doctree new file mode 100644 index 0000000..984b2cf Binary files /dev/null and b/master/doctrees/functionlib/dsplib/avgpooling.doctree differ diff --git a/master/doctrees/functionlib/dsplib/avgpoolinggrad.doctree b/master/doctrees/functionlib/dsplib/avgpoolinggrad.doctree index 232a32a..f690179 100644 Binary files a/master/doctrees/functionlib/dsplib/avgpoolinggrad.doctree and b/master/doctrees/functionlib/dsplib/avgpoolinggrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/batchnorm.doctree b/master/doctrees/functionlib/dsplib/batchnorm.doctree new file mode 100644 index 0000000..9bad90d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/batchnorm.doctree differ diff --git a/master/doctrees/functionlib/dsplib/batchnormgrad.doctree b/master/doctrees/functionlib/dsplib/batchnormgrad.doctree new file mode 100644 index 0000000..efd9364 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/batchnormgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/biasadd.doctree b/master/doctrees/functionlib/dsplib/biasadd.doctree new file mode 100644 index 0000000..de4c0ab Binary files /dev/null and b/master/doctrees/functionlib/dsplib/biasadd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/biasaddgrad.doctree b/master/doctrees/functionlib/dsplib/biasaddgrad.doctree new file mode 100644 index 0000000..01c3015 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/biasaddgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/binarycrossentropy.doctree b/master/doctrees/functionlib/dsplib/binarycrossentropy.doctree new file mode 100644 index 0000000..ec09ae2 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/binarycrossentropy.doctree differ diff --git a/master/doctrees/functionlib/dsplib/binarycrossentropygrad.doctree b/master/doctrees/functionlib/dsplib/binarycrossentropygrad.doctree new file mode 100644 index 0000000..e38d08d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/binarycrossentropygrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/cast.doctree b/master/doctrees/functionlib/dsplib/cast.doctree new file mode 100644 index 0000000..a85cf0d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/cast.doctree differ diff --git a/master/doctrees/functionlib/dsplib/ceil.doctree b/master/doctrees/functionlib/dsplib/ceil.doctree new file mode 100644 index 0000000..61385e1 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/ceil.doctree differ diff --git a/master/doctrees/functionlib/dsplib/clip.doctree b/master/doctrees/functionlib/dsplib/clip.doctree new file mode 100644 index 0000000..85ce06c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/clip.doctree differ diff --git a/master/doctrees/functionlib/dsplib/concat.doctree b/master/doctrees/functionlib/dsplib/concat.doctree new file mode 100644 index 0000000..604c379 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/concat.doctree differ diff --git a/master/doctrees/functionlib/dsplib/constant_of_shape.doctree b/master/doctrees/functionlib/dsplib/constant_of_shape.doctree new file mode 100644 index 0000000..3f03e41 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/constant_of_shape.doctree differ diff --git a/master/doctrees/functionlib/dsplib/conv2d.doctree b/master/doctrees/functionlib/dsplib/conv2d.doctree index e51c7ef..56560bb 100644 Binary files a/master/doctrees/functionlib/dsplib/conv2d.doctree and b/master/doctrees/functionlib/dsplib/conv2d.doctree differ diff --git a/master/doctrees/functionlib/dsplib/cos.doctree b/master/doctrees/functionlib/dsplib/cos.doctree new file mode 100644 index 0000000..ca9d34b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/cos.doctree differ diff --git a/master/doctrees/functionlib/dsplib/cumsum.doctree b/master/doctrees/functionlib/dsplib/cumsum.doctree new file mode 100644 index 0000000..5df0ceb Binary files /dev/null and b/master/doctrees/functionlib/dsplib/cumsum.doctree differ diff --git a/master/doctrees/functionlib/dsplib/customextractfeatures.doctree b/master/doctrees/functionlib/dsplib/customextractfeatures.doctree new file mode 100644 index 0000000..f8ff246 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/customextractfeatures.doctree differ diff --git a/master/doctrees/functionlib/dsplib/customnormalize.doctree b/master/doctrees/functionlib/dsplib/customnormalize.doctree new file mode 100644 index 0000000..0fbb7dc Binary files /dev/null and b/master/doctrees/functionlib/dsplib/customnormalize.doctree differ diff --git a/master/doctrees/functionlib/dsplib/custompredict.doctree b/master/doctrees/functionlib/dsplib/custompredict.doctree new file mode 100644 index 0000000..d7ad419 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/custompredict.doctree differ diff --git a/master/doctrees/functionlib/dsplib/deconvgradfilter.doctree b/master/doctrees/functionlib/dsplib/deconvgradfilter.doctree new file mode 100644 index 0000000..b19b629 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/deconvgradfilter.doctree differ diff --git a/master/doctrees/functionlib/dsplib/detection_post_process.doctree b/master/doctrees/functionlib/dsplib/detection_post_process.doctree new file mode 100644 index 0000000..5a2a486 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/detection_post_process.doctree differ diff --git a/master/doctrees/functionlib/dsplib/div_fusion.doctree b/master/doctrees/functionlib/dsplib/div_fusion.doctree new file mode 100644 index 0000000..4e75523 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/div_fusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/divgrad.doctree b/master/doctrees/functionlib/dsplib/divgrad.doctree new file mode 100644 index 0000000..6131e1e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/divgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/dropout.doctree b/master/doctrees/functionlib/dsplib/dropout.doctree new file mode 100644 index 0000000..22f3a8f Binary files /dev/null and b/master/doctrees/functionlib/dsplib/dropout.doctree differ diff --git a/master/doctrees/functionlib/dsplib/dropoutgrad.doctree b/master/doctrees/functionlib/dsplib/dropoutgrad.doctree new file mode 100644 index 0000000..102335c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/dropoutgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/dsplib_index.doctree b/master/doctrees/functionlib/dsplib/dsplib_index.doctree index a7d05d7..7f6e185 100644 Binary files a/master/doctrees/functionlib/dsplib/dsplib_index.doctree and b/master/doctrees/functionlib/dsplib/dsplib_index.doctree differ diff --git a/master/doctrees/functionlib/dsplib/dynamicquant.doctree b/master/doctrees/functionlib/dsplib/dynamicquant.doctree new file mode 100644 index 0000000..caee1c7 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/dynamicquant.doctree differ diff --git a/master/doctrees/functionlib/dsplib/eltwise.doctree b/master/doctrees/functionlib/dsplib/eltwise.doctree index 8dd4880..0e62444 100644 Binary files a/master/doctrees/functionlib/dsplib/eltwise.doctree and b/master/doctrees/functionlib/dsplib/eltwise.doctree differ diff --git a/master/doctrees/functionlib/dsplib/elu.doctree b/master/doctrees/functionlib/dsplib/elu.doctree new file mode 100644 index 0000000..53a9670 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/elu.doctree differ diff --git a/master/doctrees/functionlib/dsplib/embeddinglookup.doctree b/master/doctrees/functionlib/dsplib/embeddinglookup.doctree index 1845acc..d2ed94b 100644 Binary files a/master/doctrees/functionlib/dsplib/embeddinglookup.doctree and b/master/doctrees/functionlib/dsplib/embeddinglookup.doctree differ diff --git a/master/doctrees/functionlib/dsplib/erf.doctree b/master/doctrees/functionlib/dsplib/erf.doctree new file mode 100644 index 0000000..cfb40b0 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/erf.doctree differ diff --git a/master/doctrees/functionlib/dsplib/expfusion.doctree b/master/doctrees/functionlib/dsplib/expfusion.doctree index a1a8773..92d2de0 100644 Binary files a/master/doctrees/functionlib/dsplib/expfusion.doctree and b/master/doctrees/functionlib/dsplib/expfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fake_quant_with_min_max_vars.doctree b/master/doctrees/functionlib/dsplib/fake_quant_with_min_max_vars.doctree new file mode 100644 index 0000000..fdb799b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fake_quant_with_min_max_vars.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.doctree b/master/doctrees/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.doctree new file mode 100644 index 0000000..533b7ca Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fftimag.doctree b/master/doctrees/functionlib/dsplib/fftimag.doctree new file mode 100644 index 0000000..36c6b48 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fftimag.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fftreal.doctree b/master/doctrees/functionlib/dsplib/fftreal.doctree new file mode 100644 index 0000000..39cea84 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fftreal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fill.doctree b/master/doctrees/functionlib/dsplib/fill.doctree new file mode 100644 index 0000000..4304c4a Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fill.doctree differ diff --git a/master/doctrees/functionlib/dsplib/flatten.doctree b/master/doctrees/functionlib/dsplib/flatten.doctree new file mode 100644 index 0000000..c3bb1b3 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/flatten.doctree differ diff --git a/master/doctrees/functionlib/dsplib/flattengrad.doctree b/master/doctrees/functionlib/dsplib/flattengrad.doctree new file mode 100644 index 0000000..587f742 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/flattengrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/floormod.doctree b/master/doctrees/functionlib/dsplib/floormod.doctree new file mode 100644 index 0000000..0dd8b5d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/floormod.doctree differ diff --git a/master/doctrees/functionlib/dsplib/formattranspose.doctree b/master/doctrees/functionlib/dsplib/formattranspose.doctree new file mode 100644 index 0000000..c3b163c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/formattranspose.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fullconnection.doctree b/master/doctrees/functionlib/dsplib/fullconnection.doctree new file mode 100644 index 0000000..d06561b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/fullconnection.doctree differ diff --git a/master/doctrees/functionlib/dsplib/fusedbatchnorm.doctree b/master/doctrees/functionlib/dsplib/fusedbatchnorm.doctree index 76e3305..5401cb5 100644 Binary files a/master/doctrees/functionlib/dsplib/fusedbatchnorm.doctree and b/master/doctrees/functionlib/dsplib/fusedbatchnorm.doctree differ diff --git a/master/doctrees/functionlib/dsplib/gather.doctree b/master/doctrees/functionlib/dsplib/gather.doctree new file mode 100644 index 0000000..ef67125 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/gather.doctree differ diff --git a/master/doctrees/functionlib/dsplib/gather_nd.doctree b/master/doctrees/functionlib/dsplib/gather_nd.doctree new file mode 100644 index 0000000..9e0ef3d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/gather_nd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/gatherd.doctree b/master/doctrees/functionlib/dsplib/gatherd.doctree new file mode 100644 index 0000000..97d533c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/gatherd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/glu.doctree b/master/doctrees/functionlib/dsplib/glu.doctree new file mode 100644 index 0000000..a3a2390 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/glu.doctree differ diff --git a/master/doctrees/functionlib/dsplib/greater.doctree b/master/doctrees/functionlib/dsplib/greater.doctree new file mode 100644 index 0000000..3f75759 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/greater.doctree differ diff --git a/master/doctrees/functionlib/dsplib/greaterequal.doctree b/master/doctrees/functionlib/dsplib/greaterequal.doctree new file mode 100644 index 0000000..af29f7f Binary files /dev/null and b/master/doctrees/functionlib/dsplib/greaterequal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/groupnormfusion.doctree b/master/doctrees/functionlib/dsplib/groupnormfusion.doctree index d58a68b..b4341b4 100644 Binary files a/master/doctrees/functionlib/dsplib/groupnormfusion.doctree and b/master/doctrees/functionlib/dsplib/groupnormfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/hashtablelookup.doctree b/master/doctrees/functionlib/dsplib/hashtablelookup.doctree new file mode 100644 index 0000000..a0c44f4 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/hashtablelookup.doctree differ diff --git a/master/doctrees/functionlib/dsplib/instancenorm.doctree b/master/doctrees/functionlib/dsplib/instancenorm.doctree new file mode 100644 index 0000000..beb01de Binary files /dev/null and b/master/doctrees/functionlib/dsplib/instancenorm.doctree differ diff --git a/master/doctrees/functionlib/dsplib/invertpermutation.doctree b/master/doctrees/functionlib/dsplib/invertpermutation.doctree new file mode 100644 index 0000000..fabf159 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/invertpermutation.doctree differ diff --git a/master/doctrees/functionlib/dsplib/isfinite.doctree b/master/doctrees/functionlib/dsplib/isfinite.doctree new file mode 100644 index 0000000..abcce0c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/isfinite.doctree differ diff --git a/master/doctrees/functionlib/dsplib/l2norm.doctree b/master/doctrees/functionlib/dsplib/l2norm.doctree new file mode 100644 index 0000000..8dd6110 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/l2norm.doctree differ diff --git a/master/doctrees/functionlib/dsplib/layernormfusion.doctree b/master/doctrees/functionlib/dsplib/layernormfusion.doctree new file mode 100644 index 0000000..74d687e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/layernormfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/layernormgrad.doctree b/master/doctrees/functionlib/dsplib/layernormgrad.doctree new file mode 100644 index 0000000..cee7e6e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/layernormgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/less.doctree b/master/doctrees/functionlib/dsplib/less.doctree new file mode 100644 index 0000000..0cca289 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/less.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lessequal.doctree b/master/doctrees/functionlib/dsplib/lessequal.doctree new file mode 100644 index 0000000..f4e5b90 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lessequal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/linspace.doctree b/master/doctrees/functionlib/dsplib/linspace.doctree index d624e3e..dede35b 100644 Binary files a/master/doctrees/functionlib/dsplib/linspace.doctree and b/master/doctrees/functionlib/dsplib/linspace.doctree differ diff --git a/master/doctrees/functionlib/dsplib/log.doctree b/master/doctrees/functionlib/dsplib/log.doctree new file mode 100644 index 0000000..c9417e1 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/log.doctree differ diff --git a/master/doctrees/functionlib/dsplib/log1p.doctree b/master/doctrees/functionlib/dsplib/log1p.doctree new file mode 100644 index 0000000..f6a0ad2 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/log1p.doctree differ diff --git a/master/doctrees/functionlib/dsplib/loggrad.doctree b/master/doctrees/functionlib/dsplib/loggrad.doctree new file mode 100644 index 0000000..0d408af Binary files /dev/null and b/master/doctrees/functionlib/dsplib/loggrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/logical_not.doctree b/master/doctrees/functionlib/dsplib/logical_not.doctree new file mode 100644 index 0000000..43341a6 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/logical_not.doctree differ diff --git a/master/doctrees/functionlib/dsplib/logical_or.doctree b/master/doctrees/functionlib/dsplib/logical_or.doctree new file mode 100644 index 0000000..c7eeef3 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/logical_or.doctree differ diff --git a/master/doctrees/functionlib/dsplib/logicaland.doctree b/master/doctrees/functionlib/dsplib/logicaland.doctree new file mode 100644 index 0000000..7b80d5d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/logicaland.doctree differ diff --git a/master/doctrees/functionlib/dsplib/logsoftmax.doctree b/master/doctrees/functionlib/dsplib/logsoftmax.doctree new file mode 100644 index 0000000..edeb8cf Binary files /dev/null and b/master/doctrees/functionlib/dsplib/logsoftmax.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lpnormalization.doctree b/master/doctrees/functionlib/dsplib/lpnormalization.doctree new file mode 100644 index 0000000..a476c43 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lpnormalization.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lrn.doctree b/master/doctrees/functionlib/dsplib/lrn.doctree new file mode 100644 index 0000000..5bc4256 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lrn.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lsh_projection.doctree b/master/doctrees/functionlib/dsplib/lsh_projection.doctree new file mode 100644 index 0000000..1aefcf2 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lsh_projection.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lstm.doctree b/master/doctrees/functionlib/dsplib/lstm.doctree index 74cf74b..b55ac65 100644 Binary files a/master/doctrees/functionlib/dsplib/lstm.doctree and b/master/doctrees/functionlib/dsplib/lstm.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lstmgrad.doctree b/master/doctrees/functionlib/dsplib/lstmgrad.doctree new file mode 100644 index 0000000..f0c822d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lstmgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lstmgraddata.doctree b/master/doctrees/functionlib/dsplib/lstmgraddata.doctree new file mode 100644 index 0000000..e32b22a Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lstmgraddata.doctree differ diff --git a/master/doctrees/functionlib/dsplib/lstmgradweight.doctree b/master/doctrees/functionlib/dsplib/lstmgradweight.doctree new file mode 100644 index 0000000..c3b3af8 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/lstmgradweight.doctree differ diff --git a/master/doctrees/functionlib/dsplib/maximum.doctree b/master/doctrees/functionlib/dsplib/maximum.doctree new file mode 100644 index 0000000..03e0573 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/maximum.doctree differ diff --git a/master/doctrees/functionlib/dsplib/maximumgrad.doctree b/master/doctrees/functionlib/dsplib/maximumgrad.doctree new file mode 100644 index 0000000..d27922f Binary files /dev/null and b/master/doctrees/functionlib/dsplib/maximumgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/maxpoolfusion.doctree b/master/doctrees/functionlib/dsplib/maxpoolfusion.doctree new file mode 100644 index 0000000..5ef3930 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/maxpoolfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/maxpoolgrad.doctree b/master/doctrees/functionlib/dsplib/maxpoolgrad.doctree new file mode 100644 index 0000000..cbbfe7b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/maxpoolgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/mfcc.doctree b/master/doctrees/functionlib/dsplib/mfcc.doctree new file mode 100644 index 0000000..4fc2d1d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/mfcc.doctree differ diff --git a/master/doctrees/functionlib/dsplib/minimum.doctree b/master/doctrees/functionlib/dsplib/minimum.doctree new file mode 100644 index 0000000..bd08c87 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/minimum.doctree differ diff --git a/master/doctrees/functionlib/dsplib/minimumgrad.doctree b/master/doctrees/functionlib/dsplib/minimumgrad.doctree new file mode 100644 index 0000000..a3246f3 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/minimumgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/mod.doctree b/master/doctrees/functionlib/dsplib/mod.doctree new file mode 100644 index 0000000..515dae9 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/mod.doctree differ diff --git a/master/doctrees/functionlib/dsplib/mul.doctree b/master/doctrees/functionlib/dsplib/mul.doctree new file mode 100644 index 0000000..9119b07 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/mul.doctree differ diff --git a/master/doctrees/functionlib/dsplib/mulgrad.doctree b/master/doctrees/functionlib/dsplib/mulgrad.doctree new file mode 100644 index 0000000..e90bbc9 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/mulgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/neg.doctree b/master/doctrees/functionlib/dsplib/neg.doctree new file mode 100644 index 0000000..66371c2 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/neg.doctree differ diff --git a/master/doctrees/functionlib/dsplib/neg_grad.doctree b/master/doctrees/functionlib/dsplib/neg_grad.doctree new file mode 100644 index 0000000..3b865f4 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/neg_grad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/nllloss.doctree b/master/doctrees/functionlib/dsplib/nllloss.doctree new file mode 100644 index 0000000..375930d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/nllloss.doctree differ diff --git a/master/doctrees/functionlib/dsplib/nlllossgrad.doctree b/master/doctrees/functionlib/dsplib/nlllossgrad.doctree new file mode 100644 index 0000000..44b9dfa Binary files /dev/null and b/master/doctrees/functionlib/dsplib/nlllossgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/non_max_suppression.doctree b/master/doctrees/functionlib/dsplib/non_max_suppression.doctree new file mode 100644 index 0000000..c5438de Binary files /dev/null and b/master/doctrees/functionlib/dsplib/non_max_suppression.doctree differ diff --git a/master/doctrees/functionlib/dsplib/nonzero.doctree b/master/doctrees/functionlib/dsplib/nonzero.doctree new file mode 100644 index 0000000..41d56cf Binary files /dev/null and b/master/doctrees/functionlib/dsplib/nonzero.doctree differ diff --git a/master/doctrees/functionlib/dsplib/not_equal.doctree b/master/doctrees/functionlib/dsplib/not_equal.doctree new file mode 100644 index 0000000..2c777d0 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/not_equal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/onehot.doctree b/master/doctrees/functionlib/dsplib/onehot.doctree new file mode 100644 index 0000000..8886b13 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/onehot.doctree differ diff --git a/master/doctrees/functionlib/dsplib/ones_like.doctree b/master/doctrees/functionlib/dsplib/ones_like.doctree new file mode 100644 index 0000000..ef5ed5e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/ones_like.doctree differ diff --git a/master/doctrees/functionlib/dsplib/padfusion.doctree b/master/doctrees/functionlib/dsplib/padfusion.doctree new file mode 100644 index 0000000..ab43e74 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/padfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/pow_fusion.doctree b/master/doctrees/functionlib/dsplib/pow_fusion.doctree new file mode 100644 index 0000000..0faee08 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/pow_fusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/power_grad.doctree b/master/doctrees/functionlib/dsplib/power_grad.doctree new file mode 100644 index 0000000..18cb4e7 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/power_grad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/prelufusion.doctree b/master/doctrees/functionlib/dsplib/prelufusion.doctree new file mode 100644 index 0000000..762a966 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/prelufusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/priorbox.doctree b/master/doctrees/functionlib/dsplib/priorbox.doctree new file mode 100644 index 0000000..a7f1dc6 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/priorbox.doctree differ diff --git a/master/doctrees/functionlib/dsplib/quantdtypecast.doctree b/master/doctrees/functionlib/dsplib/quantdtypecast.doctree new file mode 100644 index 0000000..0ee0d36 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/quantdtypecast.doctree differ diff --git a/master/doctrees/functionlib/dsplib/random_normal.doctree b/master/doctrees/functionlib/dsplib/random_normal.doctree new file mode 100644 index 0000000..64467d1 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/random_normal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/random_standard_normal.doctree b/master/doctrees/functionlib/dsplib/random_standard_normal.doctree new file mode 100644 index 0000000..81f50ba Binary files /dev/null and b/master/doctrees/functionlib/dsplib/random_standard_normal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/range.doctree b/master/doctrees/functionlib/dsplib/range.doctree index 36f5946..b1ef0af 100644 Binary files a/master/doctrees/functionlib/dsplib/range.doctree and b/master/doctrees/functionlib/dsplib/range.doctree differ diff --git a/master/doctrees/functionlib/dsplib/rank.doctree b/master/doctrees/functionlib/dsplib/rank.doctree new file mode 100644 index 0000000..0337fa7 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/rank.doctree differ diff --git a/master/doctrees/functionlib/dsplib/real_div.doctree b/master/doctrees/functionlib/dsplib/real_div.doctree new file mode 100644 index 0000000..85b3ace Binary files /dev/null and b/master/doctrees/functionlib/dsplib/real_div.doctree differ diff --git a/master/doctrees/functionlib/dsplib/reciprocal.doctree b/master/doctrees/functionlib/dsplib/reciprocal.doctree new file mode 100644 index 0000000..63f6106 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/reciprocal.doctree differ diff --git a/master/doctrees/functionlib/dsplib/reducescatter.doctree b/master/doctrees/functionlib/dsplib/reducescatter.doctree new file mode 100644 index 0000000..e575c73 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/reducescatter.doctree differ diff --git a/master/doctrees/functionlib/dsplib/reshape.doctree b/master/doctrees/functionlib/dsplib/reshape.doctree new file mode 100644 index 0000000..f841f71 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/reshape.doctree differ diff --git a/master/doctrees/functionlib/dsplib/resizegrad.doctree b/master/doctrees/functionlib/dsplib/resizegrad.doctree new file mode 100644 index 0000000..4e15b6d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/resizegrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/rfft.doctree b/master/doctrees/functionlib/dsplib/rfft.doctree new file mode 100644 index 0000000..2b86aed Binary files /dev/null and b/master/doctrees/functionlib/dsplib/rfft.doctree differ diff --git a/master/doctrees/functionlib/dsplib/roipooling.doctree b/master/doctrees/functionlib/dsplib/roipooling.doctree new file mode 100644 index 0000000..ae18a12 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/roipooling.doctree differ diff --git a/master/doctrees/functionlib/dsplib/round.doctree b/master/doctrees/functionlib/dsplib/round.doctree new file mode 100644 index 0000000..d15ec8e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/round.doctree differ diff --git a/master/doctrees/functionlib/dsplib/rsqrt.doctree b/master/doctrees/functionlib/dsplib/rsqrt.doctree new file mode 100644 index 0000000..f7733fa Binary files /dev/null and b/master/doctrees/functionlib/dsplib/rsqrt.doctree differ diff --git a/master/doctrees/functionlib/dsplib/rsqrtgrad.doctree b/master/doctrees/functionlib/dsplib/rsqrtgrad.doctree new file mode 100644 index 0000000..fd06d9d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/rsqrtgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/scatter_nd.doctree b/master/doctrees/functionlib/dsplib/scatter_nd.doctree new file mode 100644 index 0000000..6090089 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/scatter_nd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/scatter_nd_update.doctree b/master/doctrees/functionlib/dsplib/scatter_nd_update.doctree new file mode 100644 index 0000000..50adf0b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/scatter_nd_update.doctree differ diff --git a/master/doctrees/functionlib/dsplib/select.doctree b/master/doctrees/functionlib/dsplib/select.doctree new file mode 100644 index 0000000..2aaa9b5 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/select.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sgd.doctree b/master/doctrees/functionlib/dsplib/sgd.doctree index beacfbe..0f8f96e 100644 Binary files a/master/doctrees/functionlib/dsplib/sgd.doctree and b/master/doctrees/functionlib/dsplib/sgd.doctree differ diff --git a/master/doctrees/functionlib/dsplib/shape.doctree b/master/doctrees/functionlib/dsplib/shape.doctree new file mode 100644 index 0000000..ff02fbc Binary files /dev/null and b/master/doctrees/functionlib/dsplib/shape.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.doctree b/master/doctrees/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.doctree new file mode 100644 index 0000000..379df34 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sigmoidcrossentropywithlogits.doctree b/master/doctrees/functionlib/dsplib/sigmoidcrossentropywithlogits.doctree new file mode 100644 index 0000000..d670a86 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sigmoidcrossentropywithlogits.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sin.doctree b/master/doctrees/functionlib/dsplib/sin.doctree new file mode 100644 index 0000000..748f6cb Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sin.doctree differ diff --git a/master/doctrees/functionlib/dsplib/size.doctree b/master/doctrees/functionlib/dsplib/size.doctree new file mode 100644 index 0000000..67240ed Binary files /dev/null and b/master/doctrees/functionlib/dsplib/size.doctree differ diff --git a/master/doctrees/functionlib/dsplib/skipgram.doctree b/master/doctrees/functionlib/dsplib/skipgram.doctree new file mode 100644 index 0000000..f3d899f Binary files /dev/null and b/master/doctrees/functionlib/dsplib/skipgram.doctree differ diff --git a/master/doctrees/functionlib/dsplib/slice.doctree b/master/doctrees/functionlib/dsplib/slice.doctree new file mode 100644 index 0000000..980d089 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/slice.doctree differ diff --git a/master/doctrees/functionlib/dsplib/smooth1loss.doctree b/master/doctrees/functionlib/dsplib/smooth1loss.doctree new file mode 100644 index 0000000..e1c2888 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/smooth1loss.doctree differ diff --git a/master/doctrees/functionlib/dsplib/smoothl1lossgrad.doctree b/master/doctrees/functionlib/dsplib/smoothl1lossgrad.doctree new file mode 100644 index 0000000..9d99e71 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/smoothl1lossgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/softmax.doctree b/master/doctrees/functionlib/dsplib/softmax.doctree new file mode 100644 index 0000000..64b3f2b Binary files /dev/null and b/master/doctrees/functionlib/dsplib/softmax.doctree differ diff --git a/master/doctrees/functionlib/dsplib/softmax_cross_entropy_with_logits.doctree b/master/doctrees/functionlib/dsplib/softmax_cross_entropy_with_logits.doctree new file mode 100644 index 0000000..9caa277 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/softmax_cross_entropy_with_logits.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sparse_softmax_cross_entropy_with_logits.doctree b/master/doctrees/functionlib/dsplib/sparse_softmax_cross_entropy_with_logits.doctree new file mode 100644 index 0000000..3bca445 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sparse_softmax_cross_entropy_with_logits.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sparsefillemptyrows.doctree b/master/doctrees/functionlib/dsplib/sparsefillemptyrows.doctree new file mode 100644 index 0000000..171eed2 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sparsefillemptyrows.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sparsereshape.doctree b/master/doctrees/functionlib/dsplib/sparsereshape.doctree new file mode 100644 index 0000000..576d870 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sparsereshape.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sparsesegmentsum.doctree b/master/doctrees/functionlib/dsplib/sparsesegmentsum.doctree new file mode 100644 index 0000000..6337ae2 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sparsesegmentsum.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sparsetodense.doctree b/master/doctrees/functionlib/dsplib/sparsetodense.doctree new file mode 100644 index 0000000..af41a6a Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sparsetodense.doctree differ diff --git a/master/doctrees/functionlib/dsplib/splice.doctree b/master/doctrees/functionlib/dsplib/splice.doctree new file mode 100644 index 0000000..c1fbd88 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/splice.doctree differ diff --git a/master/doctrees/functionlib/dsplib/split.doctree b/master/doctrees/functionlib/dsplib/split.doctree new file mode 100644 index 0000000..9097523 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/split.doctree differ diff --git a/master/doctrees/functionlib/dsplib/split_with_overlap.doctree b/master/doctrees/functionlib/dsplib/split_with_overlap.doctree new file mode 100644 index 0000000..084ddd9 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/split_with_overlap.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sqrt.doctree b/master/doctrees/functionlib/dsplib/sqrt.doctree new file mode 100644 index 0000000..be20d2d Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sqrt.doctree differ diff --git a/master/doctrees/functionlib/dsplib/sqrtgrad.doctree b/master/doctrees/functionlib/dsplib/sqrtgrad.doctree new file mode 100644 index 0000000..4f4a364 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/sqrtgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/square.doctree b/master/doctrees/functionlib/dsplib/square.doctree new file mode 100644 index 0000000..570882c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/square.doctree differ diff --git a/master/doctrees/functionlib/dsplib/squaredifference.doctree b/master/doctrees/functionlib/dsplib/squaredifference.doctree new file mode 100644 index 0000000..921c6f5 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/squaredifference.doctree differ diff --git a/master/doctrees/functionlib/dsplib/stack.doctree b/master/doctrees/functionlib/dsplib/stack.doctree new file mode 100644 index 0000000..bc1ec60 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/stack.doctree differ diff --git a/master/doctrees/functionlib/dsplib/stridedslice.doctree b/master/doctrees/functionlib/dsplib/stridedslice.doctree new file mode 100644 index 0000000..ceacd98 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/stridedslice.doctree differ diff --git a/master/doctrees/functionlib/dsplib/stridedslicegrad.doctree b/master/doctrees/functionlib/dsplib/stridedslicegrad.doctree new file mode 100644 index 0000000..64af6c7 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/stridedslicegrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/subfusion.doctree b/master/doctrees/functionlib/dsplib/subfusion.doctree new file mode 100644 index 0000000..03627ea Binary files /dev/null and b/master/doctrees/functionlib/dsplib/subfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/subgrad.doctree b/master/doctrees/functionlib/dsplib/subgrad.doctree new file mode 100644 index 0000000..b7b0675 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/subgrad.doctree differ diff --git a/master/doctrees/functionlib/dsplib/switch.doctree b/master/doctrees/functionlib/dsplib/switch.doctree new file mode 100644 index 0000000..e9cbf89 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/switch.doctree differ diff --git a/master/doctrees/functionlib/dsplib/switchlayer.doctree b/master/doctrees/functionlib/dsplib/switchlayer.doctree new file mode 100644 index 0000000..bc8f129 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/switchlayer.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensor_scatter_add.doctree b/master/doctrees/functionlib/dsplib/tensor_scatter_add.doctree new file mode 100644 index 0000000..c2a0ba3 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensor_scatter_add.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorarray.doctree b/master/doctrees/functionlib/dsplib/tensorarray.doctree new file mode 100644 index 0000000..a32dff8 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorarray.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorarrayread.doctree b/master/doctrees/functionlib/dsplib/tensorarrayread.doctree new file mode 100644 index 0000000..44c2c80 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorarrayread.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorarraywrite.doctree b/master/doctrees/functionlib/dsplib/tensorarraywrite.doctree new file mode 100644 index 0000000..860e6a0 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorarraywrite.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorlistfromtensor.doctree b/master/doctrees/functionlib/dsplib/tensorlistfromtensor.doctree new file mode 100644 index 0000000..f21941e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorlistfromtensor.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorlistgetitem.doctree b/master/doctrees/functionlib/dsplib/tensorlistgetitem.doctree new file mode 100644 index 0000000..6212462 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorlistgetitem.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorlistreserve.doctree b/master/doctrees/functionlib/dsplib/tensorlistreserve.doctree new file mode 100644 index 0000000..683aed4 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorlistreserve.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorlistsetitem.doctree b/master/doctrees/functionlib/dsplib/tensorlistsetitem.doctree new file mode 100644 index 0000000..bff5a94 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorlistsetitem.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tensorliststack.doctree b/master/doctrees/functionlib/dsplib/tensorliststack.doctree new file mode 100644 index 0000000..1f30ccc Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tensorliststack.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tile.doctree b/master/doctrees/functionlib/dsplib/tile.doctree new file mode 100644 index 0000000..fab96a5 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tile.doctree differ diff --git a/master/doctrees/functionlib/dsplib/topkfusion.doctree b/master/doctrees/functionlib/dsplib/topkfusion.doctree new file mode 100644 index 0000000..5ae3606 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/topkfusion.doctree differ diff --git a/master/doctrees/functionlib/dsplib/transpose.doctree b/master/doctrees/functionlib/dsplib/transpose.doctree new file mode 100644 index 0000000..8c89f8e Binary files /dev/null and b/master/doctrees/functionlib/dsplib/transpose.doctree differ diff --git a/master/doctrees/functionlib/dsplib/tril.doctree b/master/doctrees/functionlib/dsplib/tril.doctree new file mode 100644 index 0000000..c2960ee Binary files /dev/null and b/master/doctrees/functionlib/dsplib/tril.doctree differ diff --git a/master/doctrees/functionlib/dsplib/triu.doctree b/master/doctrees/functionlib/dsplib/triu.doctree new file mode 100644 index 0000000..89061ec Binary files /dev/null and b/master/doctrees/functionlib/dsplib/triu.doctree differ diff --git a/master/doctrees/functionlib/dsplib/uniform_real.doctree b/master/doctrees/functionlib/dsplib/uniform_real.doctree new file mode 100644 index 0000000..23a5ed7 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/uniform_real.doctree differ diff --git a/master/doctrees/functionlib/dsplib/unique.doctree b/master/doctrees/functionlib/dsplib/unique.doctree new file mode 100644 index 0000000..c9cc32c Binary files /dev/null and b/master/doctrees/functionlib/dsplib/unique.doctree differ diff --git a/master/doctrees/functionlib/dsplib/unsortedsegmentsum.doctree b/master/doctrees/functionlib/dsplib/unsortedsegmentsum.doctree new file mode 100644 index 0000000..6132ca6 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/unsortedsegmentsum.doctree differ diff --git a/master/doctrees/functionlib/dsplib/unstack.doctree b/master/doctrees/functionlib/dsplib/unstack.doctree new file mode 100644 index 0000000..fea5586 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/unstack.doctree differ diff --git a/master/doctrees/functionlib/dsplib/where.doctree b/master/doctrees/functionlib/dsplib/where.doctree new file mode 100644 index 0000000..caa1962 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/where.doctree differ diff --git a/master/doctrees/functionlib/dsplib/zeroslike.doctree b/master/doctrees/functionlib/dsplib/zeroslike.doctree new file mode 100644 index 0000000..0071430 Binary files /dev/null and b/master/doctrees/functionlib/dsplib/zeroslike.doctree differ diff --git a/master/doctrees/functionlib/rstfiles.doctree b/master/doctrees/functionlib/rstfiles.doctree new file mode 100644 index 0000000..61331f5 Binary files /dev/null and b/master/doctrees/functionlib/rstfiles.doctree differ diff --git a/master/doctrees/rstfiles.doctree b/master/doctrees/rstfiles.doctree new file mode 100644 index 0000000..44b5e62 Binary files /dev/null and b/master/doctrees/rstfiles.doctree differ diff --git a/master/html/.buildinfo b/master/html/.buildinfo index 4a02113..09e6ba8 100644 --- a/master/html/.buildinfo +++ b/master/html/.buildinfo @@ -1,4 +1,4 @@ # Sphinx build info version 1 # This file records the configuration used when building these files. When it is not found, a full rebuild will be done. -config: b811412b4dcb327d38006d71aa1bfbad +config: 1f8651fb2d45b357aeb36cd10b73d834 tags: 645f666f9bcd5a90fca523b33c5a78b7 diff --git a/master/html/_sources/functionlib/dsplib/abs.rst.txt b/master/html/_sources/functionlib/dsplib/abs.rst.txt new file mode 100644 index 0000000..a1d9091 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/abs.rst.txt @@ -0,0 +1,95 @@ +Abs +================= + +对输入数组逐元素计算绝对值。 + +对于实数类型,返回其数值的绝对值; +对于复数类型,返回其模长。 + +.. math:: + + dst_i = + \begin{cases} + |src_i|, & \text{实数类型} \\ + \sqrt{\Re(src_i)^2 + \Im(src_i)^2}, & \text{复数类型} + \end{cases} + +输入: + - **src_data** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst_data** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp, dp, int8, int16, int32, cplx64, cplx128 + - MT7004 支持 hp, fp, int16, int32, cplx64 + - 对于整数类型,结果为对应数值的绝对值 + - 对于复数类型,输出为模长(实数) + +**共享存储版本:** + +.. c:function:: void i8_abs_s(int8_t* src_data, int8_t* dst_data, int length, int core_mask) +.. c:function:: void i16_abs_s(int16_t* src_data, int16_t* dst_data, int length, int core_mask) +.. c:function:: void i32_abs_s(int32_t* src_data, int32_t* dst_data, int length, int core_mask) +.. c:function:: void hp_abs_s(half* src_data, half* dst_data, int length, int core_mask) +.. c:function:: void fp_abs_s(float* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void dp_abs_s(double* src_data, double* dst_data, int length, int core_mask) +.. c:function:: void c64_abs_s(float* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void c128_abs_s(double* src_data, double* dst_data, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1024; + int core_mask = 0xff; + + fp_abs_s(input0, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_abs_p(int8_t* src_data, int8_t* dst_data, int length) +.. c:function:: void i16_abs_p(int16_t* src_data, int16_t* dst_data, int length) +.. c:function:: void i32_abs_p(int32_t* src_data, int32_t* dst_data, int length) +.. c:function:: void hp_abs_p(half* src_data, half* dst_data, int length) +.. c:function:: void fp_abs_p(float* src_data, float* dst_data, int length) +.. c:function:: void dp_abs_p(double* src_data, double* dst_data, int length) +.. c:function:: void c64_abs_p(float* src_data, float* dst_data, int length) +.. c:function:: void c128_abs_p(double* src_data, double* dst_data, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; // input在L2空间 + float *output = (float *)0x10820000; + int length = 1024; + + fp_abs_p(input0, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/absgrad.rst.txt b/master/html/_sources/functionlib/dsplib/absgrad.rst.txt new file mode 100644 index 0000000..025cfad --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/absgrad.rst.txt @@ -0,0 +1,80 @@ +Absgrad +================= + + +逐元素计算绝对值梯度 + +.. math:: + + output_i = \begin{cases} + input0_i, & \text{if } input1_i \geq 0 \\ + -input0_i, & \text{if } input1_i < 0 + \end{cases} + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **size** - 计算长度。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_absgrad_s(float *input0, float *input1, float *output, int size, int core_mask) +.. c:function:: void hp_absgrad_s(half *input0, half *input1, half *output, int size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *input1 = (float *)0xB0000000; + bool *output = (bool *)0xC0000000; + int size = 1000; + int core_mask = 0xff; + fp_absgrad_s(input0, input1, output, size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_absgrad_p(float *input0, float *input1, float *output, int size) +.. c:function:: void hp_absgrad_p(half *input0, half *input1, half *output, int size) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; //input在L2空间 + float *input1 = (float *)0x10001000; + bool *output = (bool *)0xC0000000; + int size = 1000; + fp_absgrad_p(input0, input1, output, size); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/activation.rst.txt b/master/html/_sources/functionlib/dsplib/activation.rst.txt index c33dc8a..41d3ae2 100644 --- a/master/html/_sources/functionlib/dsplib/activation.rst.txt +++ b/master/html/_sources/functionlib/dsplib/activation.rst.txt @@ -19,7 +19,7 @@ Activation .. math:: - output_i = \min(\max(input_i, \text{min_val}), \text{max_val}) + output_i = \min(\max(input_i, \text{min\_val}), \text{max\_val}) - ``LRelu`` - 带泄露的线性整流单元(Leaky Rectified Linear Unit),它在输入为正时保持线性,在输入为负时也保留一个很小的斜率,以避免标准 ReLU 中的“死亡神经元”问题。 @@ -97,7 +97,7 @@ Activation output_i = \begin{cases} - input_i, & input_i \gt 88.0 \\ + input_i, & input_i > 88.0 \\ \ln(1 + e^{input_i}), & \text{otherwise} \end{cases} @@ -107,7 +107,7 @@ Activation output_i = \begin{cases} - input_i, & input_i \ge 0 \\ + input_i, & input_i >= 0 \\ \alpha (e^{input_i} - 1), & input_i < 0 \end{cases} @@ -117,7 +117,7 @@ Activation .. math:: output_i = \begin{cases} - input_i, & input_i \ge 0 \\ + input_i, & input_i >= 0 \\ \alpha (e^{\frac{input_i}{alpha}} - 1), & input_i < 0 \end{cases} diff --git a/master/html/_sources/functionlib/dsplib/activation_grad.rst.txt b/master/html/_sources/functionlib/dsplib/activation_grad.rst.txt new file mode 100644 index 0000000..693c620 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/activation_grad.rst.txt @@ -0,0 +1,115 @@ +ActivationGrad +================= + +激活函数梯度计算算子系列。该系列算子根据激活函数的输入(或输出)以及上一层传回的梯度,计算并输出当前层的梯度。 + +.. math:: + + \text{通用公式:}\quad dst_i = src0_i \cdot f'(src1_i) + +其中 :math:`src0` 为梯度输入(Output Gradient),:math:`src1` 为激活函数的原始输入或输出(取决于具体激活函数类型),:math:`f'` 为激活函数的导数。 + +包含算子列表: + - **ReluGrad**: ReLU 激活梯度。 + - **Relu6Grad**: ReLU6 激活梯度。 + - **LeakyReluGrad (l_relu_grad)**: Leaky ReLU 激活梯度,需传入参数 ``alpha``。 + - **SigmoidGrad**: Sigmoid 激活梯度。 + - **TanhGrad**: Tanh 激活梯度。 + - **HSigmoidGrad**: Hard Sigmoid 激活梯度。 + - **HSwishGrad**: Hard Swish 激活梯度。 + - **GeluGrad**: GELU 激活梯度。 + - **EluGrad**: ELU 激活梯度,需传入参数 ``alpha``。 + - **SoftplusGrad**: Softplus 激活梯度。 + - **HardShrinkGrad**: Hard Shrink 激活梯度,需传入参数 ``lambd``。 + - **SoftshrinkGrad**: Softshrink 激活梯度,需传入参数 ``lambd``。 + +输入: + - **src0** - 梯度输入数据地址。 + - **src1** - 原始输入/输出数据地址。 + - **length** - 计算长度。 + - **alpha / lambd (float, 可选)** - 特定激活函数所需的系数。 + - **core_mask (int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst** - 梯度计算结果输出地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 仅支持 fp32。 + - MT7004 支持 fp32, fp16。 + - 对于不同的算子,``src1`` 的含义可能不同(例如 SigmoidGrad 通常使用前向传播的输出作为 src1,而 ReluGrad 使用前向传播的输入作为 src1),需确保上层传入地址正确。 + +**共享存储版本:** + +.. c:function:: void fp_relu_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void hp_relu_grad_s(half* src0, half* src1, half* dst, int length, int core_mask) +.. c:function:: void fp_relu6_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void fp_l_relu_grad_s(float* src0, float* src1, float* dst, int length, float alpha, int core_mask) +.. c:function:: void fp_sigmoid_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void fp_tanh_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void fp_h_sigmoid_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void fp_h_swish_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void fp_gelu_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void fp_elu_grad_s(float* src0, float* src1, float* dst, int length, float alpha, int core_mask) +.. c:function:: void fp_softplus_grad_s(float* src0, float* src1, float* dst, int length, int core_mask) +.. c:function:: void fp_hard_shrink_grad_s(float* src0, float* src1, float* dst, int length, float lambd, int core_mask) +.. c:function:: void fp_softshrink_grad_s(float* src0, float* src1, float* dst, int length, float lambd, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例(共享存储) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *src0 = (float *)0xA0000000; // 梯度输入在共享存储 + float *src1 = (float *)0xA1000000; // 原始数据在共享存储 + float *dst = (float *)0xB0000000; // 结果输出到共享存储 + int length = 1024; + int core_mask = 0xff; + fp_relu_grad_s(src0, src1, dst, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_relu_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void hp_relu_grad_p(half* src0, half* src1, half* dst, int length) +.. c:function:: void fp_relu6_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void fp_l_relu_grad_p(float* src0, float* src1, float* dst, int length, float alpha) +.. c:function:: void fp_sigmoid_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void fp_tanh_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void fp_h_sigmoid_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void fp_h_swish_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void fp_gelu_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void fp_elu_grad_p(float* src0, float* src1, float* dst, int length, float alpha) +.. c:function:: void fp_softplus_grad_p(float* src0, float* src1, float* dst, int length) +.. c:function:: void fp_hard_shrink_grad_p(float* src0, float* src1, float* dst, int length, float lambd) +.. c:function:: void fp_softshrink_grad_p(float* src0, float* src1, float* dst, int length, float lambd) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + //MT7004 示例 + #include + + int main(int argc, char* argv[]) { + float *src0 = (float *)0x10000000; // 私有存储空间地址 + float *src1 = (float *)0x10001000; + float *dst = (float *)0x10002000; + int length = 139; + float alpha = 0.01f; + fp_l_relu_grad_p(src0, src1, dst, length, alpha); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/adam.rst.txt b/master/html/_sources/functionlib/dsplib/adam.rst.txt new file mode 100644 index 0000000..f1904ff --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/adam.rst.txt @@ -0,0 +1,128 @@ +Adam +================= + + 使用 Adam 算法更新参数权重。支持 Nesterov 动量。 + + 算法逻辑如下: + + .. math:: + + m_t = m_{t-1} + (g_t - m_{t-1}) \cdot (1 - \beta_1) \\ + v_t = v_{t-1} + (g_t^2 - v_{t-1}) \cdot (1 - \beta_2) \\ + \hat{lr} = lr \cdot \frac{\sqrt{1 - \beta_2^t}}{1 - \beta_1^t} + + 如果不启用 Nesterov: + + .. math:: + + w_t = w_{t-1} - \hat{lr} \cdot \frac{m_t}{\sqrt{v_t} + \epsilon} + + 如果启用 Nesterov: + + .. math:: + + w_t = w_{t-1} - \hat{lr} \cdot \frac{m_t \cdot \beta_1 + (1 - \beta_1) \cdot g_t}{\sqrt{v_t} + \epsilon} + + 输入: + - **m** - 一阶矩向量地址(输入/输出)。 + - **v** - 二阶矩向量地址(输入/输出)。 + - **gradient** - 梯度向量地址。 + - **weight** - 权重向量地址(输入/输出)。 + - **beta1** - 一阶矩估计的指数衰减率。 + - **beta2** - 二阶矩估计的指数衰减率。 + - **beta1_power** - :math:`\beta_1^t` 的值(指针形式传入)。 + - **beta2_power** - :math:`\beta_2^t` 的值(指针形式传入)。 + - **eps** - 数值稳定性项 epsilon。 + - **learning_rate** - 学习率。 + - **nesterov** - 是否启用 Nesterov 动量(0: 不启用, 1: 启用)。 + - **start** - 计算的起始索引(包含)。 + - **end** - 计算的结束索引(不包含)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **m** - 更新后的一阶矩。 + - **v** - 更新后的二阶矩。 + - **weight** - 更新后的权重。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - 仅支持 fp32 数据类型。 + +**共享存储版本:** + +.. c:function:: void fp_adam_s(float* m, float* v, const float* gradient, float* weight, float beta1, float beta2, float* beta1_power, float* beta2_power, float eps, float learning_rate, int nesterov, int start, int end, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 25-26 + + #include + + int main(int argc, char* argv[]) { + // 假设所有数据均位于DDR空间 + float* m = (float*)0xC0000000; + float* v = (float*)0xC1000000; + float* gradient = (float*)0xC2000000; + float* weight = (float*)0xC3000000; + + // 标量参数 + float beta1 = 0.9f; + float beta2 = 0.999f; + float beta1_power_val = 0.9f; // beta1^1 + float beta2_power_val = 0.999f; // beta2^1 + float* beta1_power = &beta1_power_val; + float* beta2_power = &beta2_power_val; + float eps = 1e-8f; + float learning_rate = 0.001f; + int nesterov = 1; + + int start = 0; + int end = 800000; // 元素总数 + int core_mask = 0xff; // 使用所有核心 + + fp_adam_s(m, v, gradient, weight, beta1, beta2, beta1_power, beta2_power, + eps, learning_rate, nesterov, start, end, core_mask); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_adam_p(float* m, float* v, const float* gradient, float* weight, float beta1, float beta2, float* beta1_power, float* beta2_power, float eps, float learning_rate, int nesterov, int start, int end) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 20-21 + + #include + + int main(int argc, char* argv[]) { + // 假设所有数据均位于L2/AM空间 + float* m = (float*)0x10820000; + float* v = (float*)0x10830000; + float* gradient = (float*)0x10840000; + float* weight = (float*)0x10850000; + + float beta1 = 0.9f; + float beta2 = 0.999f; + float beta1_power_val = 0.9f; + float beta2_power_val = 0.999f; + float eps = 1e-8f; + float learning_rate = 0.001f; + int nesterov = 0; + int start = 0; + int end = 2000; + + fp_adam_p(m, v, gradient, weight, beta1, beta2, &beta1_power_val, &beta2_power_val, + eps, learning_rate, nesterov, start, end); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/adamweightdecay.rst.txt b/master/html/_sources/functionlib/dsplib/adamweightdecay.rst.txt index a83c44d..f2ae39a 100644 --- a/master/html/_sources/functionlib/dsplib/adamweightdecay.rst.txt +++ b/master/html/_sources/functionlib/dsplib/adamweightdecay.rst.txt @@ -48,7 +48,7 @@ AdamWeightDecay .. code-block:: c :linenos: - :emphasize-lines: 17 + :emphasize-lines: 17-19 // FT78NE 多核示例 #include @@ -84,7 +84,7 @@ AdamWeightDecay .. code-block:: c :linenos: - :emphasize-lines: 15 + :emphasize-lines: 15-17 // MT7004 单核示例 #include diff --git a/master/html/_sources/functionlib/dsplib/addfusion.rst.txt b/master/html/_sources/functionlib/dsplib/addfusion.rst.txt new file mode 100644 index 0000000..7df3026 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/addfusion.rst.txt @@ -0,0 +1,274 @@ +Addfusion +================= + + +带参数alpha的逐元素求和 + +.. math:: + + output_i = in0_i + in1_i*alpha + +输入: + - **in0** - 第一个输入数据地址。 + - **in1** - 第二个输入数据地址。 + - **alpha** - 乘法因子。 + - **size** - 计算长度。 + - **core_mask** - 核掩码。 + +输出: + - **out** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + + +**共享存储版本:** + +.. c:function:: void i8_addext_s(int8_t *in0, int8_t *in1, int8_t alpha, int8_t *out, int size, int core_mask) +.. c:function:: void i16_addext_s(int16_t *in0, int16_t *in1, int16_t alpha, int16_t *out, int size, int core_mask) +.. c:function:: void i32_addext_s(int32_t *in0, int32_t *in1, int32_t alpha, int32_t *out, int size, int core_mask) +.. c:function:: void hp_addext_s(half *in0, half *in1, half alpha, half *out, int size, int core_mask) +.. c:function:: void fp_addext_s(float *in0, float *in1, float alpha, float *out, int size, int core_mask) +.. c:function:: void dp_addext_s(double *in0, double *in1, double alpha, double *out, int size, int core_mask) +.. c:function:: void c64_addext_s(float *in0, float *in1, float alpha, float *out, int size, int core_mask) +.. c:function:: void c128_addext_s(double *in0, double *in1, double alpha, double *out, int size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *in0 = (float *)0xA0000000; //input在DDR空间 + float *in1 = (float *)0xB0000000; + bool *out = (bool *)0xC0000000; + float alpha = 0.1; + int size = 1000; + int core_mask = 0xff; + fp_addext_s(in0, in1, alpha, out, size, core_mask); + return 0; + } + + + +**私有存储版本:** + +.. c:function:: void i8_addext_p(int8_t *in0, int8_t *in1, int8_t alpha, int8_t *out, int size) +.. c:function:: void i16_addext_p(int16_t *in0, int16_t *in1, int16_t alpha, int16_t *out, int size) +.. c:function:: void i32_addext_p(int32_t *in0, int32_t *in1, int32_t alpha, int32_t *out, int size) +.. c:function:: void hp_addext_p(half *in0, half *in1, half alpha, half *out, int size) +.. c:function:: void fp_addext_p(float *in0, float *in1, float alpha, float *out, int size) +.. c:function:: void dp_addext_p(double *in0, double *in1, double alpha, double *out, int size) +.. c:function:: void c64_addext_p(float *in0, float *in1, float alpha, float *out, int size) +.. c:function:: void c128_addext_p(double *in0, double *in1, double alpha, double *out, int size) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *in0 = (float *)0x10000000; //input在L2空间 + float *in1 = (float *)0x10001000; + bool *out = (bool *)0xC0000000; + int length = 1000; + fp_addext_s(in0, in1, alpha, out, size); + return 0; + } + + + +逐元素求和并计算relu激活函数 + +.. math:: + + out_i = max(0, in0_i + in1_i) + +输入: + - **in0** - 第一个输入数据地址。 + - **in1** - 第二个输入数据地址。 + - **size** - 计算长度。 + - **core_mask** - 核掩码。 + +输出: + - **out** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_addrelu_s(int8_t *in0, int8_t *in1, int8_t *out, int size, int core_mask) +.. c:function:: void i16_addrelu_s(int16_t *in0, int16_t *in1, int16_t *out, int size, int core_mask) +.. c:function:: void i32_addrelu_s(int32_t *in0, int32_t *in1, int32_t *out, int size, int core_mask) +.. c:function:: void hp_addrelu_s(half *in0, half *in1, half *out, int size, int core_mask) +.. c:function:: void fp_addrelu_s(float *in0, float *in1, float *out, int size, int core_mask) +.. c:function:: void dp_addrelu_s(double *in0, double *in1, double *out, int size, int core_mask) +.. c:function:: void c64_addrelu_s(float *in0, float *in1, float *out, int size, int core_mask) +.. c:function:: void c128_addrelu_s(double *in0, double *in1, double *out, int size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *in0 = (float *)0xA0000000; //input在DDR空间 + float *in1 = (float *)0xB0000000; + bool *out = (bool *)0xC0000000; + int size = 1000; + int core_mask = 0xff; + fp_addrelu_s(in0, in1, out, size, core_mask); + return 0; + } + + + +**私有存储版本:** + +.. c:function:: void i8_addrelu_p(int8_t *in0, int8_t *in1, int8_t *out, int size) +.. c:function:: void i16_addrelu_p(int16_t *in0, int16_t *in1, int16_t *out, int size) +.. c:function:: void i32_addrelu_p(int32_t *in0, int32_t *in1, int32_t *out, int size) +.. c:function:: void hp_addrelu_p(half *in0, half *in1, half *out, int size) +.. c:function:: void fp_addrelu_p(float *in0, float *in1, float *out, int size) +.. c:function:: void dp_addrelu_p(double *in0, double *in1, double *out, int size) +.. c:function:: void c64_addrelu_p(float *in0, float *in1, float *out, int size) +.. c:function:: void c128_addrelu_p(double *in0, double *in1, double *out, int size) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *in0 = (float *)0x10000000; //input在L2空间 + float *in1 = (float *)0x10001000; + bool *out = (bool *)0xC0000000; + int size = 1000; + fp_addrelu_s(in0, in1, out, size); + return 0; + } + + + +逐元素求和并计算relu6激活函数 + +.. math:: + + out_i = min(max(0, in0_i + in1_i), 6) + +输入: + - **in0** - 第一个输入数据地址。 + - **in1** - 第二个输入数据地址。 + - **size** - 计算长度。 + - **core_mask** - 核掩码。 + +输出: + - **out** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + + + +**共享存储版本:** + +.. c:function:: void i8_addrelu6_s(int8_t *in0, int8_t *in1, int8_t *out, int size, int core_mask) +.. c:function:: void i16_addrelu6_s(int16_t *in0, int16_t *in1, int16_t *out, int size, int core_mask) +.. c:function:: void i32_addrelu6_s(int32_t *in0, int32_t *in1, int32_t *out, int size, int core_mask) +.. c:function:: void hp_addrelu6_s(half *in0, half *in1, half *out, int size, int core_mask) +.. c:function:: void fp_addrelu6_s(float *in0, float *in1, float *out, int size, int core_mask) +.. c:function:: void dp_addrelu6_s(double *in0, double *in1, double *out, int size, int core_mask) +.. c:function:: void c64_addrelu6_s(float *in0, float *in1, float *out, int size, int core_mask) +.. c:function:: void c128_addrelu6_s(double *in0, double *in1, double *out, int size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *in0 = (float *)0xA0000000; //input在DDR空间 + float *in1 = (float *)0xB0000000; + bool *out = (bool *)0xC0000000; + int size = 1000; + int core_mask = 0xff; + fp_addrelu6_s(in0, in1, out, size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_addrelu6_p(int8_t *in0, int8_t *in1, int8_t *out, int size) +.. c:function:: void i16_addrelu6_p(int16_t *in0, int16_t *in1, int16_t *out, int size) +.. c:function:: void i32_addrelu6_p(int32_t *in0, int32_t *in1, int32_t *out, int size) +.. c:function:: void hp_addrelu6_p(half *in0, half *in1, half *out, int size) +.. c:function:: void fp_addrelu6_p(float *in0, float *in1, float *out, int size) +.. c:function:: void dp_addrelu6_p(double *in0, double *in1, double *out, int size) +.. c:function:: void c64_addrelu6_p(float *in0, float *in1, float *out, int size) +.. c:function:: void c128_addrelu6_p(double *in0, double *in1, double *out, int size) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *in0 = (float *)0x10000000; //input在L2空间 + float *in1 = (float *)0x10001000; + bool *out = (bool *)0xC0000000; + int size = 1000; + fp_addrelu6_s(in0, in1, out, size); + return 0; + } + + \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/addgrad.rst.txt b/master/html/_sources/functionlib/dsplib/addgrad.rst.txt new file mode 100644 index 0000000..3c6d28a --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/addgrad.rst.txt @@ -0,0 +1,88 @@ +Addgrad +================= + + +逐元素计算加法梯度 + +.. math:: + + dx1 = \frac{\partial L}{\partial X1} = \frac{\partial L}{\partial Y} * 1 = \frac{\partial L}{\partial Y}\\ + dx2 = \frac{\partial L}{\partial X2} = \frac{\partial L}{\partial Y} * 1 = \frac{\partial L}{\partial Y} + +输入: + - **dy** - dy数据地址。 + - **dx1_dim** - x1的维度信息。 + - **dx2_dim** - x2的维度信息。 + - **dy_dims** - dy的维度信息。 + - **num_dims** - 维度数 + - **core_mask** - 核掩码。 + +输出: + - **dx1** - dx1的数据地址。 + - **dx2** - dx2的数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_addgrad_s(float* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, float* dx1, float* dx2, int core_mask) +.. c:function:: void hp_addgrad_s(half* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, half* dx1, half* dx2, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *dy = (float *)0xA0000000; //input在DDR空间 + float *dx1 = (float *)0xB0000000; //dx1 + float *dx2 = (float *)0xC0000000; //dxx + int dx1_dims[] = {13, 183, 47}; // 2x2 + int dx2_dims[] = {1, 183, 47}; // 2x2 + int dy_dims[] = {13, 183, 47}; // 2x2 + int num_dims = 3; + int core_mask = 0xff; + fp_addgrad_s(dy, dx1_dims, dx2_dims, dy_dims, num_dims, dx1, dx2, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_addgrad_p(float* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, float* dx1, float* dx2) +.. c:function:: void hp_addgrad_p(half* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, half* dx1, half* dx2) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *dy = (float *)0x10000000; //input在L2空间 + float *dx1 = (float *)0x10001000; //dx1 + float *dx2 = (float *)0x10002000; //dxx + int dx1_dims[] = {13, 183, 47}; // 2x2 + int dx2_dims[] = {1, 183, 47}; // 2x2 + int dy_dims[] = {13, 183, 47}; // 2x2 + int num_dims = 3; + fp_addgrad_p(dy, dx1_dims, dx2_dims, dy_dims, num_dims, dx1, dx2); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/addn.rst.txt b/master/html/_sources/functionlib/dsplib/addn.rst.txt new file mode 100644 index 0000000..cc0bab6 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/addn.rst.txt @@ -0,0 +1,96 @@ +AddN +================= + + + +对多个输入张量进行 **逐元素相加**,并将结果输出。 +当前实现为二输入版本,可扩展用于多输入逐元素求和的场景。 + +数学表达式为: + +.. math:: + + dst_i = src0_i + src1_i + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **length** - 计算长度(元素个数)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32``、``fp64``、``int8``、``int16``、``int32``、``cplx64``、``cplx128`` 类型 + - MT7004 支持 ``fp16``、``fp32``、``int16``、``int32``、``cplx64`` 类型 + - 所有输入与输出张量需具有相同的长度与数据布局 + +**共享存储版本:** + +.. c:function:: void fp_addn_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_addn_s(double* input0, double* input1, double* output, int length, int core_mask) +.. c:function:: void i8_addn_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_addn_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_addn_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void c64_addn_s(cplx64* input0, cplx64* input1, cplx64* output, int length, int core_mask) +.. c:function:: void c128_addn_s(cplx128* input0, cplx128* input1, cplx128* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; + float *input1 = (float *)0xA0010000; + float *output = (float *)0xC0000000; + + int length = 1024; + int core_mask = 0xff; + + fp_addn_s(input0, input1, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_addn_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_addn_p(double* input0, double* input1, double* output, int length) +.. c:function:: void i8_addn_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_addn_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_addn_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void c64_addn_p(cplx64* input0, cplx64* input1, cplx64* output, int length) +.. c:function:: void c128_addn_p(cplx128* input0, cplx128* input1, cplx128* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; // L2 空间 + float *input1 = (float *)0x10814000; + float *output = (float *)0x10820000; + + int length = 1024; + + fp_addn_p(input0, input1, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/affine.rst.txt b/master/html/_sources/functionlib/dsplib/affine.rst.txt new file mode 100644 index 0000000..b562f00 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/affine.rst.txt @@ -0,0 +1,276 @@ +Affine +================= +将来自输入数据的若干行按照一个“context”(偏移集合)拼接/裁切成一个中间矩阵,再对该中间矩阵与权重做一次矩阵乘和偏置加法操作,最后应用激活函数(如果指定),得到输出结果。 + +该算子支持全量运行和增量运行两种模式,并维护了上一次全窗口的输出(previous\_output)以支持增量更新。 + +输入: + - **input0** - 输入数据张量地址。 + - **input1** - 权重矩阵地址。 + - **input2** - 偏置向量地址。 + - **input0_shape** - 输入数据形状数组,长度为3,第一维值为1。 + - **input1_shape** - 权重矩阵形状数组,长度为3,第一维值为1。 + - **input2_shape** - 偏置向量形状数组,长度为3,第一维值为1。 + - **output_shape** - 输出张量形状数组,长度为3,第一维值为1。 + - **context** - 上下文索引数组,值递增。 + - **context_size** - 上下文大小。 + - **output_dim** - 输出维度,即输入张量最后一维大小乘以上下文大小。 + - **activation_type** - 激活函数类型,0-8。 + - **is_full_run** - 全量运行标志指针,所指值为1时为全量更新(运行后修改为0),为0时为增量更新。 + - **full_input** - 全量输入缓冲区地址。 + - **full_input_shape** - 全量输入形状数组。 + - **increment_input** - 增量输入缓冲区地址。 + - **increment_input_shape** - 增量输入形状数组。 + - **increment_output** - 增量输出缓冲区地址。 + - **increment_output_shape** - 增量输出形状数组。 + - **previous_output** - 先前输出缓冲区地址。 + - **previous_output_shape** - 先前输出形状数组。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 仿射变换结果张量地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:int8, fp32 + - MT7004 支持的数据类型:fp16, fp32 + +**激活函数类型定义:** + +.. code-block:: c + :linenos: + + #define ActivationType_NO_ACTIVATION 0 // 无激活函数 + #define ActivationType_RELU 1 // ReLU激活函数 + #define ActivationType_RELU6 2 // ReLU6激活函数 + #define ActivationType_SIGMOID 3 // Sigmoid激活函数 + #define ActivationType_TANH 4 // Tanh激活函数 + #define ActivationType_SWISH 5 // Swish激活函数 + #define ActivationType_HSWISH 6 // Hard Swish激活函数 + #define ActivationType_HSIGMOID 7 // Hard Sigmoid激活函数 + #define ActivationType_SOFTPLUS 8 // Softplus激活函数 + +**激活函数数学公式:** + +- **ReLU**: :math:`f(x) = \max(0, x)` +- **ReLU6**: :math:`f(x) = \min(\max(0, x), 6)` +- **Sigmoid**: :math:`f(x) = \frac{1}{1 + e^{-x}}` +- **Tanh**: :math:`f(x) = \frac{e^x - e^{-x}}{e^x + e^{-x}}` +- **Swish**: :math:`f(x) = x \cdot \sigma(x) = \frac{x}{1 + e^{-x}}` +- **Hard Swish**: :math:`f(x) = x \cdot \frac{\min(\max(x + 3, 0), 6)}{6}` +- **Hard Sigmoid**: :math:`f(x) = \frac{\min(\max(x + 3, 0), 6)}{6}` +- **Softplus**: :math:`f(x) = ln(1 + e^x)` + +**参数数组结构:** + +.. code-block:: c + :linenos: + + long long params[21]; + params[0] = (long long)input0; // 输入数据张量地址 + params[1] = (long long)input1; // 权重矩阵地址 + params[2] = (long long)input2; // 偏置向量地址 + params[3] = (long long)output; // 输出张量地址 + params[4] = (long long)input0_shape; // 输入数据形状数组 + params[5] = (long long)input1_shape; // 权重矩阵形状数组 + params[6] = (long long)input2_shape; // 偏置向量形状数组 + params[7] = (long long)output_shape; // 输出张量形状数组 + params[8] = (long long)context; // 上下文索引数组 + params[9] = (long long)context_size; // 上下文大小 + params[10] = (long long)output_dim; // 输出维度 + params[11] = (long long)activation_type; // 激活函数类型 + params[12] = (long long)&is_full_run; // 全量运行标志指针 + params[13] = (long long)full_input; // 全量输入缓冲区地址 + params[14] = (long long)full_input_shape; // 全量输入形状数组 + params[15] = (long long)increment_input; // 增量输入缓冲区地址 + params[16] = (long long)increment_input_shape; // 增量输入形状数组 + params[17] = (long long)increment_output; // 增量输出缓冲区地址 + params[18] = (long long)increment_output_shape; // 增量输出形状数组 + params[19] = (long long)previous_output; // 先前输出缓冲区地址 + params[20] = (long long)previous_output_shape; // 先前输出形状数组 + +**共享存储版本:** + +.. c:function:: void i8_affine_s(long long* params, int core_mask) +.. c:function:: void fp_affine_s(long long* params, int core_mask) +.. c:function:: void hp_affine_s(long long* params, int core_mask) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 87 + + // FT78NE 多核示例 + #include + #include + #include + #include + + void test_fp_affine_s(int a, int b, int c, int o, int activation_type, int full_run, int core_mask) { + int i = 0, j = 0; + srand(time(0)); + + int core_id = DNUM; + int logic_core_id = GetLogicCoreId(core_mask, core_id); + int num = GetCoreNum(core_mask); + + int is_full_run = full_run; + int context[] = {-1, 0, 1, 2}; + int context_size = c; + int output_dim = b * c; + + // 形状定义 + int input0_shape[3] = {1, a, b}; + int input1_shape[3] = {1, b * c, o}; + int input2_shape[3] = {1, a - c + 1, o}; + int output_shape[3] = {1, a - c + 1, o}; + + // 中间缓冲区形状 + int full_input_shape[3] = {1, input0_shape[1] - (context[context_size - 1] - context[0]), output_dim}; + int increment_input_shape[3] = {1, 1, output_dim}; + int increment_output_shape[3] = {1, 1, output_shape[2]}; + int previous_output_shape[3] = {1, output_shape[1], output_shape[2]}; + + // 内存分配 + float* input0 = (float*)(0xA0400000); + float* input1 = (float*)(0xA0400000 + 0x100000); + float* input2 = (float*)(0xA0400000 + 0x200000); + float* output = (float*)(0xA0400000 + 0x300000); + float* full_input = (float*)(0xA0400000 + 0x400000); + float* increment_input = (float*)(0xA0400000 + 0x500000); + float* increment_output = (float*)(0xA0400000 + 0x600000); + float* previous_output = (float*)(0xA0400000 + 0x700000); + + // 初始化数据 + if (logic_core_id == 0) { + int input0_len = input0_shape[0] * input0_shape[1] * input0_shape[2]; + int input1_len = input1_shape[0] * input1_shape[1] * input1_shape[2]; + int input2_len = input2_shape[0] * input2_shape[1] * input2_shape[2]; + + for (i = 0; i < input0_len; i++) { + input0[i] = ((float)rand() / RAND_MAX) * 2 - 1; + } + for (i = 0; i < input1_len; i++) { + input1[i] = ((float)rand() / RAND_MAX) * 2 - 1; + } + for (i = 0; i < input2_shape[2]; i++) { + input2[i] = ((float)rand() / RAND_MAX) * 2 - 1; + for (j = 1; j < input2_shape[1]; j++) { + input2[i + j * input2_shape[2]] = input2[i]; + } + } + } + + // 准备参数数组 + long long params[21]; + params[0] = (long long)input0; + params[1] = (long long)input1; + params[2] = (long long)input2; + params[3] = (long long)output; + params[4] = (long long)input0_shape; + params[5] = (long long)input1_shape; + params[6] = (long long)input2_shape; + params[7] = (long long)output_shape; + params[8] = (long long)context; + params[9] = (long long)context_size; + params[10] = (long long)output_dim; + params[11] = (long long)activation_type; + params[12] = (long long)&is_full_run; + params[13] = (long long)full_input; + params[14] = (long long)full_input_shape; + params[15] = (long long)increment_input; + params[16] = (long long)increment_input_shape; + params[17] = (long long)increment_output; + params[18] = (long long)increment_output_shape; + params[19] = (long long)previous_output; + params[20] = (long long)previous_output_shape; + + // 执行 Affine 操作 + fp_affine_s(params, core_mask); + } + + int main(void) { + int a = 23, b = 31, c = 4, o = 29; + int activation_type = 0; // 激活函数类型 + int full_run = 1; // 全量运行标志 + int core_mask = 0xff; // 核掩码 + + test_fp_affine_s(a, b, c, o, activation_type, full_run, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_affine_p(long long* params) +.. c:function:: void fp_affine_p(long long* params) +.. c:function:: void hp_affine_p(long long* params) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 57 + + // FT78NE 单核示例 + #include + #include + + int main(void) { + // 参数设置(与共享版本类似) + int a = 32, b = 16, c = 4, o = 16; + int is_full_run = full_run; + int context[] = {-1, 0, 1, 2}; + int context_size = c; + int output_dim = b * c; + + int input0_shape[3] = {1, a, b}; + int input1_shape[3] = {1, b * c, o}; + int input2_shape[3] = {1, a - c + 1, o}; + int output_shape[3] = {1, a - c + 1, o}; + + int full_input_shape[3] = {1, input0_shape[1] - (context[context_size - 1] - context[0]), output_dim}; + int increment_input_shape[3] = {1, 1, output_dim}; + int increment_output_shape[3] = {1, 1, output_shape[2]}; + int previous_output_shape[3] = {1, output_shape[1], output_shape[2]}; + + float* input0 = (float*)(0x10810000); + float* input1 = (float*)(0x10810000 + 0x100000); + float* input2 = (float*)(0x10810000 + 0x200000); + float* output = (float*)(0x10810000 + 0x300000); + float* full_input = (float*)(0x10810000 + 0x400000); + float* increment_input = (float*)(0x10810000 + 0x500000); + float* increment_output = (float*)(0x10810000 + 0x600000); + float* previous_output = (float*)(0x10810000 + 0x700000); + + // 准备参数数组(与共享版本相同) + long long params[21]; + params[0] = (long long)input0; + params[1] = (long long)input1; + params[2] = (long long)input2; + params[3] = (long long)output; + params[4] = (long long)input0_shape; + params[5] = (long long)input1_shape; + params[6] = (long long)input2_shape; + params[7] = (long long)output_shape; + params[8] = (long long)context; + params[9] = (long long)context_size; + params[10] = (long long)output_dim; + params[11] = (long long)activation_type; + params[12] = (long long)&is_full_run; + params[13] = (long long)full_input; + params[14] = (long long)full_input_shape; + params[15] = (long long)increment_input; + params[16] = (long long)increment_input_shape; + params[17] = (long long)increment_output; + params[18] = (long long)increment_output_shape; + params[19] = (long long)previous_output; + params[20] = (long long)previous_output_shape; + + // 调用 Affine + fp_affine_p(params); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/all.rst.txt b/master/html/_sources/functionlib/dsplib/all.rst.txt new file mode 100644 index 0000000..31a70c6 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/all.rst.txt @@ -0,0 +1,97 @@ +All +================= + + +沿指定轴计算张量的逻辑与 (Logical AND)。如果轴上的所有元素都为真(非零),则输出为1.0,否则为0.0。 + +.. math:: + + Y_i = \prod_{x \in S_i} B(x) + +其中 :math:`S_i` 是输入张量中用于计算输出 :math:`Y_i` 的切片, :math:`B(x)` 是一个辅助函数: + +.. math:: + + B(x)=\begin{cases} 1, & \text{if } x \neq 0 \\ 0, & \text{if } x = 0 \end{cases} + +输入: + - **outer_size** - 规约轴之前的所有维度大小的乘积。 + - **inner_size** - 规约轴之后的所有维度大小的乘积。 + - **axis_size** - 规约轴本身的大小。 + - **src_data** - 输入数据地址。 + - **core_mask** - 核掩码。 + +输出: + - **dst_data** - 输出数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + + +**共享存储版本:** + +.. c:function:: void i8_reduceall_s(int outer_size, int inner_size, int axis_size, int8_t *src_data, int8_t *dst_data, int core_mask) +.. c:function:: void i16_reduceall_s(int outer_size, int inner_size, int axis_size, int16_t *src_data, int16_t *dst_data, int core_mask) +.. c:function:: void i32_reduceall_s(int outer_size, int inner_size, int axis_size, int *src_data, int *dst_data, int core_mask) +.. c:function:: void hp_reduceall_s(int outer_size, int inner_size, int axis_size, half *src_data, half *dst_data, int core_mask) +.. c:function:: void fp_reduceall_s(int outer_size, int inner_size, int axis_size, float *src_data, float *dst_data, int core_mask) +.. c:function:: void dp_reduceall_s(int outer_size, int inner_size, int axis_size, double *src_data, double *dst_data, int core_mask) +.. c:function:: void c64_reduceall_s(int outer_size, int inner_size, int axis_size, float *src_data, float *dst_data, int core_mask) +.. c:function:: void c128_reduceall_s(int outer_size, int inner_size, int axis_size, double *src_data, double *dst_data, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *src_data = (float *)0xA0000000; //input在DDR空间 + float *dst_data = (float *)0xB0000000; //output + int outer_size = 2; + int inner_size = 4; + int axis_size = 3; + int core_mask = 0xff; + fp_reduceall_s(outer_size, inner_size, axis_size, src_data, dst_data, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_reduceall_p(int outer_size, int inner_size, int axis_size, int8_t *src_data, int8_t *dst_data) +.. c:function:: void i16_reduceall_p(int outer_size, int inner_size, int axis_size, int16_t *src_data, int16_t *dst_data) +.. c:function:: void i32_reduceall_p(int outer_size, int inner_size, int axis_size, int *src_data, int *dst_data) +.. c:function:: void hp_reduceall_p(int outer_size, int inner_size, int axis_size, half *src_data, half *dst_data) +.. c:function:: void fp_reduceall_p(int outer_size, int inner_size, int axis_size, float *src_data, float *dst_data) +.. c:function:: void dp_reduceall_p(int outer_size, int inner_size, int axis_size, double *src_data, double *dst_data) +.. c:function:: void c64_reduceall_p(int outer_size, int inner_size, int axis_size, float *src_data, float *dst_data) +.. c:function:: void c128_reduceall_p(int outer_size, int inner_size, int axis_size, double *src_data, double *dst_data) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *src_data = (float *)0x10000000; //input在DDR空间 + float *dst_data = (float *)0x10001000; //output + int outer_size = 2; + int inner_size = 4; + int axis_size = 3; + fp_reduceall_p(outer_size, inner_size, axis_size, src_data, dst_data); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/allgather.rst.txt b/master/html/_sources/functionlib/dsplib/allgather.rst.txt new file mode 100644 index 0000000..bddcfa7 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/allgather.rst.txt @@ -0,0 +1,105 @@ +AllGather +================= + + + +将各输入分片的数据进行 **全量收集(All-Gather)**, +并在输出中按指定维度进行拼接复制。 +该算子通常用于并行/多核场景中,将每个 rank 的本地数据 +收集成一个完整的输出张量。 + +算子行为可概括为两步: +1. 将 ``input`` 中每个 ``input_rank`` 对应的数据块拷贝到 ``output`` 的首段连续区域; +2. 将首段完整数据复制到其余 ``output_rank - 1`` 个位置,形成完整聚合结果。 + +输入: + - **input** - 输入数据地址,按 ``[input_rank, data_size]`` 形式连续存储。 + - **input_rank** - 输入 rank 数量(通常对应参与 AllGather 的并行单元数)。 + - **output_rank** - 输出 rank 数量(通常等于或大于 ``input_rank``)。 + - **data_size** - 单个 rank 对应的数据元素个数。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - AllGather 结果输出地址, + 数据按 ``[output_rank, input_rank, data_size]`` 的逻辑顺序存储。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32``、``fp64``、``int8``、``int16``、``int32``、``cplx64``、``cplx128`` 类型 + - MT7004 支持 ``fp16``、``fp32``、``int16``、``int32``、``cplx64`` 类型 + - 当前实现基于 DMA进行数据搬运 + - 输入与输出地址需保证足够的连续空间以容纳聚合结果 + +**共享存储版本:** + +.. c:function:: void fp_allgather_s(float* input, float* output, int input_rank, int output_rank, int data_size, int core_mask) +.. c:function:: void dp_allgather_s(double* input, double* output, int input_rank, int output_rank, int data_size, int core_mask) +.. c:function:: void i8_allgather_s(int8_t* input, int8_t* output, int input_rank, int output_rank, int data_size, int core_mask) +.. c:function:: void i16_allgather_s(int16_t* input, int16_t* output, int input_rank, int output_rank, int data_size, int core_mask) +.. c:function:: void i32_allgather_s(int32_t* input, int32_t* output, int input_rank, int output_rank, int data_size, int core_mask) +.. c:function:: void c64_allgather_s(cplx64* input, cplx64* output, int input_rank, int output_rank, int data_size, int core_mask) +.. c:function:: void c128_allgather_s(cplx128* input, cplx128* output, int input_rank, int output_rank, int data_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14-16 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + float *output = (float *)0xC0000000; + + int input_rank = 4; + int output_rank = 4; + int data_size = 256; + int core_mask = 0xff; + + fp_allgather_s(input, output, + input_rank, output_rank, + data_size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_allgather_p(float* input, float* output, int input_rank, int output_rank, int data_size) +.. c:function:: void dp_allgather_p(double* input, double* output, int input_rank, int output_rank, int data_size) +.. c:function:: void i8_allgather_p(int8_t* input, int8_t* output, int input_rank, int output_rank, int data_size) +.. c:function:: void i16_allgather_p(int16_t* input, int16_t* output, int input_rank, int output_rank, int data_size) +.. c:function:: void i32_allgather_p(int32_t* input, int32_t* output, int input_rank, int output_rank, int data_size) +.. c:function:: void c64_allgather_p(cplx64* input, cplx64* output, int input_rank, int output_rank, int data_size) +.. c:function:: void c128_allgather_p(cplx128* input, cplx128* output, int input_rank, int output_rank, int data_size) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13-15 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; // L2 空间 + float *output = (float *)0x10820000; + + int input_rank = 4; + int output_rank = 4; + int data_size = 256; + + fp_allgather_p(input, output, + input_rank, output_rank, + data_size); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/applymomentum.rst.txt b/master/html/_sources/functionlib/dsplib/applymomentum.rst.txt index 4c291ab..fe1f1f3 100644 --- a/master/html/_sources/functionlib/dsplib/applymomentum.rst.txt +++ b/master/html/_sources/functionlib/dsplib/applymomentum.rst.txt @@ -48,7 +48,7 @@ ApplyMomentum .. code-block:: c :linenos: - :emphasize-lines: 15 + :emphasize-lines: 15-17 // FT78NE 多核示例 #include @@ -80,7 +80,7 @@ ApplyMomentum .. code-block:: c :linenos: - :emphasize-lines: 13 + :emphasize-lines: 13-15 // MT7004 单核示例 #include diff --git a/master/html/_sources/functionlib/dsplib/argmax.rst.txt b/master/html/_sources/functionlib/dsplib/argmax.rst.txt new file mode 100644 index 0000000..bc7c7d3 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/argmax.rst.txt @@ -0,0 +1,111 @@ +Argmax +================= + + +沿指定轴查找最大 `topk` 个值的索引。当 `topk=1` 时,该算子等价于 `ArgMax`。 + +.. math:: + + Y_i = \underset{k}{\operatorname{argmax}} (X_{slice_i}) + +其中 :math:`X_{slice_i}` 是输入张量中沿指定轴的一个切片,函数返回该切片中最大值的索引 :math:`k`。 + +输入: + - **input** - 输入数据地址。 + - **output** - 输出索引的数据地址,数据类型通常为int32。 + - **output_value** - (可选) 输出值的数据地址。 + - **in_shape** - 输入张量的维度信息数组。 + - **in_strides** - 输入张量的步长信息数组。 + - **out_strides** - 输出张量的步长信息数组。 + - **arg_elements** - 用于存放候选值的临时工作空间地址。 + - **index** - 用于存放候选索引的临时工作空间地址。 + - **topk** - 需要查找的最大值的数量。设置为1以执行ArgMax操作。 + - **out_value** - 是否返回数值的标志。若为非0,则 `output_value` 必须提供有效地址。 + - **input_shape_size** - 输入张量的维度数 (即 `in_shape` 数组的长度)。 + - **axis** - 执行查找操作的轴。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 存储索引的输出张量。 + - **output_value** - 如果 `return_values` 为 true,则此处存储找到的值。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_argmax_s(float *input, void *output, float *output_value, int32_t *in_shape, int *in_strides, int *out_strides, float *arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis, int core_mask) +.. c:function:: void hp_argmax_s(half *input, void *output, half *output_value, int32_t *in_shape, int *in_strides, int *out_strides, half *arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 21 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input在DDR空间 + int *output = (int *)0xB0000000; // output indices + float *output_value = (float *)0xC0000000; // output values + float *arg_elements = (float *)0xD0000000; // temp workspace 1 + int *index = (int *)0xE0000000; // temp workspace 2 + + int in_shape[] = {2, 3, 4}; // input shape: (2, 3, 4) + int in_strides[] = {12, 4, 1}; // input strides for contiguous layout + int out_strides[] = {4, 1}; // output strides, shape is (2, 4) + int input_shape_size = 3; + + int axis = 1; // 沿第1轴操作 + int topk = 1; // ArgMax + int out_value = 1; // 同时返回值 + int core_mask = 0xff; + + fp_argmax_s(input, output, output_value, in_shape, in_strides, out_strides, arg_elements, index, topk, out_value, input_shape_size, axis, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_argmax_p(float *input, void *output, float *output_value, int32_t *in_shape, int *in_strides, int *out_strides, float *arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis) +.. c:function:: void hp_argmax_p(half *input, void *output, half *output_value, int32_t *in_shape, int *in_strides, int *out_strides, half *arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 20-21 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0x10001000; // input在DDR空间 + int *output = (int *)0x10002000; // output indices + float *output_value = (float *)0x10003000; // output values + float *arg_elements = (float *)0x10004000; // temp workspace 1 + int *index = (int *)0x10005000; // temp workspace 2 + + int in_shape[] = {2, 3, 4}; // input shape: (2, 3, 4) + int in_strides[] = {12, 4, 1}; // input strides for contiguous layout + int out_strides[] = {4, 1}; // output strides, shape is (2, 4) + int input_shape_size = 3; + + int axis = 1; // 沿第1轴操作 + int topk = 1; // ArgMax + int out_value = 1; // 同时返回值 + + fp_argmax_p(input, output, output_value, in_shape, in_strides, out_strides, + arg_elements, index, topk, out_value, input_shape_size, axis); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/argmin.rst.txt b/master/html/_sources/functionlib/dsplib/argmin.rst.txt new file mode 100644 index 0000000..e773013 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/argmin.rst.txt @@ -0,0 +1,112 @@ +Argmin +================= + + +沿指定轴查找最小 `topk` 个值的索引。当 `topk=1` 时,该算子等价于 `ArgMin`。 + +.. math:: + + Y_i = \underset{k}{\operatorname{argmin}} (X_{slice_i}) + +其中 :math:`X_{slice_i}` 是输入张量中沿指定轴的一个切片,函数返回该切片中最小值的索引 :math:`k`。 + +输入: + - **input** - 输入数据地址。 + - **output** - 输出索引的数据地址,数据类型通常为int32。 + - **output_value** - (可选) 输出值的数据地址。 + - **in_shape** - 输入张量的维度信息数组。 + - **in_strides** - 输入张量的步长信息数组。 + - **out_strides** - 输出张量的步长信息数组。 + - **arg_elements** - 用于存放候选值的临时工作空间地址。 + - **index** - 用于存放候选索引的临时工作空间地址。 + - **topk** - 需要查找的最小值的数量。设置为1以执行ArgMin操作。 + - **out_value** - 是否返回数值的标志。若为非0,则 `output_value` 必须提供有效地址。 + - **input_shape_size** - 输入张量的维度数 (即 `in_shape` 数组的长度)。 + - **axis** - 执行查找操作的轴。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 存储索引的输出张量。 + - **output_value** - 如果 `return_values` 为 true,则此处存储找到的值。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_argmin_s(float* input, void* output, float* output_value, int* in_shape, int* in_strides, int* out_strides, float* arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis, int core_mask) +.. c:function:: void hp_argmin_s(half* input, void* output, half* output_value, int* in_shape, int* in_strides, int* out_strides, half* arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 21-22 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input在DDR空间 + int *output = (int *)0xB0000000; // output indices + float *output_value = (float *)0xC0000000; // output values + float *arg_elements = (float *)0xD0000000; // temp workspace 1 + int *index = (int *)0xE0000000; // temp workspace 2 + + int in_shape[] = {2, 3, 4}; // input shape: (2, 3, 4) + int in_strides[] = {12, 4, 1}; // input strides for contiguous layout + int out_strides[] = {4, 1}; // output strides, shape is (2, 4) + int input_shape_size = 3; + + int axis = 1; // 沿第1轴操作 + int topk = 1; // ArgMax + int out_value = 1; // 同时返回值 + int core_mask = 0xff; + + fp_argmin_s(input, output, output_value, in_shape, in_strides, out_strides, + arg_elements, index, topk, out_value, input_shape_size, axis, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_argmin_p(float *input, void *output, float *output_value, int32_t *in_shape, int *in_strides, int *out_strides, float *arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis) +.. c:function:: void hp_argmin_p(half *input, void *output, half *output_value, int32_t *in_shape, int *in_strides, int *out_strides, half *arg_elements, int* index, int topk, int out_value, int input_shape_size, int axis) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 20-21 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0x10001000; // input在DDR空间 + int *output = (int *)0x10002000; // output indices + float *output_value = (float *)0x10003000; // output values + float *arg_elements = (float *)0x10004000; // temp workspace 1 + int *index = (int *)0x10005000; // temp workspace 2 + + int in_shape[] = {2, 3, 4}; // input shape: (2, 3, 4) + int in_strides[] = {12, 4, 1}; // input strides for contiguous layout + int out_strides[] = {4, 1}; // output strides, shape is (2, 4) + int input_shape_size = 3; + + int axis = 1; // 沿第1轴操作 + int topk = 1; // ArgMin + int out_value = 1; // 同时返回值 + + fp_argmin_p(input, output, output_value, in_shape, in_strides, out_strides, + arg_elements, index, topk, out_value, input_shape_size, axis); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/assign.rst.txt b/master/html/_sources/functionlib/dsplib/assign.rst.txt new file mode 100644 index 0000000..101ec19 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/assign.rst.txt @@ -0,0 +1,83 @@ +Assign +================= + +将输入张量的值复制到输出张量中,实现张量赋值操作。 + +.. math:: + + dst_i = src_i + +输入: + - **src** - 输入数据地址。 + - **length** - 数组长度(元素个数)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst** - 输出数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持的数据类型:fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_assign_s(int8_t* src, int8_t* dst, int length, int core_mask) +.. c:function:: void i16_assign_s(int16_t* src, int16_t* dst, int length, int core_mask) +.. c:function:: void i32_assign_s(int32_t* src, int32_t* dst, int length, int core_mask) +.. c:function:: void fp_assign_s(float* src, float* dst, int length, int core_mask) +.. c:function:: void dp_assign_s(double* src, double* dst, int length, int core_mask) +.. c:function:: void c64_assign_s(float* src, float* dst, int length, int core_mask) +.. c:function:: void c128_assign_s(double* src, double* dst, int length, int core_mask) +.. c:function:: void hp_assign_s(half* src, half* dst, int length, int core_mask) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *src = (float *)0xA0000000; // src在DDR空间 + float *dst = (float *)0xB0000000; + int length = 1000; + int core_mask = 0xff; + fp_assign_s(src, dst, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_assign_p(int8_t* src, int8_t* dst, int length) +.. c:function:: void i16_assign_p(int16_t* src, int16_t* dst, int length) +.. c:function:: void i32_assign_p(int32_t* src, int32_t* dst, int length) +.. c:function:: void fp_assign_p(float* src, float* dst, int length) +.. c:function:: void dp_assign_p(double* src, double* dst, int length) +.. c:function:: void c64_assign_p(float* src, float* dst, int length) +.. c:function:: void c128_assign_p(double* src, double* dst, int length) +.. c:function:: void hp_assign_p(half* src, half* dst, int length) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *src = (half *)0x10000000; // src在L2空间 + half *dst = (half *)0x10004000; + int length = 1000; + hp_assign_p(src, dst, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/assignadd.rst.txt b/master/html/_sources/functionlib/dsplib/assignadd.rst.txt new file mode 100644 index 0000000..fc0b4ec --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/assignadd.rst.txt @@ -0,0 +1,88 @@ +AssignAdd +================= + +将输入张量的值累加到输出张量中,实现张量原地加法操作。 + +.. math:: + + output_i = output_i + input_i + +输入: + - **input** - 输入数据地址。 + - **output** - 输出数据地址(同时作为输入和输出)。 + - **length** - 数组长度(元素个数)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 原地写回累加结果。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持的数据类型:fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_assignadd_s(int8_t* input, int8_t* output, int length, int core_mask) +.. c:function:: void i16_assignadd_s(int16_t* input, int16_t* output, int length, int core_mask) +.. c:function:: void i32_assignadd_s(int32_t* input, int32_t* output, int length, int core_mask) +.. c:function:: void fp_assignadd_s(float* input, float* output, int length, int core_mask) +.. c:function:: void dp_assignadd_s(double* input, double* output, int length, int core_mask) +.. c:function:: void c64_assignadd_s(float* input, float* output, int length, int core_mask) +.. c:function:: void c128_assignadd_s(double* input, double* output, int length, int core_mask) +.. c:function:: void hp_assignadd_s(half* input, half* output, int length, int core_mask) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input在DDR空间 + float *output = (float *)0xB0000000; // output在DDR空间,同时作为输入和输出 + int length = 1000; + int core_mask = 0xff; + + // 执行 output[i] += input[i] + fp_assignadd_s(input, output, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_assignadd_p(int8_t* input, int8_t* output, int length) +.. c:function:: void i16_assignadd_p(int16_t* input, int16_t* output, int length) +.. c:function:: void i32_assignadd_p(int32_t* input, int32_t* output, int length) +.. c:function:: void fp_assignadd_p(float* input, float* output, int length) +.. c:function:: void dp_assignadd_p(double* input, double* output, int length) +.. c:function:: void c64_assignadd_p(float* input, float* output, int length) +.. c:function:: void c128_assignadd_p(double* input, double* output, int length) +.. c:function:: void hp_assignadd_p(half* input, half* output, int length) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input = (half *)0x10000000; // input在L2空间 + half *output = (half *)0x10004000; // output在L2空间,同时作为输入和输出 + int length = 1000; + + // 执行 output[i] += input[i] + hp_assignadd_p(input, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/audio_spectrogram.rst.txt b/master/html/_sources/functionlib/dsplib/audio_spectrogram.rst.txt new file mode 100644 index 0000000..798b6a3 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/audio_spectrogram.rst.txt @@ -0,0 +1,137 @@ +AudioSpectrogram +=================== + +音频频谱(Audio Spectrogram)是一种常用的音频信号特征表示方式,通过将时域音频信号转换为频域信号,提供了时间和频率的双重信息。 + +参数说明: + - **params** - 频谱图参数配置结构体指针 + - **workspace** - 工作空间缓冲区结构体指针 + - **core_mask** - 核掩码,指定参与计算的处理器核(仅共享存储版本) + +**结构体定义:** + .. code-block:: c + :linenos: + + typedef struct { + // 输入 + float* input; // 输入数据地址 + int input_len; // 输入数据长度 + + // 输出 + float* output; // 输出数据地址 + int* output_shape; // 输出数据形状 + + // 配置参数 + int pad; // 填充大小 + WindowType window_type; // 窗函数类型 + int n_fft; // FFT点数 + int hop_length; // 帧移长度 + int win_length; // 窗长度 + float power; // 功率值 + bool normalized; // 是否归一化 + bool center; // 是否中心化 + BorderType pad_mode; // 填充模式 + bool onesided; // 是否单边频谱 + } SpectrogramParam; + typedef struct { + float* fft_window; // FFT窗函数缓冲区 + float* fft_window_later; // 后续FFT窗缓冲区 + float* input_data_pad; // 填充后的输入数据 + float* input_data; // 输入数据缓冲区 + float* input_win; // 加窗输入数据 + float* exp_complex; // 复数指数缓冲区 + float* spec_f; // 频谱频率缓冲区 + float* output_onsided; // 单边输出缓冲区 + } WorkspaceParam; +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp32 + +**共享存储版本:** + +.. c:function:: void fp_audio_spectrogram_s(SpectrogramParam* params, WorkspaceParam* workspace, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 30 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float* input_data = (float*)0x81000000; + float* output = (float*)0x82000000; + int output_shape[2] = {0, 0, 0}; + SpectrogramParam* spec_params = (SpectrogramParam*)0x83200000; + WorkspaceParam* spec_workspace = (WorkspaceParam*)0x83400000; + + // 结构体 SpectrogramParam + spec_params->input = input_data; + spec_params->input_len = 4000; + spec_params->output = output; + spec_params->output_shape = output_shape; + spec_params->pad = 0; + spec_params->window_type = kHann; + spec_params->n_fft = 32; + spec_params->hop_length = 16; + spec_params->win_length = 32; + spec_params->power = 2.0f; + spec_params->normalized = false; + spec_params->center = false; + spec_params->pad_mode = kConstant; + spec_params->onesided = true; + + int core_mask = 0xff; + //spec_workspace里每个中间缓冲区指针分配地址 + //... + fp_audio_spectrogram_s(spec_params, spec_workspace, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_audio_spectrogram_p(SpectrogramParam* params, WorkspaceParam* workspace) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 29 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float* input_data = (float*)0x10810000; + float* output_mfcc = (float*)0x10820000; + SpectrogramParam* spec_params = (SpectrogramParam*)0x10830000; + WorkspaceParam* spec_workspace = (WorkspaceParam*)0x10840000; + int output_shape[2] = {0, 0, 0}; + + spec_params->input = input_data; + spec_params->input_len = 4000; + spec_params->output = output; + spec_params->output_shape = output_shape; + spec_params->pad = 0; + spec_params->window_type = kHann; + spec_params->n_fft = 32; + spec_params->hop_length = 16; + spec_params->win_length = 32; + spec_params->power = 2.0f; + spec_params->normalized = false; + spec_params->center = false; + spec_params->pad_mode = kConstant; + spec_params->onesided = true; + + //为spec_workspace里每个中间缓冲区指针分配地址 + //... + + fp_audio_spectrogram_p(spec_params, spec_workspace); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/avgpooling.rst.txt b/master/html/_sources/functionlib/dsplib/avgpooling.rst.txt new file mode 100644 index 0000000..000a134 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/avgpooling.rst.txt @@ -0,0 +1,136 @@ +Avgpooling +================= + + +对NHWC格式的输入张量执行2D平均池化,并随后进行范围裁剪(Clip)激活。 + +该算子融合了两个步骤: + +1. **平均池化 (Average Pooling)**: + +.. math:: + + \text{Pool}_{i,j} = \frac{1}{k_h \times k_w} \sum_{m=0}^{k_h-1} \sum_{n=0}^{k_w-1} \text{Input}_{i \cdot s_h + m, j \cdot s_w + n} + +2. **裁剪激活 (Clipping Activation)**: + +.. math:: + + \text{Output} = \max(\min\_val, \min(\text{Pool}, \max\_val)) + +输入: + - **input** - 输入张量的数据地址。格式: NHWC。 + - **batch** (N) - 批处理大小。 + - **in_h** (H) - 输入特征图的高度。 + - **in_w** (W) - 输入特征图的宽度。 + - **channel** (C) - 输入特征图的通道数。 + - **win_h** - 池化核的高度。 + - **win_w** - 池化核的宽度。 + - **stride_h** - 垂直方向的步长。 + - **stride_w** - 水平方向的步长。 + - **pad_top** - 上边距填充。 + - **pad_bottom** - 下边距填充。 + - **pad_left** - 左边距填充。 + - **pad_right** - 右边距填充。 + - **min_val** - 裁剪范围的最小值。 + - **max_val** - 裁剪范围的最大值。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void i8_avgpool_fusion_s(int8_t* input, int8_t* output, int batch, int in_h, int in_w, int channel, int win_h, int win_w, int stride_h, int stride_w, int pad_top, int pad_bottom, int pad_left, int pad_right, int8_t min_val, int8_t max_val, int core_mask) +.. c:function:: void fp_avgpool_fusion_s(float* input, float* output, int batch, int in_h, int in_w, int channel, int win_h, int win_w, int stride_h, int stride_w, int pad_top, int pad_bottom, int pad_left, int pad_right, float min_val, float max_val, int core_mask) +.. c:function:: void hp_avgpool_fusion_s(half* input, half* output, int batch, int in_h, int in_w, int channel, int win_h, int win_w, int stride_h, int stride_w, int pad_top, int pad_bottom, int pad_left, int pad_right, half min_val, half max_val, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 25-28 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input在DDR空间 + float *output = (float *)0xB0000000; // output + + // Shape parameters + int batch = 2; + int in_h = 224; + int in_w = 224; + int channel = 3; + + // Pooling parameters + int win_h = 3, win_w = 3; + int stride_h = 2, stride_w = 2; + int pad_top = 1, pad_bottom = 1, pad_left = 1, pad_right = 1; + + // Fusion parameters (e.g., for ReLU6 activation) + float min_val = 0.0f; + float max_val = 6.0f; + + int core_mask = 0xff; + + fp_avgpool_fusion_s(input, output, batch, in_h, in_w, channel, + win_h, win_w, stride_h, stride_w, + pad_top, pad_bottom, pad_left, pad_right, + min_val, max_val, core_mask); + return 0; + } + + + +**私有存储版本:** + +.. c:function:: void i8_avgpool_fusion_p(int8_t* input, int8_t* output, int batch, int in_h, int in_w, int channel, int win_h, int win_w, int stride_h, int stride_w, int pad_top, int pad_bottom, int pad_left, int pad_right, int8_t min_val, int8_t max_val) +.. c:function:: void fp_avgpool_fusion_p(float* input, float* output, int batch, int in_h, int in_w, int channel, int win_h, int win_w, int stride_h, int stride_w, int pad_top, int pad_bottom, int pad_left, int pad_right, float min_val, float max_val) +.. c:function:: void hp_avgpool_fusion_p(half* input, half* output, int batch, int in_h, int in_w, int channel, int win_h, int win_w, int stride_h, int stride_w, int pad_top, int pad_bottom, int pad_left, int pad_right, half min_val, half max_val) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 23-26 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // input在L2空间 + float *output = (float *)0x10100000; // output + + // Shape parameters + int batch = 2; + int in_h = 32; + int in_w = 32; + int channel = 16; + + // Pooling parameters + int win_h = 2, win_w = 2; + int stride_h = 2, stride_w = 2; + int pad_top = 0, pad_bottom = 0, pad_left = 0, pad_right = 0; + + // Fusion parameters + float min_val = -127.0f; + float max_val = 127.0f; + + fp_avgpool_fusion_p(input, output, batch, in_h, in_w, channel, + win_h, win_w, stride_h, stride_w, + pad_top, pad_bottom, pad_left, pad_right, + min_val, max_val); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/avgpoolinggrad.rst.txt b/master/html/_sources/functionlib/dsplib/avgpoolinggrad.rst.txt index 9a4093d..81c13aa 100644 --- a/master/html/_sources/functionlib/dsplib/avgpoolinggrad.rst.txt +++ b/master/html/_sources/functionlib/dsplib/avgpoolinggrad.rst.txt @@ -47,7 +47,7 @@ AvgPoolingGrad .. code-block:: c :linenos: - :emphasize-lines: 20 + :emphasize-lines: 20-25 // FT78NE 多核示例 #include @@ -88,7 +88,7 @@ AvgPoolingGrad .. code-block:: c :linenos: - :emphasize-lines: 21 + :emphasize-lines: 21-26 // MT7004 单核示例 #include diff --git a/master/html/_sources/functionlib/dsplib/batchnorm.rst.txt b/master/html/_sources/functionlib/dsplib/batchnorm.rst.txt new file mode 100644 index 0000000..3bbc65f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/batchnorm.rst.txt @@ -0,0 +1,99 @@ +BatchNorm +================= + +对输入数组按通道执行批归一化(Batch Normalization)计算。 +该算子使用给定的均值与方差对输入进行标准化,并通过 epsilon 保证数值稳定性。 + +.. math:: + + dst_{i,j} = \frac{src_{i,j} - mean_j}{\sqrt{variance_j + \epsilon}} + +其中: + +- :math:`i` 表示第 ``unit`` 个样本 +- :math:`j` 表示通道索引 +- :math:`mean_j`、:math:`variance_j` 为第 :math:`j` 个通道的统计量 + +对于 ``int8`` 类型输入,内部以浮点方式计算,最终结果按实现规则取整并输出为 ``int8``。 + +输入: + - **input** - 输入数据地址,形状为 ``[unit, channel]``。 + - **mean** - 均值数组地址,长度为 ``channel``。 + - **variance** - 方差数组地址,长度为 ``channel``。 + - **unit** - 样本数(或展开后的空间维度)。 + - **channel** - 通道数。 + - **epsilon** - 数值稳定因子。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 批归一化后的输出数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``int8``、``fp32`` 类型 + - MT7004 支持 ``fp16``、``fp32`` 类型 + - 当前实现不包含 scale 与 bias,仅执行标准化操作 + +**共享存储版本:** + +.. c:function:: void i8_batchnorm_s(int8_t* input, int8_t* output, float* mean, float* variance, int unit, int channel, float epsilon, int core_mask) +.. c:function:: void fp_batchnorm_s(float* input, float* output, float* mean, float* variance, int unit, int channel, float epsilon, int core_mask) +.. c:function:: void hp_batchnorm_s(half* input, half* output, float* mean, float* variance, int unit, int channel, float epsilon, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + int8_t *input = (int8_t *)0xA0000000; // input 在 DDR 空间 + int8_t *output = (int8_t *)0xC0000000; + float *mean = (float *)0xA1000000; + float *var = (float *)0xA2000000; + int unit = 128; + int channel = 64; + float epsilon = 1e-5f; + int core_mask = 0xff; + + i8_batchnorm_s(input, output, mean, var, unit, channel, epsilon, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_batchnorm_p(int8_t* input, int8_t* output, float* mean, float* variance, int unit, int channel, float epsilon) +.. c:function:: void fp_batchnorm_p(float* input, float* output, float* mean, float* variance, int unit, int channel, float epsilon) +.. c:function:: void hp_batchnorm_p(half* input, half* output, float* mean, float* variance, int unit, int channel, float epsilon) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + int8_t *input = (int8_t *)0x10810000; // input 在 L2 空间 + int8_t *output = (int8_t *)0x10820000; + float *mean = (float *)0x10830000; + float *var = (float *)0x10840000; + int unit = 128; + int channel = 64; + float epsilon = 1e-5f; + + i8_batchnorm_p(input, output, mean, var, unit, channel, epsilon); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/batchnormgrad.rst.txt b/master/html/_sources/functionlib/dsplib/batchnormgrad.rst.txt new file mode 100644 index 0000000..ae08364 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/batchnormgrad.rst.txt @@ -0,0 +1,113 @@ +Batchnormgrad +================= + + +逐元素计算加法梯度 + +计算批标准化 (Batch Normalization) 的梯度。 + +该算子计算损失函数 L 分别对输入 `x`、缩放因子 `scale` (γ) 和偏置 `bias` (β) 的梯度。其中 `bias` 的梯度为 `dbias`。 + +.. math:: + + dscale(\gamma) &= \sum_{i=1}^{m} dy_i \cdot \hat{x}_i \\ + dbias(\beta) &= \sum_{i=1}^{m} dy_i + +.. math:: + + dx_i = \frac{\gamma}{m\sqrt{\sigma^2 + \epsilon}} \left[ m \cdot dy_i - \sum_{j=1}^{m}dy_j - \hat{x}_i \sum_{j=1}^{m}dy_j \hat{x}_j \right] + +其中 :math:`m` 是批处理大小 (batch),:math:`\hat{x}` 是归一化后的 :math:`x`。 + +输入: + - **x** - 前向传播时的输入张量。 + - **dy** - 来自后一层的上游梯度。 + - **mean** - 前向传播时计算的均值。 + - **invar** - 前向传播时计算的逆方差 (1 / sqrt(variance + epsilon))。 + - **scale** - 前向传播时使用的缩放因子 (gamma, γ)。 + - **batch** - 批处理大小。 + - **channel** - 通道数。 + - **is_train** - 是否为训练模式。梯度计算通常在训练时进行。 + - **core_mask** - 核掩码。 + +输出: + - **dx** - 对输入 `x` 的梯度。 + - **dbias** - 对偏置 `bias` (β) 的梯度。 + - **dscale** - 对缩放因子 `scale` (γ) 的梯度。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_batchnormgrad_s(float* x, float* dy, float* mean, float* invar, float* scale, int batch, int channel, int is_train, float* dx, float* dbias, float* dscale, int core_mask) +.. c:function:: void hp_batchnormgrad_s(half* x, half* dy, half* mean, half* invar, half* scale, int batch, int channel, int is_train, half* dx, half* dbias, half* dscale, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 20 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *x = (float *)0xA0000000; // forward input x + float *dy = (float *)0xB0000000; // upstream gradient dy + float *mean = (float *)0xC0000000; // forward mean + float *invar = (float *)0xD0000000; // forward inverse variance + float *scale = (float *)0xE0000000; // forward scale (gamma) + + float *dx = (float *)0xA1000000; // output gradient dx + float *dbias = (float *)0xB1000000; // output gradient dbias + float *dscale = (float *)0xC1000000; // output gradient dscale + + int batch = 4; + int channel = 64; + int is_train = true; + int core_mask = 0xff; + + fp_batchnormgrad_s(x, dy, mean, invar, scale, batch, channel, is_train, dx, dbias, dscale, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_batchnormgrad_p(float* x, float* dy, float* mean, float* invar, float* scale, int batch, int channel, int is_train, float* dx, float* dbias, float* dscale) +.. c:function:: void hp_batchnormgrad_p(half* x, half* dy, half* mean, half* invar, half* scale, int batch, int channel, int is_train, half* dx, half* dbias, half* dscale) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 19 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *x = (float *)0x10000000; // forward input x in L2 space + float *dy = (float *)0x10100000; // upstream gradient dy + float *mean = (float *)0x10200000; // forward mean + float *invar = (float *)0x10300000; // forward inverse variance + float *scale = (float *)0x10400000; // forward scale (gamma) + + float *dx = (float *)0x10500000; // output gradient dx + float *dbias = (float *)0x10600000; // output gradient dbias + float *dscale = (float *)0x10700000; // output gradient dscale + + int batch = 4; + int channel = 32; + int is_train = true; + + fp_batchnormgrad_p(x, dy, mean, invar, scale, batch, channel, is_train, dx, dbias, dscale); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/biasadd.rst.txt b/master/html/_sources/functionlib/dsplib/biasadd.rst.txt new file mode 100644 index 0000000..29bb55e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/biasadd.rst.txt @@ -0,0 +1,108 @@ +Biasadd +================= + + + +将偏置向量 `input_bias` 加到输入张量 `input_x` 上。这是一个特殊的广播加法,其中偏置向量会沿着非通道维度进行广播。 + +.. math:: + + \text{Output} = \text{Input} + \text{Bias} + +对于NCHW格式,计算为 :math:`\text{Output}(n, c, h, w) = \text{Input}(n, c, h, w) + \text{Bias}(c)`。 +对于NHWC格式,计算为 :math:`\text{Output}(n, h, w, c) = \text{Input}(n, h, w, c) + \text{Bias}(c)`。 + +输入: + - **input_x** - 输入张量的数据地址。 + - **input_bias** - 1D偏置张量的数据地址。其长度必须等于输入张量的通道维度。 + - **dims** - 输入张量的维度信息数组。 + - **shape_size** - 输入张量的维度数。 + - **data_format** - 数据布局格式,支持 "NCHW" 和 "NHWC"。 + - **length** - 输入张量 `input_x` 的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其维度与`input_x`相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_biasadd_s(int8_t* input_x, int8_t* input_bias, int8_t* output, int* dims, int shape_size, char* data_format, int length, int core_mask) +.. c:function:: void i16_biasadd_s(int16_t* input_x, int16_t* input_bias, int16_t* output, int* dims, int shape_size, char* data_format, int length, int core_mask) +.. c:function:: void i32_biasadd_s(int32_t* input_x, int32_t* input_bias, int32_t* output, int* dims, int shape_size, char* data_format, int length, int core_mask) +.. c:function:: void fp_biasadd_s(float* input_x, float* input_bias, float* output, int* dims, int shape_size, char* data_format, int length, int core_mask) +.. c:function:: void hp_biasadd_s(half* input_x, half* input_bias, half* output, int* dims, int shape_size, char* data_format, int length, int core_mask) +.. c:function:: void dp_biasadd_s(double* input_x, double* input_bias, double* output, int* dims, int shape_size, char* data_format, int length, int core_mask) +.. c:function:: void c64_biasadd_s(float* input_x, float* input_bias, float* output, int* dims, int shape_size, char* data_format, int length, int core_mask) +.. c:function:: void c128_biasadd_s(double* input_x, double* input_bias, double* output, int* dims, int shape_size, char* data_format, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input_x = (float *)0xA0000000; // input_x 在DDR空间 + float *input_bias = (float *)0xB0000000; // input_bias + float *output = (float *)0xC0000000; // output + + // NHWC format + int dims[] = {2, 224, 224, 3}; // N, H, W, C + int shape_size = 4; + int length = 2 * 224 * 224 * 3; + const char* data_format = "NHWC"; + int core_mask = 0xff; + + // The length of input_bias should be dims, which is 3. + fp_biasadd_s(input_x, input_bias, output, dims, shape_size, data_format, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_biasadd_p(int8_t* input_x, int8_t* input_bias, int8_t* output, int* dims, int shape_size, char* data_format, int length) +.. c:function:: void i16_biasadd_p(int16_t* input_x, int16_t* input_bias, int16_t* output, int* dims, int shape_size, char* data_format, int length) +.. c:function:: void i32_biasadd_p(int32_t* input_x, int32_t* input_bias, int32_t* output, int* dims, int shape_size, char* data_format, int length) +.. c:function:: void fp_biasadd_p(float* input_x, float* input_bias, float* output, int* dims, int shape_size, char* data_format, int length) +.. c:function:: void hp_biasadd_p(half* input_x, half* input_bias, half* output, int* dims, int shape_size, char* data_format, int length) +.. c:function:: void dp_biasadd_p(double* input_x, double* input_bias, double* output, int* dims, int shape_size, char* data_format, int length) +.. c:function:: void c64_biasadd_p(float* input_x, float* input_bias, float* output, int* dims, int shape_size, char* data_format, int length) +.. c:function:: void c128_biasadd_p(double* input_x, double* input_bias, double* output, int* dims, int shape_size, char* data_format, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input_x = (float *)0x10000000; // input_x 在L2空间 + float *input_bias = (float *)0x11000000; // input_bias + float *output = (float *)0x12000000; // output + + // NCHW format + int dims[] = {2, 3, 224, 224}; // N, C, H, W + int shape_size = 4; + int length = 2 * 3 * 224 * 224; + const char* data_format = "NCHW"; + + // The length of input_bias should be dims, which is 3. + fp_biasadd_p(input_x, input_bias, output, dims, shape_size, data_format, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/biasaddgrad.rst.txt b/master/html/_sources/functionlib/dsplib/biasaddgrad.rst.txt new file mode 100644 index 0000000..dd6fd09 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/biasaddgrad.rst.txt @@ -0,0 +1,86 @@ +Biasaddgrad +================= + + +将偏置向量 `input_bias` 加到输入张量 `input_x` 上。这是一个特殊的广播加法,其中偏置向量会沿着非通道维度进行广播。 + +.. math:: + + dbias_c = \sum_{n,h,w} dy(n, h, w, c) + +其中 :math:`dbias_c` 是偏置梯度向量的第 `c` 个元素,:math:`dy` 是上游梯度张量。 + +输入: + - **dy** - 来自后一层的上游梯度张量。格式必须为 NHWC。 + - **dy_dims** - 上游梯度张量 `dy` 的维度信息数组。 + - **shape_size** - 上游梯度张量 `dy` 的维度数。 + - **core_mask** - 核掩码。 + +输出: + - **dbias** - 输出的偏置梯度向量。其长度等于 `dy` 的通道数。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_biasaddgrad_s(float* dy, int* dy_dims, int shape_size, float* dbias, int core_mask) +.. c:function:: void hp_biasaddgrad_s(half* dy, int* dy_dims, int shape_size, half* dbias, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0xA0000000; // dy 在DDR空间 + float *dbias = (float *)0xB0000000; // dbias output + + // NHWC format + int dy_dims[] = {2, 16, 16, 8}; // N, H, W, C + int shape_size = 4; + int core_mask = 0xff; + + // The length of dbias should be dy_dims, which is 8. + fp_biasaddgrad_s(dy, dy_dims, shape_size, dbias, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_biasaddgrad_p(float* dy, int* dy_dims, int shape_size, float* dbias) +.. c:function:: void hp_biasaddgrad_p(half* dy, int* dy_dims, int shape_size, half* dbias) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0x10000000; // dy 在L2空间 + float *dbias = (float *)0x11000000; // dbias output + + // NHWC format + int dy_dims[] = {2, 16, 16, 8}; // N, H, W, C + int shape_size = 4; + + // The length of dbias should be dy_dims, which is 8. + fp_biasaddgrad_p(dy, dy_dims, shape_size, dbias); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/binarycrossentropy.rst.txt b/master/html/_sources/functionlib/dsplib/binarycrossentropy.rst.txt new file mode 100644 index 0000000..8bced91 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/binarycrossentropy.rst.txt @@ -0,0 +1,102 @@ +Binarycrossentropy +================= + + +计算二元交叉熵损失。用于衡量二分类问题中预测值 (`input_x`) 和目标值 (`input_y`) 之间的差距。 + +.. math:: + + L_n = -w_n [y_n \cdot \log(x_n) + (1 - y_n) \cdot \log(1 - x_n)] + +其中 :math:`x_n` 是预测值, :math:`y_n` 是目标值, :math:`w_n` 是可选的样本权重。最终输出由 `reduction` 参数决定: + +- **None (0):** 不进行规约,输出每个元素的损失 :math:`L_n`。 +- **Mean (1):** 输出所有元素损失的平均值。 +- **Sum (2):** 输出所有元素损失的总和。 + +输入: + - **input_size** - 输入张量的总元素数量。 + - **reduction** - 规约类型 (0: None, 1: Mean, 2: Sum)。 + - **input_x** - 预测值的张量数据地址,通常是Sigmoid函数的输出。 + - **input_y** - 目标值(标签)的张量数据地址。 + - **weight** - (可选) 权重张量的数据地址,维度与 `input_x` 相同。 + - **loss** - (输出) 最终损失值的数据地址。如果 `reduction` 为 `None`,其大小为 `input_size`;否则大小为1。 + - **tmp_loss** - 用于存储逐元素损失的临时工作空间地址,大小必须为 `input_size`。 + - **weight_defined** - 权重是否有效的标志。若为非0,则`weight`参数必须提供。 + - **core_mask** - 核掩码。 + + +输出: + - **loss** - 写入最终损失值的数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_binarycrossentropy_s(int input_size, int reduction, float* input_x, float* input_y, float* weight, float* loss, float* tmp_loss, int weight_defined, int core_mask) +.. c:function:: void hp_binarycrossentropy_s(int input_size, int reduction, half* input_x, half* input_y, half* weight, half* loss, half* tmp_loss, int weight_defined, int core_mask) +.. c:function:: void i8_binarycrossentropy_s(int input_size, int reduction, int8_t* input_x, int8_t* input_y, int8_t* weight, int8_t* loss, int8_t* tmp_loss, int weight_defined, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input_x = (float *)0xA0000000; // input_x 在DDR空间 + float *input_y = (float *)0xB0000000; // input_y + float *weight = (float *)0xC0000000; // weight + float *loss = (float *)0xD0000000; // loss output + float *tmp_loss = (float *)0xE0000000; // temp workspace + + int input_size = 1024; + int reduction = 1; // Mean + int weight_defined = 1; // true + int core_mask = 0xff; + + fp_binarycrossentropy_s(input_size, reduction, input_x, input_y, weight, loss, tmp_loss, weight_defined, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_binarycrossentropy_p(int input_size, int reduction, float* input_x, float* input_y, float* weight, float* loss, float* tmp_loss, int weight_defined) +.. c:function:: void hp_binarycrossentropy_p(int input_size, int reduction, half* input_x, half* input_y, half* weight, half* loss, half* tmp_loss, int weight_defined) +.. c:function:: void i8_binarycrossentropy_p(int input_size, int reduction, int8_t* input_x, int8_t* input_y, int8_t* weight, int8_t* loss, int8_t* tmp_loss, int weight_defined) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input_x = (float *)0x10000000; // input_x 在L2空间 + float *input_y = (float *)0x11000000; // input_y + float *weight = (float *)0x12000000; // weight + float *loss = (float *)0x13000000; // loss output + float *tmp_loss = (float *)0x14000000; // temp workspace + + int input_size = 1024; + int reduction = 1; // Mean + int weight_defined = 0; // false + + fp_binarycrossentropy_p(input_size, reduction, input_x, input_y, weight, loss, tmp_loss, weight_defined); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/binarycrossentropygrad.rst.txt b/master/html/_sources/functionlib/dsplib/binarycrossentropygrad.rst.txt new file mode 100644 index 0000000..8ab548b --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/binarycrossentropygrad.rst.txt @@ -0,0 +1,95 @@ +Binarycrossentropygrad +================= + + +计算二元交叉熵损失函数的梯度。 + +.. math:: + + dx_n = dL'_n \cdot w_n \cdot \frac{x_n - y_n}{x_n(1 - x_n) + \epsilon} + +其中 :math:`x_n` 是前向预测值, :math:`y_n` 是目标值, :math:`w_n` 是样本权重, :math:`dL'_n` 是上游梯度。`reduction` 参数决定了 :math:`dL'_n` 的广播方式。 + +输入: + - **input_size** - 输入张量的总元素数量。 + - **reduction** - 前向传播时使用的规约类型 (0: None, 1: Mean, 2: Sum)。 + - **input_x** - 前向传播时的预测值张量。 + - **input_y** - 前向传播时的目标值(标签)张量。 + - **weight** - (可选) 前向传播时使用的权重张量。 + - **dloss** - 来自后一层的上游梯度。 + - **dx** - (输出) 梯度结果的存储地址。 + - **weight_defined** - 权重是否有效的标志。若为非0,则`weight`参数必须提供。 + - **core_mask** - 核掩码。 + +输出: + - **dx** - 写入计算出的对 `input_x` 的梯度。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_binarycrossentropygrad_s(int input_size, int reduction, float* input_x, float* input_y, float* weight, float* dloss, float* dx, int weight_defined, int core_mask) +.. c:function:: void hp_binarycrossentropygrad_s(int input_size, int reduction, half* input_x, half* input_y, half* weight, half* dloss, half* dx, int weight_defined, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input_x = (float *)0xA0000000; // forward input_x, DDR + float *input_y = (float *)0xB0000000; // forward input_y + float *weight = (float *)0xC0000000; // forward weight + float *dloss = (float *)0xD0000000; // upstream gradient + float *dx = (float *)0xE0000000; // output gradient dx + + int input_size = 1024; + int reduction = 1; // Mean + int weight_defined = 1; // true + int core_mask = 0xff; + + fp_binarycrossentropygrad_s(input_size, reduction, input_x, input_y, weight, dloss, dx, weight_defined, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_binarycrossentropygrad_p(int input_size, int reduction, float* input_x, float* input_y, float* weight, float* dloss, float* dx, int weight_defined) +.. c:function:: void hp_binarycrossentropygrad_p(int input_size, int reduction, half* input_x, half* input_y, half* weight, half* dloss, half* dx, int weight_defined) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input_x = (float *)0x10000000; // forward input_x, L2 + float *input_y = (float *)0x11000000; // forward input_y + float *weight = (float *)0x12000000; // forward weight + float *dloss = (float *)0x13000000; // upstream gradient + float *dx = (float *)0x14000000; // output gradient dx + + int input_size = 1024; + int reduction = 1; // Mean + int weight_defined = 0; // false + + fp_binarycrossentropygrad_p(input_size, reduction, input_x, input_y, weight, dloss, dx, weight_defined); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/cast.rst.txt b/master/html/_sources/functionlib/dsplib/cast.rst.txt new file mode 100644 index 0000000..4e06742 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/cast.rst.txt @@ -0,0 +1,128 @@ +Cast +================= + + + +对输入数组进行 **逐元素数据类型转换(Cast)**,将源类型的数据按 C 语言强制类型转换规则转换为目标类型后输出。 + +数学表达式为: + +.. math:: + + dst_i = (T_{dst}) \; src_i + +输入: + - **input** - 输入数据地址。 + - **length** - 计算长度(元素个数)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 转换后的结果数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32``、``fp64``、``int8``、``int16``、``int32``、``cplx64``、``cplx128`` 类型之间的 Cast + - MT7004 支持 ``fp16``、``fp32``、``int16``、``int32``、``cplx64`` 类型之间的 Cast + - 转换规则遵循 C 语言显式类型转换语义,可能存在精度截断或溢出风险 + - 输入与输出张量长度必须一致 + +**共享存储版本:** + +.. c:function:: void casti8Tointi8_s(int8_t* input, int8_t* output, int length, int core_mask) +.. c:function:: void casti8Tointi16_s(int8_t* input, int16_t* output, int length, int core_mask) +.. c:function:: void casti8Tointi32_s(int8_t* input, int32_t* output, int length, int core_mask) +.. c:function:: void casti8Tofp_s(int8_t* input, float* output, int length, int core_mask) + +.. c:function:: void casti16Tointi8_s(int16_t* input, int8_t* output, int length, int core_mask) +.. c:function:: void casti16Tointi16_s(int16_t* input, int16_t* output, int length, int core_mask) +.. c:function:: void casti16Tointi32_s(int16_t* input, int32_t* output, int length, int core_mask) +.. c:function:: void casti16Tofp_s(int16_t* input, float* output, int length, int core_mask) + +.. c:function:: void casti32Tointi8_s(int32_t* input, int8_t* output, int length, int core_mask) +.. c:function:: void casti32Tointi16_s(int32_t* input, int16_t* output, int length, int core_mask) +.. c:function:: void casti32Tointi32_s(int32_t* input, int32_t* output, int length, int core_mask) +.. c:function:: void casti32Tofp_s(int32_t* input, float* output, int length, int core_mask) + +.. c:function:: void castfpTofp_s(float* input, float* output, int length, int core_mask) +.. c:function:: void castfpToint8_s(float* input, int8_t* output, int length, int core_mask) +.. c:function:: void castfpToint16_s(float* input, int16_t* output, int length, int core_mask) +.. c:function:: void castfpToint32_s(float* input, int32_t* output, int length, int core_mask) + +.. c:function:: void castdpTodp_s(double* input, double* output, int length, int core_mask) +.. c:function:: void castdpTofp_s(double* input, float* output, int length, int core_mask) + +.. c:function:: void castc64Toc64_s(cplx64* input, cplx64* output, int length, int core_mask) +.. c:function:: void castc128Toc128_s(cplx128* input, cplx128* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // FT78NE 示例:int8 -> int16 + #include + #include + + int main(int argc, char* argv[]) { + int8_t *input = (int8_t *)0xA0000000; + int16_t *output = (int16_t *)0xC0000000; + + int length = 2048; + int core_mask = 0xff; + + casti8Tointi16_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void casti8Tointi8_p(int8_t* input, int8_t* output, int length) +.. c:function:: void casti8Tointi16_p(int8_t* input, int16_t* output, int length) +.. c:function:: void casti8Tointi32_p(int8_t* input, int32_t* output, int length) +.. c:function:: void casti8Tofp_p(int8_t* input, float* output, int length) + +.. c:function:: void casti16Tointi8_p(int16_t* input, int8_t* output, int length) +.. c:function:: void casti16Tointi16_p(int16_t* input, int16_t* output, int length) +.. c:function:: void casti16Tointi32_p(int16_t* input, int32_t* output, int length) +.. c:function:: void casti16Tofp_p(int16_t* input, float* output, int length) + +.. c:function:: void casti32Tointi8_p(int32_t* input, int8_t* output, int length) +.. c:function:: void casti32Tointi16_p(int32_t* input, int16_t* output, int length) +.. c:function:: void casti32Tointi32_p(int32_t* input, int32_t* output, int length) +.. c:function:: void casti32Tofp_p(int32_t* input, float* output, int length) + +.. c:function:: void castfpTofp_p(float* input, float* output, int length) +.. c:function:: void castfpToint8_p(float* input, int8_t* output, int length) +.. c:function:: void castfpToint16_p(float* input, int16_t* output, int length) +.. c:function:: void castfpToint32_p(float* input, int32_t* output, int length) + +.. c:function:: void castdpTodp_p(double* input, double* output, int length) +.. c:function:: void castdpTofp_p(double* input, float* output, int length) + +.. c:function:: void castc64Toc64_p(cplx64* input, cplx64* output, int length) +.. c:function:: void castc128Toc128_p(cplx128* input, cplx128* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 示例:int8 -> int16(私有存储) + #include + #include + + int main(int argc, char* argv[]) { + int8_t *input = (int8_t *)0x10810000; + int16_t *output = (int16_t *)0x10820000; + + int length = 2048; + casti8Tointi16_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/ceil.rst.txt b/master/html/_sources/functionlib/dsplib/ceil.rst.txt new file mode 100644 index 0000000..a3d1186 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/ceil.rst.txt @@ -0,0 +1,79 @@ +Ceil +================= + + + +逐元素计算输入张量的向上取整值。即返回大于或等于每个元素的最小整数。 + +.. math:: + + \text{Output}_i = \lceil \text{Input}_i \rceil + +输入: + - **input_x** - 输入张量的数据地址。 + - **input_size** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与`input_x`相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, double + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_ceil_s(float* input_x, float* output, int input_size, int core_mask) +.. c:function:: void hp_ceil_s(half* input_x, half* output, int input_size, int core_mask) +.. c:function:: void dp_ceil_s(double* input_x, double* output, int input_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input_x = (float *)0xA0000000; // input_x 在DDR空间 + float *output = (float *)0xB0000000; // output + + int input_size = 4096; + int core_mask = 0xff; + + fp_ceil_s(input_x, output, input_size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_ceil_p(float* input_x, float* output, int input_size) +.. c:function:: void hp_ceil_p(half* input_x, half* output, int input_size) +.. c:function:: void dp_ceil_p(double* input_x, double* output, int input_size) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input_x = (float *)0x10000000; // input_x 在L2空间 + float *output = (float *)0x11000000; // output + + int input_size = 1024; + + fp_ceil_p(input_x, output, input_size); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/clip.rst.txt b/master/html/_sources/functionlib/dsplib/clip.rst.txt new file mode 100644 index 0000000..9e7b6cc --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/clip.rst.txt @@ -0,0 +1,90 @@ +Clip +================= + + + +逐元素将输入张量 `src` 的值裁剪到指定的最小值 `min` 和最大值 `max` 范围内。 + +.. math:: + + \text{dst}_i = \max(\min, \min(\text{src}_i, \max)) + +输入: + - **src** - 输入张量的数据地址。 + - **length** - 输入张量的总元素数量。 + - **min** - 裁剪范围的最小值。 + - **max** - 裁剪范围的最大值。 + - **core_mask** - 核掩码。 + +输出: + - **dst** - 输出张量的数据地址,其大小与`src`相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64 + - MT7004 支持int16, int32, fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_clip_s(int8_t* src, int8_t* dst, int length, int8_t min, int8_t max, int core_mask) +.. c:function:: void i16_clip_s(int16_t* src, int16_t* dst, int length, int16_t min, int16_t max, int core_mask) +.. c:function:: void i32_clip_s(int32_t* src, int32_t* dst, int length, int32_t min, int32_t max, int core_mask) +.. c:function:: void fp_clip_s(float* src, float* dst, int length, float min, float max, int core_mask) +.. c:function:: void hp_clip_s(half* src, half* dst, int length, half min, half max, int core_mask) +.. c:function:: void dp_clip_s(double* src, double* dst, int length, double min, double max, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *src = (float *)0xA0000000; // src 在DDR空间 + float *dst = (float *)0xB0000000; // dst + + int length = 4096; + float min = 0.0f; + float max = 6.0f; // 例如ReLU6的裁剪范围 + int core_mask = 0xff; + + fp_clip_s(src, dst, length, min, max, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_clip_p(int8_t* src, int8_t* dst, int length, int8_t min, int8_t max) +.. c:function:: void i16_clip_p(int16_t* src, int16_t* dst, int length, int16_t min, int16_t max) +.. c:function:: void i32_clip_p(int32_t* src, int32_t* dst, int length, int32_t min, int32_t max) +.. c:function:: void fp_clip_p(float* src, float* dst, int length, float min, float max) +.. c:function:: void hp_clip_p(half* src, half* dst, int length, half min, half max) +.. c:function:: void dp_clip_p(double* src, double* dst, int length, double min, double max) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *src = (float *)0x10000000; // src 在L2空间 + float *dst = (float *)0x11000000; // dst + + int length = 1024; + float min = -1.0f; + float max = 1.0f; + + fp_clip_p(src, dst, length, min, max); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/concat.rst.txt b/master/html/_sources/functionlib/dsplib/concat.rst.txt new file mode 100644 index 0000000..96011ea --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/concat.rst.txt @@ -0,0 +1,110 @@ +Concat +================= + +在指定维度(axis)上连接一组张量。所有输入张量除了连接维度外,其他维度的形状必须完全相同。 + +.. math:: + + output = \text{Concat}(input_0, input_1, \dots, input_{n-1}, \text{axis}) + +输入: + - **inputs** - 包含所有输入张量起始地址的数组指针。 + - **input_shapes** - 包含所有输入张量形状(shape)的数组指针。 + - **num_inputs** - 输入张量的数量。 + - **axis** - 连接操作所在的维度索引。 + - **input_ndim** - 输入张量的维度(秩)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果存储地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 所有输入张量的维度 `input_ndim` 必须一致。 + - 除 `axis` 指定的维度外,所有输入张量的 `shape[i]` 必须相等。 + +**共享存储版本:** + +.. c:function:: void i8_concat_s(int8_t* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, int8_t* output, int core_mask) +.. c:function:: void i16_concat_s(int16_t* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, int16_t* output, int core_mask) +.. c:function:: void i32_concat_s(int32_t* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, int32_t* output, int core_mask) +.. c:function:: void hp_concat_s(half* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, half* output, int core_mask) +.. c:function:: void fp_concat_s(float* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, float* output, int core_mask) +.. c:function:: void dp_concat_s(double* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, double* output, int core_mask) +.. c:function:: void c64_concat_s(float* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, float* output, int core_mask) +.. c:function:: void c128_concat_s(double* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, double* output, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 22 + + // FT78NE 示例(共享存储) + #include + #include "78NE/utils.h" + + int main() { + int shape0[] = { 4, 10, 8, 12 }; + int shape1[] = { 4, 5, 8, 12 }; + int shape2[] = { 4, 15, 8, 12 }; + int* input_shapes[] = { shape0, shape1, shape2 }; + + float *input0 = (float *)0xA0000000; + float *input1 = (float *)0xA2000000; + float *input2 = (float *)0xA4000000; + float *inputs[] = { input0, input1, input2 }; + float *output = (float *)0xB0000000; + + int num_inputs = 3; + int axis = 1; + int input_ndim = 4; + int core_mask = 0b1011; + + fp_concat_s(inputs, input_shapes, num_inputs, axis, input_ndim, output, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_concat_p(int8_t* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, int8_t* output) +.. c:function:: void i16_concat_p(int16_t* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, int16_t* output) +.. c:function:: void i32_concat_p(int32_t* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, int32_t* output) +.. c:function:: void hp_concat_p(half* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, half* output) +.. c:function:: void fp_concat_p(float* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, float* output) +.. c:function:: void dp_concat_p(double* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, double* output) +.. c:function:: void c64_concat_p(float* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, float* output) +.. c:function:: void c128_concat_p(double* inputs[], int* input_shapes[], int num_inputs, int axis, int input_ndim, double* output) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 18 + + // MT7004 示例(私有存储) + #include + + int main() { + int shape0[] = { 2, 3, 4, 5 }; + int shape1[] = { 2, 2, 4, 5 }; + int* input_shapes[] = { shape0, shape1 }; + + float *input0 = (float *)0x10810000; + float *input1 = (float *)0x10820000; + float *inputs[] = { input0, input1 }; + float *output = (float *)0x10830000; + + int num_inputs = 2; + int axis = 1; + int input_ndim = 4; + + fp_concat_p(inputs, input_shapes, num_inputs, axis, input_ndim, output); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/constant_of_shape.rst.txt b/master/html/_sources/functionlib/dsplib/constant_of_shape.rst.txt new file mode 100644 index 0000000..e3d406b --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/constant_of_shape.rst.txt @@ -0,0 +1,110 @@ +ConstantOfShape +================= + + 将给定内存区间内的所有元素填充为指定的常量值。对于复数类型,实部和虚部需分别传入。 + + 输入: + - **output** - 输出数据的起始地址。 + - **start** - 填充的起始索引(包含)。 + - **end** - 填充的结束索引(不包含)。 + - **value** - 需要填充的常量值(针对非复数类型)。 + - **value_real** - 复数常量的实部(针对复数类型)。 + - **value_imag** - 复数常量的虚部(针对复数类型)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 填充完成后的数据地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_constant_of_shape_s(int8_t* output, int start, int end, int8_t value, int core_mask) +.. c:function:: void i16_constant_of_shape_s(int16_t* output, int start, int end, int16_t value, int core_mask) +.. c:function:: void i32_constant_of_shape_s(int32_t* output, int start, int end, int32_t value, int core_mask) +.. c:function:: void fp_constant_of_shape_s(float* output, int start, int end, float value, int core_mask) +.. c:function:: void dp_constant_of_shape_s(double* output, int start, int end, double value, int core_mask) +.. c:function:: void c64_constant_of_shape_s(float* output, int start, int end, float value_real, float value_imag, int core_mask) +.. c:function:: void c128_constant_of_shape_s(double* output, int start, int end, double value_real, double value_imag, int core_mask) + + **C调用示例(Float32):** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + + int main(int argc, char* argv[]) { + // output在DDR空间 + float *output = (float *)0xA0000000; + int start = 0; + int end = 16000; + float value = 1.5f; + int core_mask = 0xff; + + // 多核并行填充 + fp_constant_of_shape_s(output, start, end, value, core_mask); + + return 0; + } + + **C调用示例(Complex64):** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + #include + + int main(int argc, char* argv[]) { + // output在DDR空间,c64类型通常使用float*指针访问 + float *output = (float *)0xA0000000; + int start = 0; + int end = 1000; // 1000个复数元素 + float val_r = 0.5f; + float val_i = -0.5f; + int core_mask = 0xff; + + // 注意:c64接口需分别传入实部和虚部 + c64_constant_of_shape_s(output, start, end, val_r, val_i, core_mask); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_constant_of_shape_p(int8_t* output, int start, int end, int8_t value) +.. c:function:: void i16_constant_of_shape_p(int16_t* output, int start, int end, int16_t value) +.. c:function:: void i32_constant_of_shape_p(int32_t* output, int start, int end, int32_t value) +.. c:function:: void fp_constant_of_shape_p(float* output, int start, int end, float value) +.. c:function:: void dp_constant_of_shape_p(double* output, int start, int end, double value) +.. c:function:: void c64_constant_of_shape_p(float* output, int start, int end, float value_real, float value_imag) +.. c:function:: void c128_constant_of_shape_p(double* output, int start, int end, double value_real, double value_imag) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + + int main(int argc, char* argv[]) { + // output在L2空间 + float *output = (float *)0x10810000; + int start = 0; + int end = 1024; + float value = 3.14f; + + fp_constant_of_shape_p(output, start, end, value); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/conv2d.rst.txt b/master/html/_sources/functionlib/dsplib/conv2d.rst.txt index 0c0e537..7b56584 100644 --- a/master/html/_sources/functionlib/dsplib/conv2d.rst.txt +++ b/master/html/_sources/functionlib/dsplib/conv2d.rst.txt @@ -15,7 +15,7 @@ Conv2d - :math:`j` 对应输出通道,其范围为 :math:`[0, C_{out}-1]`,其中 :math:`C_{out}` 为输出通道数,该值也等于卷积核的个数。 - :math:`k` 对应输入通道数,其范围为 :math:`[0, C_{in}-1]`,其中 :math:`C_{in}` 为输入通道数,该值也等于卷积核的通道数。 -因此,上面的公式中,:math:`bias(C_{out_j})` 为第 :math:`j` 个输出通道的偏置,:math:`weight(C_{out_j}, k)` 表示第 :math:`j` 个卷积核在第 :math:`k` 个输入通道的卷积核切片,:math:`X(N_i, k)` 为特征图第 :math:`i` 个 batch 第 :math:`k` 个输入通道的切片。卷积核 shape 为 :math:`(\text{kernel_size}[0], \text{kernel_size}[1])`,其中 kernel_size[0] 和 kernel_size[1] 是卷积核的高度和宽度。若考虑到输入输出通道以及 group,则完整卷积核的 shape 为 :math:`(C_{out}, \text{kernel_size}[0], \text{kernel_size}[1], C_{in}/\text{group})`,其中 group 是分组卷积时在通道上分割输入 :math:`x` 的组数。 +因此,上面的公式中,:math:`bias(C_{out_j})` 为第 :math:`j` 个输出通道的偏置,:math:`weight(C_{out_j}, k)` 表示第 :math:`j` 个卷积核在第 :math:`k` 个输入通道的卷积核切片,:math:`X(N_i, k)` 为特征图第 :math:`i` 个 batch 第 :math:`k` 个输入通道的切片。卷积核 shape 为 :math:`(\text{kernel\_size}[0], \text{kernel\_size}[1])`,其中 kernel_size[0] 和 kernel_size[1] 是卷积核的高度和宽度。若考虑到输入输出通道以及 group,则完整卷积核的 shape 为 :math:`(C_{out}, \text{kernel\_size}[0], \text{kernel\_size}[1], C_{in}/\text{group})`,其中 group 是分组卷积时在通道上分割输入 :math:`x` 的组数。 输入: - **input_x** - 输入数据的地址 diff --git a/master/html/_sources/functionlib/dsplib/cos.rst.txt b/master/html/_sources/functionlib/dsplib/cos.rst.txt new file mode 100644 index 0000000..c3a8a33 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/cos.rst.txt @@ -0,0 +1,88 @@ +Cos +================= + + + +对输入数组逐元素计算余弦值(cosine)。 + +.. math:: + + dst_i = \cos(src_i) + +其中输入角度以弧度(radian)为单位。 + +输入: + - **src_data** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst_data** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp, dp, int8, int16, int32 类型 + - MT7004 支持 hp, fp, int16, int32 类型 + - 当输入类型为 int8 / int16 / int32 时,输出类型统一为 fp(float) + - 输入数值将被解释为弧度值 + +**共享存储版本:** + +.. c:function:: void i8_cos_s(int8_t* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void i16_cos_s(int16_t* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void i32_cos_s(int* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void hp_cos_s(half* src_data, half* dst_data, int length, int core_mask) +.. c:function:: void fp_cos_s(float* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void dp_cos_s(double* src_data, double* dst_data, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1024; + int core_mask = 0xff; + + fp_cos_s(input0, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_cos_p(int8_t* src_data, float* dst_data, int length) +.. c:function:: void i16_cos_p(int16_t* src_data, float* dst_data, int length) +.. c:function:: void i32_cos_p(int* src_data, float* dst_data, int length) +.. c:function:: void hp_cos_p(half* src_data, half* dst_data, int length) +.. c:function:: void fp_cos_p(float* src_data, float* dst_data, int length) +.. c:function:: void dp_cos_p(double* src_data, double* dst_data, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; // input在L2空间 + float *output = (float *)0x10820000; + int length = 1024; + + fp_cos_p(input0, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/cumsum.rst.txt b/master/html/_sources/functionlib/dsplib/cumsum.rst.txt new file mode 100644 index 0000000..426930e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/cumsum.rst.txt @@ -0,0 +1,102 @@ +Cumsum +================= + +沿指定轴计算张量的累积和 + +对于输入张量,沿指定轴计算累积和。支持独占(exclusive)和包含(inclusive)两种模式。 + +.. math:: + + \text{output}[i] = \begin{cases} + 0, & \text{if } i = 0 \text{ and exclusive} = \text{True} \\ + \sum_{j=0}^{i-1} \text{input}[j], & \text{if } i > 0 \text{ and exclusive} = \text{True} \\ + \text{input}[0], & \text{if } i = 0 \text{ and exclusive} = \text{False} \\ + \sum_{j=0}^{i} \text{input}[j], & \text{if } i > 0 \text{ and exclusive} = \text{False} + \end{cases} + +输入: + - **input** - 输入数据地址。 + - **out_dim** - 输出维度(层数)。 + - **axis_dim** - 累积轴的维度大小。 + - **inner_dim** - 内部维度大小。 + - **exclusive** - 是否使用独占模式。1表示独占模式,0表示包含模式。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + - 独占模式(exclusive=1):第一个元素为0,第i个元素为前i-1个元素之和 + - 包含模式(exclusive=0):第一个元素为输入的第一个元素,第i个元素为前i个元素之和 + +**共享存储版本:** + +.. c:function:: void i8_cumsum_s(int8_t* output, int8_t* input, int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) +.. c:function:: void i16_cumsum_s(int16_t* output, int16_t* input, int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) +.. c:function:: void i32_cumsum_s(int32_t* output, int32_t* input, int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) +.. c:function:: void hp_cumsum_s(half* output, half* input, int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) +.. c:function:: void fp_cumsum_s(float* output, float* input, int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) +.. c:function:: void dp_cumsum_s(double* output, double* input, int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) +.. c:function:: void c64_cumsum_s(float (*output)[2], float (*input)[2], int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) +.. c:function:: void c128_cumsum_s(double (*output)[2], double (*input)[2], int out_dim, int axis_dim, int inner_dim, int exclusive, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xB0000000; //output在DDR空间 + int out_dim = 8; // 输出维度 + int axis_dim = 16; // 累积轴维度 + int inner_dim = 16; // 内部维度 + int exclusive = 0; // 包含模式 + int core_mask = 0xff; + fp_cumsum_s(output, input, out_dim, axis_dim, inner_dim, exclusive, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_cumsum_p(int8_t* output, int8_t* input, int out_dim, int axis_dim, int inner_dim, int exclusive) +.. c:function:: void i16_cumsum_p(int16_t* output, int16_t* input, int out_dim, int axis_dim, int inner_dim, int exclusive) +.. c:function:: void i32_cumsum_p(int32_t* output, int32_t* input, int out_dim, int axis_dim, int inner_dim, int exclusive) +.. c:function:: void hp_cumsum_p(half* output, half* input, int out_dim, int axis_dim, int inner_dim, int exclusive) +.. c:function:: void fp_cumsum_p(float* output, float* input, int out_dim, int axis_dim, int inner_dim, int exclusive) +.. c:function:: void dp_cumsum_p(double* output, double* input, int out_dim, int axis_dim, int inner_dim, int exclusive) +.. c:function:: void c64_cumsum_p(float (*output)[2], float (*input)[2], int out_dim, int axis_dim, int inner_dim, int exclusive) +.. c:function:: void c128_cumsum_p(double (*output)[2], double (*input)[2], int out_dim, int axis_dim, int inner_dim, int exclusive) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; //input在L2空间 + float *output = (float *)0x10850000; //output在L2空间 + int out_dim = 8; // 输出维度 + int axis_dim = 16; // 累积轴维度 + int inner_dim = 16; // 内部维度 + int exclusive = 0; // 包含模式 + fp_cumsum_p(output, input, out_dim, axis_dim, inner_dim, exclusive); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/customextractfeatures.rst.txt b/master/html/_sources/functionlib/dsplib/customextractfeatures.rst.txt new file mode 100644 index 0000000..b8ce741 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/customextractfeatures.rst.txt @@ -0,0 +1,131 @@ +CustomExtractFeatures +======================= + + +对输入字符串序列逐条进行特征提取,生成对应的离散标签(label)和权重(weight)。 + +该算子主要用于文本特征工程,包含以下逻辑: + +- 对黑名单字符串进行过滤 +- 对非黑名单字符串计算哈希特征 +- 根据字符串中空格数量生成权重 + + +.. math:: + + \text{label}_i = + \begin{cases} + 0, & \text{if } s_i \in \text{blacklist} \\ + \operatorname{hash}(s_i) \bmod K, & \text{otherwise} + \end{cases} + +.. math:: + + \text{weight}_i = + \begin{cases} + 0, & \text{if } s_i \in \text{blacklist} \\ + \text{space\_count}(s_i) + 1, & \text{otherwise} + \end{cases} + +其中: + +- :math:`s_i` 表示第 :math:`i` 条输入字符串 +- :math:`K` 为固定的哈希空间大小(默认 :math:`10^6`) +- ``blacklist`` = {``""``, ``""``, ``" "``} + + +输入: + - **string_pointers** - 指向字符串首地址的指针数组。 + - **string_lengths** - 各字符串对应的长度数组。 + - **num_strings** - 输入字符串的数量。 + - **core_mask** - 核掩码。 + +输出: + - **output_labels** - 输出标签数组地址(int32)。 + - **output_weights** - 输出权重数组地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型: + - fp32, fp64 + - int8, int16, int32 + - cplx64, cplx128 + - MT7004 支持的数据类型: + - fp16, fp32 + - int16, int32 + - cplx64 + - 当 ``num_strings == 0`` 时,输出的 ``label`` 和 ``weight`` 被置为 0 + - 黑名单字符串不会参与哈希计算 + + +**共享存储版本:** + +.. c:function:: void fp_extract_features_s(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, float* output_weights, int core_mask) +.. c:function:: void dp_extract_features_s(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, double* output_weights, int core_mask) +.. c:function:: void i8_extract_features_s(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, int8_t* output_weights, int core_mask) +.. c:function:: void i16_extract_features_s(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, int16_t* output_weights, int core_mask) +.. c:function:: void i32_extract_features_s(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, int32_t* output_weights, int core_mask) +.. c:function:: void c64_extract_features_s(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, float* output_weights, int core_mask) +.. c:function:: void c128_extract_features_s(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, double* output_weights, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + char* strings[] = {"hello world", "", "test data"}; + int lengths[] = {11, 3, 9}; + int num_strings = 3; + + int *labels = (int *)0xA0000000; + float *weights = (float *)0xB0000000; + + int core_mask = 0xff; + + fp_extract_features_s(strings, lengths, num_strings, labels, weights, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_extract_features_p(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, float* output_weights) +.. c:function:: void dp_extract_features_p(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, double* output_weights) +.. c:function:: void i8_extract_features_p(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, int8_t* output_weights) +.. c:function:: void i16_extract_features_p(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, int16_t* output_weights) +.. c:function:: void i32_extract_features_p(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, int32_t* output_weights) +.. c:function:: void c64_extract_features_p(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, float* output_weights) +.. c:function:: void c128_extract_features_p(char** string_pointers, int* string_lengths, int num_strings, int* output_labels, double* output_weights) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + char* strings[] = {"example text"}; + int lengths[] = {12}; + int num_strings = 1; + + int *labels = (int *)0x10000000; + float *weights = (float *)0x11000000; + + fp_extract_features_p(strings, lengths, num_strings, labels, weights); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/customnormalize.rst.txt b/master/html/_sources/functionlib/dsplib/customnormalize.rst.txt new file mode 100644 index 0000000..2c41439 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/customnormalize.rst.txt @@ -0,0 +1,89 @@ +Customnormalize +================= + +对输入字符串进行一系列自定义的文本标准化处理。 + +该算子按顺序执行以下操作: +1. 将所有大写字母转换为小写字母。 +2. 去除字符串首尾的空白字符 (space, ``\t``, ``\n``, ``\v``, ``\f``, ``\r``)。 +3. 删除字符串中预定义的一组标点符号 (例如 ``.*()\"``)。 +4. 标准化缩写词,例如,将 ``we 're`` 转换为 ``we're``。 +5. 展开常见的英文缩写词 (例如, ``n't`` -> `` not``, ``'ll`` -> `` will``, ``i'm`` -> ``i am`` 等)。 +6. 标准化问号和感叹号,例如合并连续的符号 (``!!!`` -> ``!``) 并在符号和单词之间添加空格 (``Hello?`` -> ``Hello ?``)。 +7. 去除字符串首尾预定义的另一组字符 (例如 ``[\\s,:;\\-&'\"]+``)。 +8. 再次去除字符串首尾的空白字符。 +9. 如果处理后的字符串长度超过300个字符,则将其截断为300个字符。 +10. 在字符串首部添加前缀 `` ``,尾部添加后缀 `` ``。 + +输入: + - **str** - 输入字符串的地址。 + - **str_len** - 输入字符串的长度。 + - **tmp_str** - 用于中间计算的临时缓冲区的地址,其大小应足以容纳处理过程中的字符串。 + - **params** - 一个 ``long long`` 类型的数组,其中每个元素是指向标准化操作所需的配置字符串或参数的指针。 + - **core_mask** - 核掩码 (仅共享存储版本需要)。 + +输出: + - **result** - 存储标准化后字符串的输出缓冲区的地址。 + - **result_len** - 指向一个 ``int`` 变量的指针,用于存储结果字符串的长度。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分数据类型 + +**共享存储版本:** + +.. c:function:: void customnormalize_s(char *str, int str_len, char *result, int *result_len, char *tmp_str, long long *params, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + #include + #include + #include "customnormalize.h" + + int main() { + char *str = (char *)0xA0000000; // str in DDR + char *result = (char *)0xB0000000; // result in DDR + char *tmp_str = (char *)0xC0000000; // tmp_str in DDR + int str_len = 50; // 假设长度 + int result_len = 0; + long long params[20]; // 假设参数已填充 + int core_mask = 0xff; + + customnormalize_s(str, str_len, result, &result_len, tmp_str, params, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void customnormalize_p(char *str, int str_len, char *result, int *result_len, char *tmp_str, long long *params) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + #include + #include + #include "customnormalize.h" + + int main() { + char *str = (char *)0x10000000; // str in L2 + char *result = (char *)0x10001000; // result in L2 + char *tmp_str = (char *)0x10002000; // tmp_str in L2 + int str_len = 50; // 假设长度 + int result_len = 0; + long long params[20]; // 假设参数已填充 + + customnormalize_p(str, str_len, result, &result_len, tmp_str, params); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/custompredict.rst.txt b/master/html/_sources/functionlib/dsplib/custompredict.rst.txt new file mode 100644 index 0000000..ecc72f9 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/custompredict.rst.txt @@ -0,0 +1,77 @@ +CustomPredict +================= + +根据置信度(weight)对候选标签(label)进行排序和筛选,以生成最终的预测结果。 + +输入: + - **input** - ``LabelInfo`` 结构体数组的地址。每个结构体包含一个整数 ``label`` 和一个浮点数 ``weight``。 + - **input_size** - 输入数组中的元素数量。 + - **output_num** - 期望输出的预测结果数量(Top-K 中的 K)。 + - **weight_threshold** - 一个浮点数阈值,用于过滤掉置信度过低的预测结果。 + - **core_mask** - 核掩码 (仅共享存储版本需要)。 + +输出: + - **output_label** - 存储最终预测标签的整数数组地址。 + - **output_weight** - 存储最终预测权重的浮点数数组地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分数据类型,但其处理的 ``LabelInfo`` 结构体具有固定的成员类型(int, float)。 + +**共享存储版本:** + +.. c:function:: void custompredict_s(LabelInfo * input, int input_size, int *output_label, float *output_weight, int output_num, float weight_threshold, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + #include + #include "custompredict.h" + + int main() { + LabelInfo *input = (LabelInfo *)0xA0000000; // input in DDR + int *output_label = (int *)0xB0000000; // output_label in DDR + float *output_weight = (float *)0xC0000000; // output_weight in DDR + + int input_size = 100; // 假设值 + int output_num = 10; + float weight_threshold = 0.5f; + int core_mask = 0xff; + + custompredict_s(input, input_size, output_label, output_weight, output_num, weight_threshold, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void custompredict_p(LabelInfo * input, int input_size, int *output_label, float *output_weight, int output_num, float weight_threshold) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + #include + #include "custompredict.h" + + int main() { + LabelInfo *input = (LabelInfo *)0x10000000; // input in L2 + int *output_label = (int *)0x10001000; // output_label in L2 + float *output_weight = (float *)0x10002000; // output_weight in L2 + + int input_size = 100; // 假设值 + int output_num = 10; + float weight_threshold = 0.5f; + + custompredict_p(input, input_size, output_label, output_weight, output_num, weight_threshold); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/deconvgradfilter.rst.txt b/master/html/_sources/functionlib/dsplib/deconvgradfilter.rst.txt new file mode 100644 index 0000000..f465cf2 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/deconvgradfilter.rst.txt @@ -0,0 +1,97 @@ +DeconvGradFilter +================= + + + +计算反卷积(Deconvolution / Transposed Convolution)算子的权重梯度,用于反向传播阶段。 +该算子根据输出梯度 ``dy`` 与输入特征 ``x``,通过 ``im2row + GEMM`` 的方式累加得到卷积核梯度 ``dw``,并支持分组(group)计算。 + +.. math:: + + \frac{\partial W}{\partial L} + = + \sum_{b=0}^{B-1} + \text{Im2Row}(dY_b) \cdot X_b + +其中每个 Group 独立计算,最终在 Group 维度上拼接。 + +输入: + - **dy_data** - 输出特征梯度地址,形状为 ``[batch, out_h, out_w, out_c]``。 + - **x_data** - 输入特征地址,形状为 ``[batch, in_h, in_w, in_c]``。 + - **param** - 参数数组地址,用于描述反卷积计算相关参数与工作空间。 + - ``param[1]`` : in_h + - ``param[2]`` : in_w + - ``param[3]`` : in_c + - ``param[4]`` : batch + - ``param[5]`` : out_h + - ``param[6]`` : out_w + - ``param[7]`` : out_c + - ``param[8]`` : kernel_h + - ``param[9]`` : kernel_w + - ``param[16]`` : group + - ``param[17]`` : im2row 工作缓冲区地址 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dw_data** - 权重梯度输出地址,布局为 ``[group, out_c/group * k_h * k_w, in_c/group]``。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 仅支持 fp 类型 + - MT7004 支持 hp, fp 类型 + - 输入与输出数据格式为 NHWC + +**共享存储版本:** + +.. c:function:: void hp_deconvgradfilter_s(half* dy_data, half* x_data, half* dw_data, long long* param, int core_mask) +.. c:function:: void fp_deconvgradfilter_s(float* dy_data, float* x_data, float* dw_data, long long* param, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *dy_data = (float *)0xA0000000; + float *x_data = (float *)0xA1000000; + float *dw_data = (float *)0xC0000000; + long long *param = (long long *)0xA2000000; + int core_mask = 0xff; + + fp_deconvgradfilter_s(dy_data, x_data, dw_data, param, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_deconvgradfilter_p(half* dy_data, half* x_data, half* dw_data, long long* param) +.. c:function:: void fp_deconvgradfilter_p(float* dy_data, float* x_data, float* dw_data, long long* param) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *dy_data = (float *)0x10810000; // L2空间 + float *x_data = (float *)0x10820000; + float *dw_data = (float *)0x10830000; + long long *param = (long long *)0x10840000; + + fp_deconvgradfilter_p(dy_data, x_data, dw_data, param); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/detection_post_process.rst.txt b/master/html/_sources/functionlib/dsplib/detection_post_process.rst.txt new file mode 100644 index 0000000..aa13d1c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/detection_post_process.rst.txt @@ -0,0 +1,156 @@ +DetectionPostProcess +====================== + +DetectionPostProcess 算子用于目标检测模型的后处理阶段,对网络预测的候选框和置信度进行解码、筛选和非极大值抑制(NMS),输出最终检测结果。 +其主要功能包括: +1. 根据 anchor 信息对候选框进行解码; +2. 对各类别的候选框执行 NMS 操作; +3. 输出选中的检测框、类别及分数。 + +输入: + - **input_boxes** - 检测框坐标数据地址,形状为 ``[num_boxes, 4]``。 + - **input_scores** - 每个检测框对应的置信度分数,形状为 ``[num_boxes, num_classes_with_bg_]``。 + - **anchors** - Anchor 坐标数据地址,形状为 ``[num_boxes, 4]``。 + - **num_boxes** - 候选框数量。 + - **num_classes_with_bg** - 含背景类在内的类别数量。 + - **core_mask (int, 可选)** - 核掩码(仅适用于共享存储版本)。 + - **param** - 算子参数结构体,包含 NMS 阈值、检测数量限制、坐标缩放因子等。 + + **结构体定义:** + + .. code-block:: c + :linenos: + + typedef struct { + bool use_regular_nms; + int num_classes; + int max_detections; + int max_classes_per_detection; // Fast NMS使用 + int detections_per_class; // Regular NMS使用 + float nms_score_threshold; + float nms_iou_threshold; + float y_scale; + float x_scale; + float h_scale; + float w_scale; + int num_boxes; + int num_classes_with_bg; + + void *decoded_boxes; + uint8_t *nms_candidate; + int32_t *selected; + float *scores; + int32_t *indexes; + float *all_class_scores; + int32_t *all_class_indexes; + int32_t *single_class_indexes; + + // INT8 量化参数(仅在 Int8 模式下使用) + float boxes_scale; + int32_t boxes_zero_point; + float scores_scale; + int32_t scores_zero_point; + + } DetectionPostProcessParameter; +输出: + - **output_boxes** - 输出检测框坐标。 + - **output_classes** - 输出检测框类别索引。 + - **output_scores** - 输出检测框对应的置信度。 + - **output_num** - 实际输出的检测框数量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32, int8 + - MT7004 支持 fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_detection_post_process_s(const float *input_boxes, const float *input_scores, const float *anchors, float *output_boxes, float *output_classes, float *output_scores, float *output_num, DetectionPostProcessParameter *param, int core_mask) +.. c:function:: void i8_detection_post_process_s(const int8_t *input_boxes, const int8_t *input_scores, const float *anchors, float *output_boxes, float *output_classes, float *output_scores, float *output_num, DetectionPostProcessParameter *param, int core_mask) +.. c:function:: void hp_detection_post_process_s(const half *input_boxes, const half *input_scores, const half *anchors, half *output_boxes, float *output_classes, half *output_scores, float *output_num, DetectionPostProcessParameter *param, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 30 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input_boxes = (float *)0xA0000000; + float *input_scores = (float *)0xA1000000; + float *anchors = (float *)0xA2000000; + float *output_boxes = (float *)0xA3000000; + float *output_classes = (float *)0xA4000000; + float *output_scores = (float *)0xA5000000; + float *output_num = (float *)0xA6000000; + DetectionPostProcessParameter* param = (DetectionPostProcessParameter*)0xA7000000; + + param->use_regular_nms = true; + param->num_classes = 3; + param->max_detections = 50; + param->max_classes_per_detection = 1; + param->detections_per_class = 50; + param->nms_score_threshold = 0.1f; + param->nms_iou_threshold = 0.5f; + param->y_scale = 10.0f; + param->x_scale = 10.0f; + param->h_scale = 5.0f; + param->w_scale = 5.0f; + param->num_boxes = 100; + param->num_classes_with_bg = 4; + + int core_mask = 0xff; + + fp_detection_post_process_s(input_boxes, input_scores, anchors, output_boxes, output_classes, output_scores, output_num, param, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_detection_post_process_p(const float *input_boxes, const float *input_scores, const float *anchors, float *output_boxes, float *output_classes, float *output_scores, float *output_num, DetectionPostProcessParameter *param) +.. c:function:: void i8_detection_post_process_p(const int8_t *input_boxes, const int8_t *input_scores, const float *anchors, float *output_boxes, float *output_classes, float *output_scores, float *output_num, DetectionPostProcessParameter *param) +.. c:function:: void hp_detection_post_process_p(const half *input_boxes, const half *input_scores, const half *anchors, half *output_boxes, float *output_classes, half *output_scores, float *output_num, DetectionPostProcessParameter *param) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 29 + + //MT7004示例 + #include + + int main(int argc, char* argv[]) { + DetectionPostProcessParameter param; + + param.use_regular_nms = 1; + param.num_classes = 3; + param.max_detections = 50; + param.max_classes_per_detection = 1; + param.detections_per_class = 50; + param.nms_score_threshold = 0.1f; + param.nms_iou_threshold = 0.5f; + param.y_scale = 10.0f; + param.x_scale = 10.0f; + param.h_scale = 5.0f; + param.w_scale = 5.0f; + param.num_boxes = 1917; + param.num_classes_with_bg = 4; + + float *input_boxes = (float *)0xA0000000; + float *input_scores = (float *)0xA1000000; + float *anchors = (float *)0xA2000000; + float *output_boxes = (float *)0xA3000000; + float *output_classes = (float *)0xA4000000; + float *output_scores = (float *)0xA5000000; + float *output_num = (float *)0xA6000000; + + fp_detection_post_process_p(input_boxes, input_scores, anchors, output_boxes, output_classes, output_scores, output_num, param); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/div_fusion.rst.txt b/master/html/_sources/functionlib/dsplib/div_fusion.rst.txt new file mode 100644 index 0000000..1d68155 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/div_fusion.rst.txt @@ -0,0 +1,90 @@ +DivFusion +================= +对输入逐元素做除法运算,并对实数类型结果应用 ReLU激活。 + +.. math:: + + \text{对于实数类型:}\quad output_i = \max\left(\frac{input0_i}{input1_i}, 0\right) + +.. math:: + + \text{对于复数类型:}\quad output_i = \frac{input0_i}{input1_i} + +输入: + - **input0** - 被除数输入数据地址。 + - **input1** - 除数输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 对于复数类型(cplx64 / cplx128)不应用 ReLU,仅返回复数除法结果。 + - 若除数元素为 0,结果为 Inf/NaN 或未定义,需由上层处理。 + +**共享存储版本:** + +.. c:function:: void i8_div_fusion_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_div_fusion_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_div_fusion_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_div_fusion_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_div_fusion_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_div_fusion_s(double* input0, double* input1, double* output, int length, int core_mask) +.. c:function:: void c64_div_fusion_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void c128_div_fusion_s(double* input0, double* input1, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例(共享存储) + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // input0 在 DDR 空间 + float *input1 = (float *)0xA1000000; // input1 在 DDR 空间 + float *output = (float *)0xB0000000; // 输出在 DDR 空间 + int length = 1024; + int core_mask = 0xff; + fp_div_fusion_s(input0, input1, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_div_fusion_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_div_fusion_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_div_fusion_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_div_fusion_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_div_fusion_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_div_fusion_p(double* input0, double* input1, double* output, int length) +.. c:function:: void c64_div_fusion_p(float* input0, float* input1, float* output, int length) +.. c:function:: void c128_div_fusion_p(double* input0, double* input1, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //MT7004 示例 + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; + float *input1 = (float *)0x10001000; + float *output = (float *)0x10002000; + int length = 1024; + fp_div_fusion_p(input0, input1, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/divgrad.rst.txt b/master/html/_sources/functionlib/dsplib/divgrad.rst.txt new file mode 100644 index 0000000..88439b2 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/divgrad.rst.txt @@ -0,0 +1,161 @@ +Divgrad +================= + + +计算逐元素除法操作 (`Y = x1 / x2`) 的梯度。 + +.. math:: + + dx_1 = \frac{\partial L}{\partial x_1} = \frac{\partial L}{\partial Y} \cdot \frac{1}{x_2} = \frac{dy}{x_2} + +.. math:: + + dx_2 = \frac{\partial L}{\partial x_2} = \frac{\partial L}{\partial Y} \cdot \frac{-x_1}{x_2^2} = - \frac{dy \cdot x_1}{x_2^2} + +Divgrad1l版本专门用于 `x1` 张量维度大于或等于 `x2` 张量的广播场景。Divgrad2l版本专门用于 `x2` 张量维度大于或等于 `x1` 张量的广播场景。 + +输入: + - **dy** - 来自后一层的上游梯度张量。 + - **x1** - 前向传播时的第一个输入张量(被除数)。 + - **x2** - 前向传播时的第二个输入张量(除数)。 + - **large_shape** - `x1` 和 `x2` 中维度较大的张量的形状。 + - **small_shape** - `x1` 和 `x2` 中维度较小的张量的形状。 + - **out_shape** - 输出张量 `dx1` 和 `dx2` 的形状。 + - **ndims** - 张量的维度数。 + - **large_strides** - 维度较大张量的步长信息。 + - **small_strides** - 维度较小张量的步长信息。 + - **out_strides** - 输出张量的步长信息。 + - **large_multiples** - 维度较大张量的广播倍数。 + - **small_multiples** - 维度较小张量的广播倍数。 + - **tile_data0** - 临时工作空间地址。 + - **tile_data1** - 临时工作空间地址。 + - **tile_data2** - 临时工作空间地址。 + - **indices** - 用于广播计算的临时索引空间地址。 + - **core_mask** - 核掩码。 + +输出: + - **dx1** - 写入计算出的对 `x1` 的梯度。 + - **dx2** - 写入计算出的对 `x2` 的梯度。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_graddiv_s(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, float* tile_data2, int* indices, int core_mask) +.. c:function:: void hp_graddiv_s(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, half* tile_data2, int* indices, int core_mask) +.. c:function:: void fp_graddiv1l_s(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, float* tile_data2, int* indices, int core_mask) +.. c:function:: void hp_graddiv1l_s(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, half* tile_data2, int* indices, int core_mask) +.. c:function:: void fp_graddiv2l_s(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, float* tile_data2, int* indices, int core_mask) +.. c:function:: void hp_graddiv2l_s(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, half* tile_data2, int* indices, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 39 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0xA1000000; + float *dx1 = (float *)0xA2000000; + float *dx2 = (float *)0xA3000000; + float *x1_data = (float *)0xA4000000; + float *x2_data = (float *)0xA5000000; + float *tile_data0 = (float *)0xA6000000; + float *tile_data1 = (float *)0xA7000000; + float *tile_data2 = (float *)0xA8000000; + + long long ndims = 4; + long long dy_size; + long long x1_size; + long long x2_size; + + int *large_strides = (int *)0xAB000000; + int *small_strides = (int *)0xAB100000; + int *out_strides = (int *)0xAB200000; + int *large_multiples = (int *)0xAB300000; + int *small_multiples = (int *)0xAB400000; + int *indices = (int *)0xAB500000; + int *large_shape = (int *)0xAB600000; + int *small_shape = (int *)0xAB700000; + int *out_shape = (int *)0xAB800000; + + large_shape[0] = 12; large_shape[1] = 14; large_shape[2] = 3; large_shape[3] = 5; + small_shape[0] = 12; small_shape[1] = 14; small_shape[2] = 3; small_shape[3] = 5; + out_shape[0] = 12; out_shape[1] = 14; out_shape[2] = 3; out_shape[3] = 5; + + int core_mask = 0xff; + + dy_size = out_shape[0] * out_shape[1] * out_shape[2] * out_shape[3]; + x1_size = large_shape[0] * large_shape[1] * large_shape[2] * large_shape[3]; + x2_size = small_shape[0] * small_shape[1] * small_shape[2] * small_shape[3]; + + fp_graddiv_s(dy, x1, x2, large_shape, small_shape, out_shape, ndims, large_strides, small_strides, out_strides, large_multiples, small_multiples, dx1, dx2, tile_data0, tile_data1, tile_data2, indices, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_graddiv_p(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, float* tile_data2, int* indices) +.. c:function:: void hp_graddiv_p(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, half* tile_data2) +.. c:function:: void fp_graddiv1l_p(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, float* tile_data2, int* indices) +.. c:function:: void hp_graddiv1l_p(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, half* tile_data2, int* indices) +.. c:function:: void fp_graddiv2l_p(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, float* tile_data2, int* indices) +.. c:function:: void hp_graddiv2l_p(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, half* tile_data2, int* indices) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 37 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0x10000000; + float *dx1 = (float *)0x12000000; + float *dx2 = (float *)0x13000000; + float *x1_data = (float *)0x14000000; + float *x2_data = (float *)0x15000000; + float *tile_data0 = (float *)0x16000000; + float *tile_data1 = (float *)0x17000000; + float *tile_data2 = (float *)0x18000000; + + long long ndims = 4; + long long dy_size; + long long x1_size; + long long x2_size; + + int *large_strides = (int *)0x1B000000; + int *small_strides = (int *)0x1B100000; + int *out_strides = (int *)0x1B200000; + int *large_multiples = (int *)0x1B300000; + int *small_multiples = (int *)0x1B400000; + int *indices = (int *)0x1B500000; + int *large_shape = (int *)0x1B600000; + int *small_shape = (int *)0x1B700000; + int *out_shape = (int *)0x1B800000; + + large_shape[0] = 12; large_shape[1] = 14; large_shape[2] = 3; large_shape[3] = 5; + small_shape[0] = 12; small_shape[1] = 14; small_shape[2] = 3; small_shape[3] = 5; + out_shape[0] = 12; out_shape[1] = 14; out_shape[2] = 3; out_shape[3] = 5; + + dy_size = out_shape[0] * out_shape[1] * out_shape[2] * out_shape[3]; + x1_size = large_shape[0] * large_shape[1] * large_shape[2] * large_shape[3]; + x2_size = small_shape[0] * small_shape[1] * small_shape[2] * small_shape[3]; + + fp_graddiv_p(dy, x1, x2, large_shape, small_shape, out_shape, ndims, large_strides, small_strides, out_strides, large_multiples, small_multiples, dx1, dx2, tile_data0, tile_data1, tile_data2, indices); + return 0; + } + + \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/dropout.rst.txt b/master/html/_sources/functionlib/dsplib/dropout.rst.txt new file mode 100644 index 0000000..097bf20 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/dropout.rst.txt @@ -0,0 +1,86 @@ +Dropout +================= + + + +在训练期间对输入张量应用 Dropout 操作。它根据给定的 `mask` 将输入张量中的部分元素置零,并对剩余元素进行缩放。 + +.. math:: + + \text{output}_i = \text{input}_i \cdot \text{mask}_i \cdot \text{scale} + +输入: + - **input** - 输入张量的数据地址。 + - **scale** - 缩放因子。 + - **length** - 输入张量的总元素数量。 + - **mask** - 预先生成的掩码张量的数据地址,其元素为0或1,大小与`input`相同。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与`input`相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_dropout_s(float* input, float scale, int length, float* output, float mask, int core_mask) +.. c:function:: void hp_dropout_s(half* input, half scale, int length, half* output, half* mask, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在DDR空间 + float *output = (float *)0xB0000000; // output + float *mask = (float *)0xC0000000; // pre-generated mask + + int length = 4096; + // 假设 dropout 概率 p = 0.2 + float scale = 1.0f / (1.0f - 0.2f); // scale = 1.25 + int core_mask = 0xff; + + fp_dropout_s(input, scale, length, output, mask, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_dropout_p(float* input, float scale, int length, float* output, float mask) +.. c:function:: void hp_dropout_p(half* input, half scale, int length, half* output, half mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // input 在L2空间 + float *output = (float *)0x11000000; // output + float *mask = (float *)0x12000000; // pre-generated mask + + int length = 1024; + // 假设 dropout 概率 p = 0.5 + float scale = 1.0f / (1.0f - 0.5f); // scale = 2.0 + + fp_dropout_p(input, scale, length, output, mask); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/dropoutgrad.rst.txt b/master/html/_sources/functionlib/dsplib/dropoutgrad.rst.txt new file mode 100644 index 0000000..ffaebeb --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/dropoutgrad.rst.txt @@ -0,0 +1,89 @@ +Dropoutgrad +================= + + +计算 Dropout 操作的梯度。 + +.. math:: + + dx_i = dy_i \cdot \text{mask}_i \cdot \text{scale} + +其中 :math:`dy_i` 是来自后一层的上游梯度,:math:`\text{mask}_i` 是前向传播时使用的同一个掩码,:math:`\text{scale}` 是前向传播时使用的同一个缩放因子。 + +输入: + - **input** - 上游梯度张量的数据地址 (dy)。 + - **scale** - 前向传播时使用的缩放因子。 + - **length** - 张量的总元素数量。 + - **mask** - 前向传播时使用的掩码张量的数据地址。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出的梯度张量的数据地址 (dx)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_dropoutgrad_s(float* input, float scale, int length, float* output, float mask, int core_mask) +.. c:function:: void hp_dropoutgrad_s(half* input, half scale, int length, half* output, half mask, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0xA0000000; // Upstream gradient (dy), DDR + float *dx = (float *)0xB0000000; // Output gradient (dx) + float *mask = (float *)0xC0000000; // Mask from forward pass + + int length = 4096; + // 假设前向传播时 dropout 概率 p = 0.2 + float scale = 1.0f / (1.0f - 0.2f); // scale = 1.25 + int core_mask = 0xff; + + fp_dropoutgrad_s(dy, scale, length, dx, mask, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_dropoutgrad_p(float* input, float scale, int length, float* output, float mask) +.. c:function:: void hp_dropoutgrad_p(half* input, half scale, int length, half* output, half mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0x10000000; // Upstream gradient (dy), L2 + float *dx = (float *)0x11000000; // Output gradient (dx) + float *mask = (float *)0x12000000; // Mask from forward pass + + int length = 1024; + // 假设前向传播时 dropout 概率 p = 0.5 + float scale = 1.0f / (1.0f - 0.5f); // scale = 2.0 + + fp_dropoutgrad_p(dy, scale, length, dx, mask); + return 0; + } + + \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt b/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt index b63f4c0..2b214a1 100644 --- a/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt +++ b/master/html/_sources/functionlib/dsplib/dsplib_index.rst.txt @@ -1,6 +1,8 @@ DSP Library C API Reference =========================== +.. rubric:: 算子列表 + .. toctree:: :maxdepth: 1 @@ -14,7 +16,6 @@ DSP Library C API Reference unsqueeze expand_dims leaky_relu - lstm scatter_elements reduce resize @@ -50,3 +51,174 @@ DSP Library C API Reference conv2dbackpropfilterfusion sgd scalefusion + hashtablelookup + neg + neg_grad + not_equal + pow_fusion + power_grad + real_div + div_fusion + reciprocal + uniform_real + random_standard_normal + random_normal + tensor_scatter_add + non_max_suppression + detection_post_process + mfcc + audio_spectrogram + cumsum + smooth1loss + tril + triu + softmax_cross_entropy_with_logits + sparse_softmax_cross_entropy_with_logits + concat + gather + gather_nd + ones_like + reshape + scatter_nd + scatter_nd_update + slice + split + split_with_overlap + stack + tile + adam + constant_of_shape + rank + size + maxpoolfusion + maxpoolgrad + onehot + affine + assign + assignadd + padfusion + sigmoidcrossentropywithlogits + absgrad + addfusion + addgrad + all + argmax + argmin + avgpooling + batchnormgrad + biasadd + biasaddgrad + binarycrossentropy + binarycrossentropygrad + ceil + clip + customnormalize + custompredict + divgrad + dropout + dropoutgrad + elu + rfft + floormod + glu + greater + greaterequal + isfinite + l2norm + layernormgrad + less + lessequal + log1p + lrn + lstm + lstmgrad + lstmgradweight + lstmgraddata + maximumgrad + minimumgrad + mulgrad + priorbox + resizegrad + rsqrt + rsqrtgrad + select + shape + skipgram + smoothl1lossgrad + sparsereshape + sparsesegmentsum + sparsetodense + sqrtgrad + square + squaredifference + stridedslice + stridedslicegrad + subfusion + subgrad + switch + switchlayer + tensorarray + tensorarrayread + tensorarraywrite + tensorlistfromtensor + tensorlistgetitem + tensorlistreserve + tensorlistsetitem + tensorliststack + unique + where + dynamicquant + log + erf + quantdtypecast + customextractfeatures + logicaland + round + loggrad + topkfusion + unsortedsegmentsum + gatherd + fill + sparsefillemptyrows + fullconnection + sqrt + sin + nllloss + nlllossgrad + deconvgradfilter + cos + zeroslike + abs + fftreal + fftimag + batchnorm + instancenorm + lpnormalization + nonzero + allgather + addn + cast + sigmoidcrossentropwithlogitsgrad + flatten + flattengrad + unstack + invertpermutation + logsoftmax + softmax + layernormfusion + prelufusion + reducescatter + transpose + formattranspose + splice + roipooling + activation_grad + fake_quant_with_min_max_vars + fake_quant_with_min_max_vars_per_channel + logical_not + logical_or + lsh_projection + maximum + minimum + mod + mul \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/dynamicquant.rst.txt b/master/html/_sources/functionlib/dsplib/dynamicquant.rst.txt new file mode 100644 index 0000000..e700a46 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/dynamicquant.rst.txt @@ -0,0 +1,138 @@ +DynamicQuant +================= + +将 FP32 输入张量按给定的 scale 和 zero point(zp)动态量化为 INT8 输出张量。 +支持按整张量量化或按指定轴分段量化(per-axis dynamic quantization)。 + +.. math:: + + q_i = \mathrm{clip}\left( \mathrm{round}\left( \frac{x_i}{scale} + zp \right), q_{min}, q_{max} \right) + +其中: + +- :math:`x_i` 为输入浮点值 +- :math:`scale` 为量化比例因子 +- :math:`zp` 为零点(zero point) +- :math:`q_{min} = -128` +- :math:`q_{max} = 127` + +当输入为 :math:`+\infty` 或 :math:`-\infty` 时,输出分别饱和到最大或最小量化值。 + +--- + +输入: + - **real_values** - 输入 FP32 数据地址 + - **element_num** - 输入元素总数 + - **scale** - 量化 scale 数组地址 + - **zp** - 量化 zero point 数组地址 + - **segment_num** - 分段数量(用于按轴量化) + - **axis_num** - 量化轴标识 + - ``0``:整张量量化(per-tensor) + - ``!=0``:按轴分段量化(per-axis) + +输出: + - **quant_values** - 输出 INT8 量化结果地址 + +--- + +支持平台: + ``FT78NE`` + ``MT7004`` + +--- + +.. note:: + - 当前算子仅支持 **FP32\Fp16 → INT8** 的动态量化 + - 当 ``axis_num = 0`` 时,仅使用 ``scale[0]`` 与 ``zp[0]`` 进行整张量量化 + - 当 ``axis_num != 0`` 时,输入按 ``segment_num`` 进行分段,每段使用对应的 ``scale[i]`` 与 ``zp[i]`` + - 分段大小通过 ``UP_DIV(element_num, segment_num)`` 计算,最后一段自动处理剩余元素 + - 量化结果会被限制在 ``[-128, 127]`` 范围内 + +--- + +**私有存储版本:** + +.. c:function:: void fp_QuantData_p(float* real_values, int8_t* quant_values, int element_num, float* scale, int* zp, int segment_num, int axis_num) +.. c:function:: void hp_QuantData_p(half* real_values, half* quant_values, int element_num, half* scale, int* zp, int segment_num, int axis_num) + +--- + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // FP32 输入 + int8_t *output = (int8_t *)0x10004000; // INT8 输出 + float scale[4] = {0.1f, 0.12f, 0.11f, 0.09f}; + int zp[4] = {0, 0, 0, 0}; + int element_num = 1024; + int segment_num = 4; + int axis_num = 1; + + fp_QuantData_p(input, output, element_num, + scale, zp, segment_num, axis_num); + + return 0; + } + +--- + +**共享存储版本:** + +.. c:function:: void fp_QuantData_s(float* real_values, int8_t* quant_values, int element_num, float* scale, int* zp, int segment_num, int axis_num, int core_mask) +.. c:function:: void hp_QuantData_s(half* real_values, half* quant_values, int element_num, half* scale, int* zp, int segment_num, int axis_num, int core_mask) + +--- + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // FP32 输入 + int8_t *output = (int8_t *)0x10004000; // INT8 输出 + float scale[4] = {0.1f, 0.12f, 0.11f, 0.09f}; + int zp[4] = {0, 0, 0, 0}; + int element_num = 1024; + int segment_num = 4; + int axis_num = 1, core_mask = 0xff; + + fp_QuantData_s(input, output, element_num, + scale, zp, segment_num, axis_num, core_mask); + + return 0; + } + +--- + +**实现说明:** + +- 当 ``axis_num == 0`` 时: + + - 对整个输入数组执行一次统一量化 + - 等价于 per-tensor quantization + +- 当 ``axis_num != 0`` 时: + + - 输入数据按 ``segment_num`` 均分为多个分段 + - 每个分段使用独立的 ``scale[i]`` 与 ``zp[i]`` + - 等价于 per-axis dynamic quantization + +- 核心量化过程由 ``DoQuantizeFp32ToInt8`` 完成,包括: + + - 反 scale 计算 + - 四舍五入 + - 饱和裁剪 + - INF 特殊值处理 + diff --git a/master/html/_sources/functionlib/dsplib/eltwise.rst.txt b/master/html/_sources/functionlib/dsplib/eltwise.rst.txt index dbae68d..b731d87 100644 --- a/master/html/_sources/functionlib/dsplib/eltwise.rst.txt +++ b/master/html/_sources/functionlib/dsplib/eltwise.rst.txt @@ -7,9 +7,9 @@ Eltwise \mathbf{output_i} = \begin{cases} - \mathbf{Input0_i} \cdot \mathbf{Input1_i}, & \text{if } \text{eltwise_mode} = \text{Eltwise_PROD} \\[6pt] - \mathbf{Input0_i} + \mathbf{Input1_i}, & \text{if } \text{eltwise_mode} = \text{Eltwise_SUM} \\[6pt] - \max(\mathbf{Input0_i}, \mathbf{Input1_i}), & \text{if } \text{eltwise_mode} = \text{Eltwise_MAXIMUM} + \mathbf{Input0_i} \cdot \mathbf{Input1_i}, & \text{if } \text{eltwise\_mode} = \text{Eltwise\_PROD} \\[6pt] + \mathbf{Input0_i} + \mathbf{Input1_i}, & \text{if } \text{eltwise\_mode} = \text{Eltwise\_SUM} \\[6pt] + \max(\mathbf{Input0_i}, \mathbf{Input1_i}), & \text{if } \text{eltwise\_mode} = \text{Eltwise\_MAXIMUM} \end{cases} 输入: diff --git a/master/html/_sources/functionlib/dsplib/elu.rst.txt b/master/html/_sources/functionlib/dsplib/elu.rst.txt new file mode 100644 index 0000000..2e0748e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/elu.rst.txt @@ -0,0 +1,84 @@ +Elu +================= + + + +逐元素计算指数线性单元 (Exponential Linear Unit, ELU) 激活函数。 + +.. math:: + + \text{output}_i = \begin{cases} \text{input}_i & \text{if } \text{input}_i \ge 0 \\ \alpha \cdot (e^{\text{input}_i} - 1) & \text{if } \text{input}_i < 0 \end{cases} + +其中 :math:`\alpha` (alpha) 是一个可调参数,控制负值的饱和度。 + +输入: + - **input** - 输入张量的数据地址。 + - **alpha** - ELU函数的alpha值。 + - **length** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与`input`相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_elu_s(float* input, float* output, float alpha, int length, int core_mask) +.. c:function:: void hp_elu_s(half* input, half* output, half alpha, int length, int core_mask) +.. c:function:: void i8_elu_s(int8_t* input, int8_t* output, int8_t alpha, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在DDR空间 + float *output = (float *)0xB0000000; // output + + float alpha = 1.0f; + int length = 4096; + int core_mask = 0xff; + + fp_elu_s(input, output, alpha, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_elu_p(float* input, float* output, float alpha, int length) +.. c:function:: void hp_elu_p(half* input, half* output, half alpha, int length) +.. c:function:: void i8_elu_p(int8_t* input, int8_t* output, int8_t alpha, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // input 在L2空间 + float *output = (float *)0x11000000; // output + + float alpha = 1.0f; + int length = 1024; + + fp_elu_p(input, output, alpha, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/embeddinglookup.rst.txt b/master/html/_sources/functionlib/dsplib/embeddinglookup.rst.txt index 858cc87..70e6ff7 100644 --- a/master/html/_sources/functionlib/dsplib/embeddinglookup.rst.txt +++ b/master/html/_sources/functionlib/dsplib/embeddinglookup.rst.txt @@ -7,14 +7,14 @@ EmbeddingLookup \forall k \in [1, ids\_size], \quad \begin{cases} - \text{if } \textbf{is_regulated}[i_k] = 0, & + \text{if } \textbf{is\_regulated}[i\_k] = 0, & \begin{cases} - \displaystyle X_{i_k} \leftarrow - X_{i_k} \cdot \frac{\text{max_norm}} - {\sum_{j=1}^{layer\_size\_} X_{i_k, j}} \\[10pt] - \textbf{is_regulated}[i_k] \leftarrow 1 + \displaystyle X_{i\_k} \leftarrow + X_{i\_k} \cdot \frac{\text{max\_norm}} + {\sum_{j=1}^{layer\_size\_} X_{i\_k, j}} \\[10pt] + \textbf{is\_regulated}[i\_k] \leftarrow 1 \end{cases} \\[12pt] - \text{输出向量 } Y_k \leftarrow X_{i_k} + \text{输出向量 } Y_k \leftarrow X_{i\_k} \end{cases} 输入: diff --git a/master/html/_sources/functionlib/dsplib/erf.rst.txt b/master/html/_sources/functionlib/dsplib/erf.rst.txt new file mode 100644 index 0000000..35e6fb4 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/erf.rst.txt @@ -0,0 +1,86 @@ +Erf +================= + + +逐元素计算误差函数(Error Function, erf)。 + +误差函数在概率论与统计中广泛使用,其定义如下: + +.. math:: + + \text{output}_i = \operatorname{erf}(\text{input}_i) + = \frac{2}{\sqrt{\pi}} \int_{0}^{\text{input}_i} e^{-t^2} \, dt + + +输入: + - **input** - 输入张量的数据地址。 + - **length** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与 ``input`` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:fp32, fp64 + - MT7004 支持的数据类型:fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_erf_s(float* input, float* output, int length, int core_mask) +.. c:function:: void dp_erf_s(double* input, double* output, int length, int core_mask) +.. c:function:: void hp_erf_s(half* input, half* output, int length, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // C6678 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + float *output = (float *)0xB0000000; // output 在 DDR 空间 + + int length = 4096; + int core_mask = 0xff; + + fp_erf_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_erf_p(float* input, float* output, int length) +.. c:function:: void dp_erf_p(double* input, double* output, int length) +.. c:function:: void hp_erf_p(half* input, half* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input = (half *)0x10000000; // input 在 L2 空间 + half *output = (half *)0x10001000; // output 在 L2 空间 + + int length = 1024; + + hp_erf_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/expfusion.rst.txt b/master/html/_sources/functionlib/dsplib/expfusion.rst.txt index 9c67ae1..92d5f94 100644 --- a/master/html/_sources/functionlib/dsplib/expfusion.rst.txt +++ b/master/html/_sources/functionlib/dsplib/expfusion.rst.txt @@ -3,38 +3,38 @@ ExpFusion - 传入一个数组,逐元素计算其乘上输入因子(可选择)后的指数值,再将指数值乘上输出因子后输出。 +传入一个数组,逐元素计算其乘上输入因子(可选择)后的指数值,再将指数值乘上输出因子后输出。 - .. math:: +.. math:: - dst_i = \exp(src_i \cdot s_{in}) \cdot s_{out} - \quad \text{where} \quad - s_{in} = - \begin{cases} - 1, & scale = 1 \\ - in\_scale, & scale \neq 1 - \end{cases} + dst_i = \exp(src_i \cdot s_{in}) \cdot s_{out} + \quad \text{where} \quad + s_{in} = + \begin{cases} + 1, & scale = 1 \\ + in\_scale, & scale \neq 1 + \end{cases} - \quad s_{out} = out\_scale + \quad s_{out} = out\_scale - 输入: - - **src_data** - 输入数据地址。 - - **length** - 计算长度。 - - **in_scale** - 输入缩放因子,当scale != 1时启用。 - - **out_scale** - 输出缩放因子。 - - **scale** - 输入缩放因子启用控制。 - - **core_mask** - 核掩码(仅适用于共享存储版本)。 +输入: + - **src_data** - 输入数据地址。 + - **length** - 计算长度。 + - **in_scale** - 输入缩放因子,当scale != 1时启用。 + - **out_scale** - 输出缩放因子。 + - **scale** - 输入缩放因子启用控制。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 - 输出: - - **dst_data** - 计算结果地址。 +输出: + - **dst_data** - 计算结果地址。 - 支持平台: - ``FT78NE`` - ``MT7004`` +支持平台: + ``FT78NE`` + ``MT7004`` - .. note:: - - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 - - MT7004 支持fp16, fp32, int16, int32, cplx64 +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 **共享存储版本:** @@ -47,26 +47,26 @@ ExpFusion .. c:function:: void c64_expfusion_s(float* src_data, float* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) .. c:function:: void c128_expfusion_s(double* src_data, double* dst_data, int length, float in_scale, float out_scale, int scale, int core_mask) - **C调用示例:** +**C调用示例:** - .. code-block:: c - :linenos: - :emphasize-lines: 12 +.. code-block:: c + :linenos: + :emphasize-lines: 12 - //FT78NE示例 - #include - #include - - int main(int argc, char* argv[]) { - float *input0 = (float *)0xA0000000; //input在DDR空间 - float *output = (float *)0xC0000000; - int length = 1000; - float in_scale = 0.5, out_scale = 1.2; - int scale = 1; - int core_mask = 0xff; - fp_expfusion_s( input0, output, length, in_scale, out_scale, scale,core_mask); - return 0; - } + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1000; + float in_scale = 0.5, out_scale = 1.2; + int scale = 1; + int core_mask = 0xff; + fp_expfusion_s( input0, output, length, in_scale, out_scale, scale,core_mask); + return 0; + } **私有存储版本:** @@ -81,21 +81,21 @@ ExpFusion .. c:function:: void c128_expfusion_p(double* src_data, double* dst_data, int length, float in_scale, float out_scale, int scale) - **C调用示例:** +**C调用示例:** - .. code-block:: c - :linenos: - :emphasize-lines: 10 +.. code-block:: c + :linenos: + :emphasize-lines: 10 - //FT78NE示例 - #include - #include - int main(int argc, char* argv[]) { - float *input0 = (float *)0x10810000; //input在L2空间 - float *output = (float *)0x10820000; - int length = 1000; - float in_scale = 0.5, out_scale = 1.2; - int scale = 1; - fp_expfusion_p( input0, output, length, in_scale, out_scale, scale); - return 0; - } \ No newline at end of file + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10810000; //input在L2空间 + float *output = (float *)0x10820000; + int length = 1000; + float in_scale = 0.5, out_scale = 1.2; + int scale = 1; + fp_expfusion_p( input0, output, length, in_scale, out_scale, scale); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/fake_quant_with_min_max_vars.rst.txt b/master/html/_sources/functionlib/dsplib/fake_quant_with_min_max_vars.rst.txt new file mode 100644 index 0000000..4aca1c3 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fake_quant_with_min_max_vars.rst.txt @@ -0,0 +1,86 @@ +FakeQuantWithMinMaxVars +======================= + +对输入数据执行逐元素伪量化运算。该算子通过给定的最小/最大值(min_val/max_val)计算缩放因子(scale)和零点(zero_point),将浮点输入模拟量化到指定的整数范围(quant_min/quant_max),然后再将其反量化回浮点数。 + +.. math:: + + scale = \frac{max\_val - min\_val}{quant\_max - quant\_min} + +.. math:: + + output_i = \left( \text{round} \left( \frac{\text{clamp}(input_i, nudge\_min, nudge\_max) - nudge\_min}{scale} \right) \right) \times scale + nudge\_min + +输入: + - **src** - 输入数据地址。 + - **min_val** - 浮点范围的最小值。 + - **max_val** - 浮点范围的最大值。 + - **length** - 计算长度。 + - **quant_min** - 量化后的整数最小值(例如 0 或 -128)。 + - **quant_max** - 量化后的整数最大值(例如 255 或 127)。 + - **symmetric** - 是否使用对称量化(bool 类型)。若为 true,则范围调整为关于 0 对称。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 伪量化后的计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:fp32 (fp) + - MT7004 支持:fp16 (hp), fp32 (fp) + - 该算子内部包含 "Nudge" 逻辑,即会自动调整零点(Zero Point)使其为整数,并根据调整后的零点重新计算实际使用的浮点范围(nudge_min/nudge_max)。 + +**共享存储版本:** + +.. c:function:: void fp_fake_quant_with_min_max_vars_s(float* src, float min_val, float max_val, float* output, int length, int quant_min, int quant_max, bool symmetric, int core_mask) +.. c:function:: void hp_fake_quant_with_min_max_vars_s(half* src, half min_val, half max_val, half* output, int length, int quant_min, int quant_max, bool symmetric, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + // FT78NE 示例:fp32 类型共享存储多核计算 + #include + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + float *output = (float *)0xB0000000; + float min_v = -10.0f; + float max_v = 10.0f; + int length = 960001; + int core_mask = 0b1011; + fp_fake_quant_with_min_max_vars_s(input, min_v, max_v, output, length, 0, 255, false, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_fake_quant_with_min_max_vars_p(float* src, float min_val, float max_val, float* output, int length, int quant_min, int quant_max, bool symmetric) +.. c:function:: void hp_fake_quant_with_min_max_vars_p(half* src, half min_val, half max_val, half* output, int length, int quant_min, int quant_max, bool symmetric) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + // MT7004 示例:fp16 (half) 类型私有存储单核计算 + #include + #include + + int main(int argc, char* argv[]) { + half *input = (half *)0x10000000; + half *output = (half *)0x10001000; + half min_v = (half)-5.0f; + half max_v = (half)5.0f; + int length = 1024; + hp_fake_quant_with_min_max_vars_p(input, min_v, max_v, output, length, 0, 255, true); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.rst.txt b/master/html/_sources/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.rst.txt new file mode 100644 index 0000000..fa89b43 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.rst.txt @@ -0,0 +1,98 @@ +FakeQuantWithMinMaxVarsPerChannel +================================= + +对输入数据执行按通道(Per-Channel)的逐元素伪量化运算。该算子将输入数据划分为 ``channel_num`` 个等长的通道,每个通道根据其对应的最小/最大值(min_val[c] / max_val[c])独立计算缩放因子和零点,进行模拟量化与反量化。 + +.. math:: + + channel\_size = \frac{length}{channel\_num} + +.. math:: + + \text{对于通道 } c: \quad scale_c = \frac{max\_val_c - min\_val_c}{quant\_max - quant\_min} + +.. math:: + + output_{c,i} = \text{FakeQuant}(input_{c,i}, scale_c, nudge\_min_c) + +输入: + - **src** - 输入数据地址。 + - **min_val** - 每个通道最小值组成的数组地址。 + - **max_val** - 每个通道最大值组成的数组地址。 + - **length** - 输入数据总长度(需能被通道数整除)。 + - **quant_min** - 量化后的整数最小值。 + - **quant_max** - 量化后的整数最大值。 + - **symmetric** - 是否使用对称量化。 + - **channel_num** - 通道数量。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 伪量化后的计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:fp32 (fp) + - MT7004 支持:fp16 (hp), fp32 (fp) + - 输入数据的总长度 ``length`` 必须可以被 ``channel_num`` 整除。 + - 每个通道的逻辑(包括 Nudge 零点调整)与单变量版本的伪量化一致。 + +**共享存储版本:** + +.. c:function:: void fp_fake_quant_with_min_max_vars_per_channel_s(float* src, float* min_val, float* max_val, float* output, int length, int quant_min, int quant_max, bool symmetric, int channel_num, int core_mask) +.. c:function:: void hp_fake_quant_with_min_max_vars_per_channel_s(half* src, half* min_val, half* max_val, half* output, int length, int quant_min, int quant_max, bool symmetric, int channel_num, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 示例:fp32 类型共享存储多核计算 + #include + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + float *min_arr = (float *)0xA1000000; + float *max_arr = (float *)0xA1001000; + float *output = (float *)0xB0000000; + int length = 960000; + int channel_num = 100; + int core_mask = 0b1011; + + fp_fake_quant_with_min_max_vars_per_channel_s(input, min_arr, max_arr, output, + length, 0, 255, false, channel_num, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_fake_quant_with_min_max_vars_per_channel_p(float* src, float* min_val, float* max_val, float* output, int length, int quant_min, int quant_max, bool symmetric, int channel_num) +.. c:function:: void hp_fake_quant_with_min_max_vars_per_channel_p(half* src, half* min_val, half* max_val, half* output, int length, int quant_min, int quant_max, bool symmetric, int channel_num) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + // MT7004 示例:fp16 (half) 类型私有存储单核计算 + #include + #include + + int main(int argc, char* argv[]) { + half *input = (half *)0x10000000; + half *min_arr = (half *)0x10008000; + half *max_arr = (half *)0x10008100; + half *output = (half *)0x10009000; + int length = 2000; + int channel_num = 10; + + hp_fake_quant_with_min_max_vars_per_channel_p(input, min_arr, max_arr, output, + length, 0, 255, true, channel_num); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/fftimag.rst.txt b/master/html/_sources/functionlib/dsplib/fftimag.rst.txt new file mode 100644 index 0000000..88af4c7 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fftimag.rst.txt @@ -0,0 +1,104 @@ +FFTImag +========================= + +对输入复数序列虚数部分执行一维快速傅里叶变换(FFT)或逆变换(IFFT)。 +内部基于分治 FFT 算法, +并使用预计算旋转因子(twiddle factor)以提升性能。 +傅里叶变换,可以对参数进行调整,以实现FFT/IFFT/RFFT/IRFFT。 + +数学定义如下: + +.. math:: + + X(k) = \sum_{n=0}^{N-1} x(n)\,e^{-j 2\pi kn / N} \quad (\text{Forward FFT}) + +.. math:: + + x(n) = \sum_{k=0}^{N-1} X(k)\,e^{j 2\pi kn / N} \quad (\text{Inverse FFT}) + +其中 :math:`N` 为 FFT 点数。 + +输入: + - **input** - 输入复数数据地址。 + - **fft_size** - FFT 点数。 + - **dir** - 变换方向: + - ``FFT_FORWARD``:正向 FFT + - ``FFT_INVERSE``:反向 FFT + - **scratch_ptr** - 临时缓冲区地址,用于存放旋转因子及中间计算结果。 + - **twiddle** - 旋转因子地址(仅共享存储版本使用)。 + - **fft_size1** - 第一阶段 FFT 点数(仅共享存储版本使用)。 + - **fft_size2** - 第二阶段 FFT 点数(仅共享存储版本使用)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 输出复数序列地址。 + +内部核心计算公式如下: +对于FFT,它计算以下表达式: + +.. math:: + X[\omega_1, \dots, \omega_d] = + \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{-j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}}, + +其中, :math:`d` = `signal_ndim` 是信号的维度,:math:`N_i` 则是信号第 :math:`i` 个维度的大小。 + +对于IFFT,它计算以下表达式: + +.. math:: + X[\omega_1, \dots, \omega_d] = + \frac{1}{\prod_{i=1}^d N_i} \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{\ j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}}, + +其中, :math:`d` = `signal_ndim` 是信号的维度,:math:`N_i` 则是信号第 :math:`i` 维的大小。 + +.. note:: + - FFT/IFFT要求complex64或complex128类型的输入,返回complex64或complex128类型的输出。 + - RFFT要求bool, uint8, int8, int16, int32, int64, float32或float64类型的输入, + 返回complex64或complex128类型的输出。 + - IRFFT要求complex64或complex128类型的输入,返回float32或float64类型的输出。 + - 共享存储版本/私有存储版本支持函数及调用见RFFT。 + +参数: + - **signal_ndim** (int) - 表示每个信号中的维数,控制着傅里叶变换的维数,其值只能为1、2或3。 + - **inverse** (bool) - 表示该操作是否为逆变换,用以选择FFT 和 RFFT 或 IFFT 和 IRFFT。 + + - 如果为 ``True`` ,则为IFFT 和 IRFFT。 + - 如果为 ``False`` ,FFT 和 RFFT。 + + - **real** (bool) - 表示该操作是否为实变换,与 `inverse` 共同决定具体的变换模式: + + - `inverse` 为 ``False`` , `real` 为 ``False`` :对应FFT模式。 + - `inverse` 为 ``True`` , `real` 为 ``False`` :对应IFFT模式。 + - `inverse` 为 ``False`` , `real` 为 ``True`` :对应RFFT模式。 + - `inverse` 为 ``True`` , `real` 为 ``True`` :对应IRFFT模式。 + + - **norm** (str,可选) - 表示该操作的规范化方式,可选值:[ ``"backward"`` , ``"forward"`` , ``"ortho"`` ]。默认值: ``"backward"`` 。 + + - "backward",正向变换不缩放,逆变换按 :math:`1/n` 缩放,其中 `n` 表示输入 `x` 的元素数量。。 + - "ortho",正向变换与逆变换均按 :math:`1/\sqrt n` 缩放。 + - "forward",正向变换按 :math:`1/n` 缩放,逆变换不缩放。 + + - **onesided** (bool,可选) - 控制输入是否减半以避免冗余。默认值: ``True`` 。 + - **signal_sizes** (tuple,可选) - 原始信号的大小(RFFT变换之前的信号,不包含batch这一维),只有在IRFFT模式下和设置 `onesided` 为True时需要该参数,需要满足 + 以下条件。默认值: ``()`` 。 + + - `signal_sizes` 的长度等于IRFFT的 `signal_ndim` : :math:`len(signal\_sizes)=signal\_ndim` 。 + - `signal_sizes` 的最后一个维度除以2等于IRFFT输入的最后一个维度: :math:`signal\_size[-1]/2+1=x.shape[-1]` 。 + - 除了最后一个维度外, `signal_sizes` 的维度与输入shape完全相同: :math:`signal\_sizes[:-1]=x.shape[:-1]` 。 + +异常: + - **TypeError** - 如果FFT/IFFT/IRFF的输入类型不是以下类型之一:complex64、complex128。 + - **TypeError** - 如果输入的类型不是Tensor。 + - **ValueError** - 如果输入 `x` 的维度小于 `signal_ndim` 。 + - **ValueError** - 如果 `signal_ndim` 大于3或小于1。 + - **ValueError** - 如果 `norm` 取值不是"backward"、"forward"或"ortho"。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +类型支持: + - FT78NE:``cplx64``、``cplx128`` + - MT7004:``cplx64`` + diff --git a/master/html/_sources/functionlib/dsplib/fftreal.rst.txt b/master/html/_sources/functionlib/dsplib/fftreal.rst.txt new file mode 100644 index 0000000..42adf86 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fftreal.rst.txt @@ -0,0 +1,104 @@ +FFTReal +========================= + +对输入复数序列实数部分执行一维快速傅里叶变换(FFT)或逆变换(IFFT)。 +内部基于分治 FFT 算法, +并使用预计算旋转因子(twiddle factor)以提升性能。 +傅里叶变换,可以对参数进行调整,以实现FFT/IFFT/RFFT/IRFFT。 + +数学定义如下: + +.. math:: + + X(k) = \sum_{n=0}^{N-1} x(n)\,e^{-j 2\pi kn / N} \quad (\text{Forward FFT}) + +.. math:: + + x(n) = \sum_{k=0}^{N-1} X(k)\,e^{j 2\pi kn / N} \quad (\text{Inverse FFT}) + +其中 :math:`N` 为 FFT 点数。 + +输入: + - **input** - 输入复数数据地址。 + - **fft_size** - FFT 点数。 + - **dir** - 变换方向: + - ``FFT_FORWARD``:正向 FFT + - ``FFT_INVERSE``:反向 FFT + - **scratch_ptr** - 临时缓冲区地址,用于存放旋转因子及中间计算结果。 + - **twiddle** - 旋转因子地址(仅共享存储版本使用)。 + - **fft_size1** - 第一阶段 FFT 点数(仅共享存储版本使用)。 + - **fft_size2** - 第二阶段 FFT 点数(仅共享存储版本使用)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 输出复数序列地址。 + +内部核心计算公式如下: +对于FFT,它计算以下表达式: + +.. math:: + X[\omega_1, \dots, \omega_d] = + \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{-j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}}, + +其中, :math:`d` = `signal_ndim` 是信号的维度,:math:`N_i` 则是信号第 :math:`i` 个维度的大小。 + +对于IFFT,它计算以下表达式: + +.. math:: + X[\omega_1, \dots, \omega_d] = + \frac{1}{\prod_{i=1}^d N_i} \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{\ j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}}, + +其中, :math:`d` = `signal_ndim` 是信号的维度,:math:`N_i` 则是信号第 :math:`i` 维的大小。 + +.. note:: + - FFT/IFFT要求complex64或complex128类型的输入,返回complex64或complex128类型的输出。 + - RFFT要求bool, uint8, int8, int16, int32, int64, float32或float64类型的输入, + 返回complex64或complex128类型的输出。 + - IRFFT要求complex64或complex128类型的输入,返回float32或float64类型的输出。 + - 共享存储版本/私有存储版本支持函数及调用见RFFT。 + +参数: + - **signal_ndim** (int) - 表示每个信号中的维数,控制着傅里叶变换的维数,其值只能为1、2或3。 + - **inverse** (bool) - 表示该操作是否为逆变换,用以选择FFT 和 RFFT 或 IFFT 和 IRFFT。 + + - 如果为 ``True`` ,则为IFFT 和 IRFFT。 + - 如果为 ``False`` ,FFT 和 RFFT。 + + - **real** (bool) - 表示该操作是否为实变换,与 `inverse` 共同决定具体的变换模式: + + - `inverse` 为 ``False`` , `real` 为 ``False`` :对应FFT模式。 + - `inverse` 为 ``True`` , `real` 为 ``False`` :对应IFFT模式。 + - `inverse` 为 ``False`` , `real` 为 ``True`` :对应RFFT模式。 + - `inverse` 为 ``True`` , `real` 为 ``True`` :对应IRFFT模式。 + + - **norm** (str,可选) - 表示该操作的规范化方式,可选值:[ ``"backward"`` , ``"forward"`` , ``"ortho"`` ]。默认值: ``"backward"`` 。 + + - "backward",正向变换不缩放,逆变换按 :math:`1/n` 缩放,其中 `n` 表示输入 `x` 的元素数量。。 + - "ortho",正向变换与逆变换均按 :math:`1/\sqrt n` 缩放。 + - "forward",正向变换按 :math:`1/n` 缩放,逆变换不缩放。 + + - **onesided** (bool,可选) - 控制输入是否减半以避免冗余。默认值: ``True`` 。 + - **signal_sizes** (tuple,可选) - 原始信号的大小(RFFT变换之前的信号,不包含batch这一维),只有在IRFFT模式下和设置 `onesided` 为True时需要该参数,需要满足 + 以下条件。默认值: ``()`` 。 + + - `signal_sizes` 的长度等于IRFFT的 `signal_ndim` : :math:`len(signal\_sizes)=signal\_ndim` 。 + - `signal_sizes` 的最后一个维度除以2等于IRFFT输入的最后一个维度: :math:`signal\_size[-1]/2+1=x.shape[-1]` 。 + - 除了最后一个维度外, `signal_sizes` 的维度与输入shape完全相同: :math:`signal\_sizes[:-1]=x.shape[:-1]` 。 + +异常: + - **TypeError** - 如果FFT/IFFT/IRFF的输入类型不是以下类型之一:complex64、complex128。 + - **TypeError** - 如果输入的类型不是Tensor。 + - **ValueError** - 如果输入 `x` 的维度小于 `signal_ndim` 。 + - **ValueError** - 如果 `signal_ndim` 大于3或小于1。 + - **ValueError** - 如果 `norm` 取值不是"backward"、"forward"或"ortho"。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +类型支持: + - FT78NE:``cplx64``、``cplx128`` + - MT7004:``cplx64`` + diff --git a/master/html/_sources/functionlib/dsplib/fill.rst.txt b/master/html/_sources/functionlib/dsplib/fill.rst.txt new file mode 100644 index 0000000..231322c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fill.rst.txt @@ -0,0 +1,103 @@ +Fill +================= + + + + 使用给定的常量值填充输出数组。该算子将输入的 **value** 按指定数据类型复制,并在多核环境下利用 DMA 高效地完成填充操作。 + + .. math:: + + dst_i = value + \quad \text{for} \quad i = 0,1,\dots,N-1 + + 输入: + - **value** - 待填充的常量值地址。 + - **param** - Fill 参数结构体指针,描述输出张量的形状与数据类型信息。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 填充结果地址。 + + FillParameter 说明: + FillParameter 用于描述 Fill 算子的输出张量信息,其定义如下: + + .. code-block:: c + + typedef struct FillParameter { + int* shape_; // 张量各维度大小 + int ndim_; // 张量维度数 + int elem_cnt_; // 张量元素总数 + int type_size_; // 单个元素字节大小 + } FillParameter; + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持fp, dp, int8, int16, int32, cplx64, cplx128 + - MT7004 支持hp, fp, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_fill_s(int8_t* value, int8_t* output, int core_mask, FillParameter* param) +.. c:function:: void i16_fill_s(int16_t* value, int16_t* output, int core_mask, FillParameter* param) +.. c:function:: void i32_fill_s(int32_t* value, int32_t* output, int core_mask, FillParameter* param) +.. c:function:: void hp_fill_s(half* value, half* output, int core_mask, FillParameter* param) +.. c:function:: void fp_fill_s(float* value, float* output, int core_mask, FillParameter* param) +.. c:function:: void dp_fill_s(double* value, double* output, int core_mask, FillParameter* param) +.. c:function:: void c64_fill_s(float* value, float* output, int core_mask, FillParameter* param) +.. c:function:: void c128_fill_s(double* value, double* output, int core_mask, FillParameter* param) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float value = 1.0f; + float *output = (float *)0xC0000000; + FillParameter param; + param.elem_cnt_ = 1024; + param.type_size_ = sizeof(float); + int core_mask = 0xff; + fp_fill_s(&value, output, core_mask, ¶m); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_fill_p(int8_t* value, int8_t* output, FillParameter* param) +.. c:function:: void i16_fill_p(int16_t* value, int16_t* output, FillParameter* param) +.. c:function:: void i32_fill_p(int32_t* value, int32_t* output, FillParameter* param) +.. c:function:: void hp_fill_p(half* value, half* output, FillParameter* param) +.. c:function:: void fp_fill_p(float* value, float* output, FillParameter* param) +.. c:function:: void dp_fill_p(double* value, double* output, FillParameter* param) +.. c:function:: void c64_fill_p(float* value, float* output, FillParameter* param) +.. c:function:: void c128_fill_p(double* value, double* output, FillParameter* param) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float value = 1.0f; + float *output = (float *)0x10820000; + FillParameter param; + param.elem_cnt_ = 1024; + param.type_size_ = sizeof(float); + fp_fill_p(&value, output, ¶m); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/flatten.rst.txt b/master/html/_sources/functionlib/dsplib/flatten.rst.txt new file mode 100644 index 0000000..a62f1c5 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/flatten.rst.txt @@ -0,0 +1,104 @@ +Flatten +================= + +将多维输入张量按行优先(row-major)顺序展平成一维连续数组。 +该算子仅改变数据的形状,不改变数据内容与顺序,本质上是一次连续内存拷贝操作。 + +数学上可表示为: + +.. math:: + + \text{output} = \text{reshape}(\text{input}, [-1]) + +输入张量的所有元素按照原有存储顺序依次写入输出张量。 + +输入: + - **input** - 输入数据地址。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + - **param** - Flatten 参数结构体指针,包含形状与类型信息。 + +输出: + - **output** - 展平后的输出数据地址。 + +FlattenParameter 结构体说明: + + - **shape_** - 输入张量各维度大小数组。 + - **ndim_** - 输入张量维度数。 + - **elem_cnt_** - 输入张量元素总数。 + - **type_size_** - 单个元素的字节数。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32``、``fp64``、``int8``、``int16``、``int32``、``cplx64``、``cplx128`` + - MT7004 支持 ``fp16``、``fp32``、``int16``、``int32``、``cplx64`` + - 该算子不涉及数值计算,仅进行内存搬运 + - 当 input 与 output 地址相同时,算子直接返回,不执行拷贝 + - 当 core_mask 对应的逻辑核无效时,算子直接返回 + +**共享存储版本:** + +.. c:function:: void Flatten(const void* input, void* output, int core_mask, FlattenParameter* param) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 18 + + // FT78NE 示例(共享存储) + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // DDR 空间 + float *output = (float *)0xC0000000; + + int shape[3] = {1, 3, 224 * 224}; + + FlattenParameter param; + param.shape_ = shape; + param.ndim_ = 3; + param.elem_cnt_ = 1 * 3 * 224 * 224; + param.type_size_ = sizeof(float); + + int core_mask = 0xff; + Flatten(input, output, core_mask, ¶m); + + return 0; + } + + +**私有存储版本:** + +该算子在私有存储模式下等价为单核内存拷贝操作, +直接调用 `Flatten` 且 core_mask 设置为单核即可。 + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + // FT78NE 示例(私有存储) + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; // L2 空间 + float *output = (float *)0x10820000; + + int shape[2] = {4, 256}; + + FlattenParameter param; + param.shape_ = shape; + param.ndim_ = 2; + param.elem_cnt_ = 4 * 256; + param.type_size_ = sizeof(float); + + Flatten(input, output, 0x1, ¶m); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/flattengrad.rst.txt b/master/html/_sources/functionlib/dsplib/flattengrad.rst.txt new file mode 100644 index 0000000..43347fe --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/flattengrad.rst.txt @@ -0,0 +1,82 @@ +FlattenGrad +================= + +Flatten 算子的反向传播算子,用于将一维梯度按照原始输入张量的形状还原。 +该算子不进行数值计算,仅依据原始形状信息对梯度数据进行内存拷贝与重解释,元素顺序保持不变。 + +.. math:: + + \text{grad\_input} = \text{reshape}(\text{grad\_output}, \text{origin\_shape}) + +输入: + - **input** - 上游梯度(Flatten 输出对应的梯度)地址。 + - **origin_shape** - 原始输入张量的形状数组。 + - **origin_ndim** - 原始输入张量的维度数。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 还原形状后的梯度数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp16, fp32 + - 该算子仅进行内存拷贝,不涉及数值运算 + - 输出元素总数等于 origin_shape 各维度之积 + +**共享存储版本:** + +.. c:function:: void fp_flattengrad_s(float* input, float* output, int* origin_shape, int origin_ndim, int core_mask) +.. c:function:: void hp_flattengrad_s(half* input, half* output, int* origin_shape, int origin_ndim, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *grad_flat = (float *)0xA0000000; // 梯度在 DDR + float *grad_out = (float *)0xC0000000; + + int origin_shape[2] = {32, 128}; + int origin_ndim = 2; + int core_mask = 0xff; + + fp_flattengrad_s(grad_flat, grad_out, origin_shape, origin_ndim, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_flattengrad_p(float* input, float* output, int* origin_shape, int origin_ndim) +.. c:function:: void hp_flattengrad_p(half* input, half* output, int* origin_shape, int origin_ndim) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // MT7004 示例 + #include + #include + + int main(int argc, char* argv[]) { + half *grad_flat = (half *)0x10810000; // 梯度在 L2 + half *grad_out = (half *)0x10820000; + + int origin_shape[3] = {1, 3, 224 * 224}; + int origin_ndim = 3; + + hp_flattengrad_p(grad_flat, grad_out, origin_shape, origin_ndim); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/floormod.rst.txt b/master/html/_sources/functionlib/dsplib/floormod.rst.txt new file mode 100644 index 0000000..8928497 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/floormod.rst.txt @@ -0,0 +1,84 @@ +Floormod +================= + + + +逐元素计算两个输入张量的 floor-modulus。 + +.. math:: + + \text{output}_i = \text{input0}_i - \lfloor \frac{\text{input0}_i}{\text{input1}_i} \rfloor \cdot \text{input1}_i + +其中 :math:`\lfloor \cdot \rfloor` 表示向下取整 (floor) 操作。 + +输入: + - **input0** - 第一个输入张量(被除数)的数据地址。 + - **input1** - 第二个输入张量(除数)的数据地址。 + - **size** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与输入张量相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_floormod_s(float* input0, float* input1, float* output, int size, int core_mask) +.. c:function:: void hp_floormod_s(half* input0, half* input1, half* output, int size, int core_mask) +.. c:function:: void dp_floormod_s(double* input0, double* input1, double* output, int size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // input0 在DDR空间 + float *input1 = (float *)0xB0000000; // input1 + float *output = (float *)0xC0000000; // output + + int size = 4096; + int core_mask = 0xff; + + fp_floormod_s(input0, input1, output, size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_floormod_p(float* input0, float* input1, float* output, int size) +.. c:function:: void hp_floormod_p(half* input0, half* input1, half* output, int size) +.. c:function:: void dp_floormod_p(double* input0, double* input1, double* output, int size) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; // input0 在L2空间 + float *input1 = (float *)0x11000000; // input1 + float *output = (float *)0x12000000; // output + + int size = 1024; + + fp_floormod_p(input0, input1, output, size); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/formattranspose.rst.txt b/master/html/_sources/functionlib/dsplib/formattranspose.rst.txt new file mode 100644 index 0000000..6733c3c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/formattranspose.rst.txt @@ -0,0 +1,90 @@ +FormatTranspose +================= + + + +将输入数据从一种存储格式转换为另一种存储格式,如 NCHW ↔ NHWC、NC4HW4、NC8HW8 等,适用于图像或特征图数据。 + +输入: + - **src_data** - 输入数据地址。 + - **src_format** - 输入数据格式标识。 + - **dst_format** - 输出数据格式标识。 + - **batch** - 批大小。 + - **channel** - 通道数。 + - **plane** - 高*宽。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst_data** - 格式转换结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, dp, int8, int16, int32, clx64, cplx128 + - MT7004 支持hp, fp, i16, i32, cplx64 + +**共享存储版本:** + +.. c:function:: void fp_formattranspose_s(int src_format, int dst_format, float* src_data, float* dst_data, int batch, int channel, int plane, int core_mask) +.. c:function:: void hp_formattranspose_s(int src_format, int dst_format, half* src_data, half* dst_data, int batch, int channel, int plane, int core_mask) +.. c:function:: void dp_formattranspose_s(int src_format, int dst_format, double* src_data, double* dst_data, int batch, int channel, int plane, int core_mask) +.. c:function:: void i8_formattranspose_s(int src_format, int dst_format, int8_t* src_data, int8_t* dst_data, int batch, int channel, int plane, int core_mask) +.. c:function:: void i16_formattranspose_s(int src_format, int dst_format, int16_t* src_data, int16_t* dst_data, int batch, int channel, int plane, int core_mask) +.. c:function:: void i32_formattranspose_s(int src_format, int dst_format, int* src_data, int* dst_data, int batch, int channel, int plane, int core_mask) +.. c:function:: void c64_formattranspose_s(int src_format, int dst_format, float* src_data, float* dst_data, int batch, int channel, int plane, int core_mask) +.. c:function:: void c128_formattranspose_s(int src_format, int dst_format, double* src_data, double* dst_data, int batch, int channel, int plane, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + #include + + int main() { + float *input = (float *)0xA0000000; // 输入在DDR空间 + float *output = (float *)0xC0000000; + int batch = 1, channel = 3, plane = 224*224; + int src_format = 1; // NHWC + int dst_format = 0; // NCHW + int core_mask = 0xff; + + fp_formattranspose_s(src_format, dst_format, input, output, batch, channel, plane, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_formattranspose_p(int src_format, int dst_format, float* src_data, float* dst_data, int batch, int channel, int plane) +.. c:function:: void hp_formattranspose_p(int src_format, int dst_format, half* src_data, half* dst_data, int batch, int channel, int plane) +.. c:function:: void dp_formattranspose_p(int src_format, int dst_format, double* src_data, double* dst_data, int batch, int channel, int plane) +.. c:function:: void i8_formattranspose_p(int src_format, int dst_format, int8_t* src_data, int8_t* dst_data, int batch, int channel, int plane) +.. c:function:: void i16_formattranspose_p(int src_format, int dst_format, int16_t* src_data, int16_t* dst_data, int batch, int channel, int plane) +.. c:function:: void i32_formattranspose_p(int src_format, int dst_format, int* src_data, int* dst_data, int batch, int channel, int plane) +.. c:function:: void c64_formattranspose_p(int src_format, int dst_format, float* src_data, float* dst_data, int batch, int channel, int plane) +.. c:function:: void c128_formattranspose_p(int src_format, int dst_format, double* src_data, double* dst_data, int batch, int channel, int plane) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main() { + float *input = (float *)0x10810000; // 输入在L2空间 + float *output = (float *)0x10820000; + int batch = 1, channel = 3, plane = 224*224; + int src_format = 1; // NHWC + int dst_format = 0; // NCHW + + fp_formattranspose_p(src_format, dst_format, input, output, batch, channel, plane); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/fullconnection.rst.txt b/master/html/_sources/functionlib/dsplib/fullconnection.rst.txt new file mode 100644 index 0000000..048e5bc --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/fullconnection.rst.txt @@ -0,0 +1,93 @@ +FullConnection +================= + + + +传入输入矩阵与权重矩阵,执行全连接计算(矩阵乘法),并可选择性地叠加偏置项与激活函数, +最终输出结果矩阵。 + +.. math:: + + dst_{i,j} = \sum_{k=0}^{K-1} A_{i,k} \cdot B_{k,j} + bias_{i,j} + + dst_{i,j} = activation(dst_{i,j}) + +其中激活函数支持 ReLU 与 ReLU6。 + +输入: + - **A** - 输入矩阵地址,形状为 ``M × K``。 + - **B** - 权重矩阵地址,形状为 ``K × N``。 + - **bias** - 偏置矩阵地址,形状为 ``M × N``,可为 ``NULL``。 + - **M** - 输出矩阵行数。 + - **N** - 输出矩阵列数。 + - **K** - 中间维度大小。 + - **activation_type** - 激活函数类型。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **C** - 输出矩阵地址,形状为 ``M × N``。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, int8 + - MT7004 支持hp, fp + - activation_type 支持 ``ACTIVATION_NONE``、``ACTIVATION_RELU``、``ACTIVATION_RELU6`` + +**共享存储版本:** + +.. c:function:: void i8_fullconnection_s(int8_t* A, int8_t* B, int8_t* C, int8_t* bias, int M, int N, int K, int activation_type, int core_mask) +.. c:function:: void fp_fullconnection_s(float* A, float* B, float* C, float* bias, int M, int N, int K, int activation_type, int core_mask) +.. c:function:: void hp_fullconnection_s(half* A, half* B, half* C, half* bias, int M, int N, int K, int activation_type, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *A = (float *)0xA0000000; + float *B = (float *)0xA0010000; + float *C = (float *)0xC0000000; + float *bias = NULL; + int M = 4, N = 8, K = 16; + int activation_type = ACTIVATION_RELU; + int core_mask = 0xff; + fp_fullconnection_s(A, B, C, bias, M, N, K, activation_type, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_fullconnection_p(int8_t* A, int8_t* B, int8_t* C, int8_t* bias, int M, int N, int K, int activation_type) +.. c:function:: void fp_fullconnection_p(float* A, float* B, float* C, float* bias, int M, int N, int K, int activation_type) +.. c:function:: void hp_fullconnection_p(half* A, half* B, half* C, half* bias, int M, int N, int K, int activation_type) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *A = (float *)0x10810000; + float *B = (float *)0x10820000; + float *C = (float *)0x10830000; + float *bias = NULL; + int M = 4, N = 8, K = 16; + int activation_type = ACTIVATION_RELU6; + fp_fullconnection_p(A, B, C, bias, M, N, K, activation_type); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/fusedbatchnorm.rst.txt b/master/html/_sources/functionlib/dsplib/fusedbatchnorm.rst.txt index 6095913..e7869d5 100644 --- a/master/html/_sources/functionlib/dsplib/fusedbatchnorm.rst.txt +++ b/master/html/_sources/functionlib/dsplib/fusedbatchnorm.rst.txt @@ -38,7 +38,7 @@ FusedBatchNorm .. code-block:: c :linenos: - :emphasize-lines: 15 + :emphasize-lines: 15-17 // FT78NE 多核示例 #include @@ -70,7 +70,7 @@ FusedBatchNorm .. code-block:: c :linenos: - :emphasize-lines: 14 + :emphasize-lines: 14-15 // MT7004 单核示例 #include diff --git a/master/html/_sources/functionlib/dsplib/gather.rst.txt b/master/html/_sources/functionlib/dsplib/gather.rst.txt new file mode 100644 index 0000000..ada27cc --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/gather.rst.txt @@ -0,0 +1,107 @@ +Gather +================= + +沿着给定的轴 ``axis``,根据 ``indices`` 张量提供的索引值,从 ``input`` 张量中收集数据。支持 ``batch_dims`` 指定的批处理维度,即在前 ``batch_dims`` 个维度上,索引和输入是对应的。 + +.. math:: + + \text{output}[i_0, ..., i_{axis-1}, j_0, ..., j_{indices\_ndim-batch\_dims-1}, i_{axis+1}, ..., i_{input\_ndim-1}] = \\ + \text{input}[i_0, ..., i_{axis-1}, \text{indices}[i_0, ..., i_{batch\_dims-1}, j_0, ..., j_{indices\_ndim-batch\_dims-1}], i_{axis+1}, ..., i_{input\_ndim-1}] + +输入: + - **output** - 计算结果输出地址。 + - **input** - 输入源张量数据地址。 + - **input_shape** - 输入张量的形状数组地址。 + - **input_ndim** - 输入张量的维度数量。 + - **indices** - 索引张量数据地址(通常为 int32 类型)。 + - **indices_shape** - 索引张量的形状数组地址。 + - **indices_ndim** - 索引张量的维度数量。 + - **axis** - 沿着哪个轴进行聚集操作。 + - **batch_dims** - 批处理维度数量。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 聚集后的计算结果。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 索引张量 ``indices`` 内部存储的索引值必须在 ``[0, input_shape[axis])`` 范围内,否则行为未定义。 + - 聚集操作涉及非连续访存,在大规模数据下建议使用共享存储版本并行处理。 + +**共享存储版本:** + +.. c:function:: void i8_gather_s(int8_t* output, int8_t* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) +.. c:function:: void i16_gather_s(int16_t* output, int16_t* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) +.. c:function:: void i32_gather_s(int32_t* output, int32_t* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) +.. c:function:: void hp_gather_s(half* output, half* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) +.. c:function:: void fp_gather_s(float* output, float* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) +.. c:function:: void dp_gather_s(double* output, double* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) +.. c:function:: void c64_gather_s(float* output, float* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) +.. c:function:: void c128_gather_s(double* output, double* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 17 + + // FT78NE 示例:多核并行聚集操作 + #include + #include "78NE/utils.h" + + int main() { + float *input = (float *)0xA0000000; + int *indices = (int *)0xB0000000; + float *output = (float *)0xC0000000; + int input_shape[] = {16, 800, 80}; + int indices_shape[] = {16, 400}; + int input_ndim = 3; + int indices_ndim = 2; + int axis = 1; + int batch_dims = 1; + int core_mask = 0xFF; // 使用8核并行 + + fp_gather_s(output, input, input_shape, input_ndim, indices, indices_shape, indices_ndim, axis, batch_dims, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_gather_p(int8_t* output, int8_t* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) +.. c:function:: void i16_gather_p(int16_t* output, int16_t* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) +.. c:function:: void i32_gather_p(int32_t* output, int32_t* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) +.. c:function:: void hp_gather_p(half* output, half* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) +.. c:function:: void fp_gather_p(float* output, float* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) +.. c:function:: void dp_gather_p(double* output, double* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) +.. c:function:: void c64_gather_p(float* output, float* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) +.. c:function:: void c128_gather_p(double* output, double* input, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int axis, int batch_dims) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // MT7004 示例:单核聚集操作 + #include + + int main() { + float *input = (float *)0x10000000; + int *indices = (int *)0x10010000; + float *output = (float *)0x10020000; + int input_shape[] = {2, 100, 10}; + int indices_shape[] = {2, 50}; + int input_ndim = 3; + int indices_ndim = 2; + int axis = 1; + int batch_dims = 1; + + fp_gather_p(output, input, input_shape, input_ndim, indices, indices_shape, indices_ndim, axis, batch_dims); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/gather_nd.rst.txt b/master/html/_sources/functionlib/dsplib/gather_nd.rst.txt new file mode 100644 index 0000000..d996f73 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/gather_nd.rst.txt @@ -0,0 +1,97 @@ +GatherNd +================= + +根据索引张量中的坐标,从输入张量中收集切片或元素。 + +GatherNd 允许根据 ``indices`` 提供的多维索引,在 ``input`` 张量的指定轴上进行切片提取。如果 ``indices`` 的最后一个维度长度为 ``K``,则它代表从 ``input`` 的前 ``K`` 个维度中提取对应的子张量(切片)。 + +输入: + - **input** - 输入数据张量地址。 + - **output** - 计算结果张量地址。 + - **input_shape** - 输入张量的形状数组地址。 + - **input_ndim** - 输入张量的维度数。 + - **indices** - 索引张量地址(类型固定为 int32)。 + - **indices_shape** - 索引张量的形状数组地址。 + - **indices_ndim** - 索引张量的维度数。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 收集后的数据存放地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 索引张量 (indices) 在所有平台上均使用 int32 类型。 + - 坐标索引必须在输入张量维度的合法范围内,否则行为未定义。 + +**共享存储版本:** + +.. c:function:: void i8_gather_nd_s(int8_t* input, int8_t* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) +.. c:function:: void i16_gather_nd_s(int16_t* input, int16_t* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) +.. c:function:: void i32_gather_nd_s(int* input, int* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) +.. c:function:: void hp_gather_nd_s(half* input, half* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) +.. c:function:: void fp_gather_nd_s(float* input, float* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) +.. c:function:: void dp_gather_nd_s(double* input, double* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) +.. c:function:: void c64_gather_nd_s(float* input, float* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) +.. c:function:: void c128_gather_nd_s(double* input, double* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 示例(共享存储) + #include + #include "78NE/utils.h" + + int main() { + float *input = (float *)0xA0000000; // 输入在 DDR 空间 + float *output = (float *)0xB0000000; // 输出在 DDR 空间 + int *indices = (int *)0xC0000000; // 索引在 DDR 空间 + int input_shape[] = {10, 10, 5}; + int indices_shape[] = {3, 2}; // 提取3个坐标,每个坐标深度为2 + int input_ndim = 3; + int indices_ndim = 2; + int core_mask = 0xFF; + + fp_gather_nd_s(input, output, input_shape, input_ndim, indices, indices_shape, indices_ndim, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_gather_nd_p(int8_t* input, int8_t* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) +.. c:function:: void i16_gather_nd_p(int16_t* input, int16_t* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) +.. c:function:: void i32_gather_nd_p(int* input, int* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) +.. c:function:: void hp_gather_nd_p(half* input, half* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) +.. c:function:: void fp_gather_nd_p(float* input, float* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) +.. c:function:: void dp_gather_nd_p(double* input, double* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) +.. c:function:: void c64_gather_nd_p(float* input, float* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) +.. c:function:: void c128_gather_nd_p(double* input, double* output, int* input_shape, int input_ndim, int* indices, int* indices_shape, int indices_ndim) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + // MT7004 示例(私有存储) + #include + + int main() { + float *input = (float *)0x10000000; + float *output = (float *)0x10010000; + int *indices = (int *)0x10020000; + int input_shape[] = {4, 5, 6}; + int indices_shape[] = {2, 2, 2}; + int input_ndim = 3; + int indices_ndim = 3; + + fp_gather_nd_p(input, output, input_shape, input_ndim, indices, indices_shape, indices_ndim); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/gatherd.rst.txt b/master/html/_sources/functionlib/dsplib/gatherd.rst.txt new file mode 100644 index 0000000..14cd24a --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/gatherd.rst.txt @@ -0,0 +1,117 @@ +GatherD +========= + +按照给定的维度 ``dim`` 和索引张量 ``index``,从输入张量中按元素位置 +抽取数据,生成新的输出张量。 + +该算子等价于在指定维度上执行逐元素 Gather 操作,其输出形状与 +``index`` 张量形状一致。 + +.. math:: + + \text{output}[i_0, \dots, i_n] + = + \text{input}_x[i_0, \dots, i_{dim-1}, \text{index}[i_0, \dots, i_n], i_{dim+1}, \dots, i_n] + +输入: + - **input_x** - 输入张量的数据地址。 + 数据类型需与所调用的 GatherD 接口类型一致。 + + - **dim** - 指定进行 Gather 操作的维度索引,取值范围为 + ``[0, input_shape_size)``。 + + - **index** - 索引张量的数据地址,类型为 ``int*``, + 用于指定在 ``dim`` 维度上的取值位置。 + + - **input_shape** - 输入张量各维度大小数组地址。 + + - **input_shape_size** - 输入张量的维度数量。 + + - **index_shape** - 索引张量的形状数组地址, + 其维度数量与 ``input_shape_size`` 相同。 + + - **core_mask** - 核掩码(仅共享存储版本使用)。 + +输出: + - **output** - 输出张量的数据地址, + 其形状与 ``index_shape`` 保持一致, + 数据类型与 ``input_x`` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - ``index`` 中的取值应满足 + ``0 <= index[...] < input_shape[dim]``。 + - 输出张量的元素个数等于 ``index_shape`` 各维度之积。 + - 该算子不对索引顺序进行任何排序或检查。 + +**共享存储版本:** + +.. c:function:: void fp_gatherd_s(float* input_x, int dim, int* index, float* output, int* input_shape, int input_shape_size, int* index_shape, int core_mask) +.. c:function:: void dp_gatherd_s(double* input_x, int dim, int* index, double* output, int* input_shape, int input_shape_size, int* index_shape, int core_mask) +.. c:function:: void i8_gatherd_s(int8_t* input_x, int dim, int* index, int8_t* output, int* input_shape, int input_shape_size, int* index_shape, int core_mask) +.. c:function:: void i16_gatherd_s(int16_t* input_x, int dim, int* index, int16_t* output, int* input_shape, int input_shape_size, int* index_shape, int core_mask) +.. c:function:: void i32_gatherd_s(int32_t* input_x, int dim, int* index, int32_t* output, int* input_shape, int input_shape_size, int* index_shape, int core_mask) +.. c:function:: void c64_gatherd_s(float* input_x, int dim, int* index, float* output, int* input_shape, int input_shape_size, int* index_shape, int core_mask) +.. c:function:: void c128_gatherd_s(double* input_x, int dim, int* index, double* output, int* input_shape, int input_shape_size, int* index_shape, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input_x = (float *)0xA0000000; // input_x 在 DDR 空间 + float *output = (float *)0xB0000000; + int *index = (int *)0xA1000000; + + int input_shape[] = {4, 8, 16}; + int index_shape[] = {4, 8, 16}; + int input_shape_size = 3; + int dim = 1; + int core_mask = 0xff; + + fp_gatherd_s(input_x, dim, index, output, input_shape, input_shape_size, index_shape, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_gatherd_p(float* input_x, int dim, int* index, float* output, int* input_shape, int input_shape_size, int* index_shape) +.. c:function:: void dp_gatherd_p(double* input_x, int dim, int* index, double* output, int* input_shape, int input_shape_size, int* index_shape) +.. c:function:: void i8_gatherd_p(int8_t* input_x, int dim, int* index, int8_t* output, int* input_shape, int input_shape_size, int* index_shape) +.. c:function:: void i16_gatherd_p(int16_t* input_x, int dim, int* index, int16_t* output, int* input_shape, int input_shape_size, int* index_shape) +.. c:function:: void i32_gatherd_p(int32_t* input_x, int dim, int* index, int32_t* output, int* input_shape, int input_shape_size, int* index_shape) +.. c:function:: void c64_gatherd_p(float* input_x, int dim, int* index, float* output, int* input_shape, int input_shape_size, int* index_shape) +.. c:function:: void c128_gatherd_p(double* input_x, int dim, int* index, double* output, int* input_shape, int input_shape_size, int* index_shape) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + // MT7004 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input_x = (float *)0x10000000; // input_x 在 L2 空间 + float *output = (float *)0x10010000; + int *index = (int *)0x10020000; + + int input_shape[] = {2, 4}; + int index_shape[] = {2, 4}; + int input_shape_size = 2; + int dim = 0; + + fp_gatherd_p(input_x, dim, index, output, input_shape, input_shape_size, index_shape); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/glu.rst.txt b/master/html/_sources/functionlib/dsplib/glu.rst.txt new file mode 100644 index 0000000..e99d872 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/glu.rst.txt @@ -0,0 +1,106 @@ +GLU +================= + +Gated Linear Unit(门控线性单元)算子。该算子将输入张量在指定维度(``split_dim``)上分割成两个相等的部分(A和B),然后对第二部分(B)应用Sigmoid激活函数,并将其与第一部分(A)逐元素相乘。 + +.. math:: + + \text{out} = A \otimes \sigma(B) + +其中,A和B是输入张量沿 ``split_dim`` 维度分割后的两个子张量,:math:`\sigma` 是Sigmoid函数,:math:`\otimes` 表示逐元素乘法。 + +输入: + - **in_data** - 输入张量的数据地址。 + - **split_data** - 一个指针数组,用于存放分割后的子张量地址(作为临时缓冲区)。 + - **out_data** - 输出张量的数据地址。 + - **ndim** - 输入张量的维度数量。 + - **split_dim** - 执行分割操作的目标维度轴。 + - **input_shape** - 指向一个整数数组的指针,该数组描述了输入张量的形状。 + - **num_split** - 分割的数量,对于GLU操作,此值应为2。 + - **split_sizes** - 指向一个整数数组的指针,该数组描述了每个子张量在 ``split_dim`` 维度上的大小。 + - **strides** - 指向一个整数数组的指针,用于存放为张量索引预先计算好的步长。 + - **len** - 输出张量中的元素总数。 + - **core_mask** - 核掩码 (仅共享存储版本需要)。 + +输出: + - **out_data** - 存储GLU计算结果的张量。其形状与输入张量相同,但在 ``split_dim`` 维度上的大小减半。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE支持fp32和int8类型。 + - MT7004支持fp16和fp32类型。 + +**共享存储版本:** + +.. c:function:: void i8_glu_s(int8_t* in_data, int8_t** split_data, int8_t* out_data, int ndim, int split_dim, int* input_shape, int num_split, int* split_sizes, int* strides, int len, int core_mask) +.. c:function:: void hp_glu_s(half* in_data, half** split_data, half* out_data, int ndim, int split_dim, int* input_shape, int num_split, int* split_sizes, int* strides, int len, int core_mask) +.. c:function:: void fp_glu_s(float* in_data, float** split_data, float* out_data, int ndim, int split_dim, int* input_shape, int num_split, int* split_sizes, int* strides, int len, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 21 + + // 7004平台, fp32示例 + #include + #include "glu.h" + + int main(int argc, char* argv[]) { + float *in_data = (float *)0xA0000000; // input在DDR空间 + float *out_data = (float *)0xC0000000; // output在DDR空间 + float *split_data_buf[2]; // 临时缓冲区指针 + // ... (需要为split_data_buf分配内存) + + // 假设张量属性已定义 + int ndim = 2; + int split_dim = 1; + int input_shape[] = {2, 4}; + int num_split = 2; + int split_sizes[] = {2, 2}; + int len = 4; + int strides[2]; // 需要预先计算 + int core_mask = 0xff; + + fp_glu_s(in_data, split_data_buf, out_data, ndim, split_dim, input_shape, num_split, split_sizes, strides, len, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_glu_p(int8_t* in_data, int8_t** split_data, int8_t* out_data, int ndim, int split_dim, int* input_shape, int num_split, int* split_sizes, int* strides, int len) +.. c:function:: void hp_glu_p(half* in_data, half** split_data, half* out_data, int ndim, int split_dim, int* input_shape, int num_split, int* split_sizes, int* strides, int len) +.. c:function:: void fp_glu_p(float* in_data, float** split_data, float* out_data, int ndim, int split_dim, int* input_shape, int num_split, int* split_sizes, int* strides, int len) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 20 + + // 7004平台, fp32示例 + #include + #include "glu.h" + + int main(int argc, char* argv[]) { + float *in_data = (float *)0x10000000; // input在L2空间 + float *out_data = (float *)0x10001000; // output在L2空间 + float *split_data_buf[2]; // 临时缓冲区指针 + // ... (需要为split_data_buf分配内存) + + // 假设张量属性已定义 + int ndim = 2; + int split_dim = 1; + int input_shape[] = {2, 4}; + int num_split = 2; + int split_sizes[] = {2, 2}; + int len = 4; + int strides[2]; // 需要预先计算 + + fp_glu_p(in_data, split_data_buf, out_data, ndim, split_dim, input_shape, num_split, split_sizes, strides, len); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/greater.rst.txt b/master/html/_sources/functionlib/dsplib/greater.rst.txt new file mode 100644 index 0000000..5b3180e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/greater.rst.txt @@ -0,0 +1,98 @@ +Greater +================= + + +逐元素比较两个输入张量,判断 `input1` 的元素是否大于 `input2` 的对应元素。支持广播。 + +.. math:: + + \text{output}_i = (\text{input1}_i > \text{input2}_i) + +输出一个布尔张量,如果比较结果为真,则对应元素为 `true`,否则为 `false`。 + +输入: + - **element_num** - 输出张量的总元素数量。在广播情况下,等于较大输入张量的元素数量。 + - **optimize** - 广播优化标志。若为非0,则启用广播模式。 + - **in_elements_num0** - 第一个输入张量 `input1` 的元素数量。用于广播判断。 + - **input1** - 第一个输入张量的数据地址。 + - **input2** - 第二个输入张量的数据地址。 + - **output** - (输出) 输出的布尔类型张量的数据地址。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 写入比较结果的布尔张量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, fp64, int8, int16, int32 + - MT7004 支持fp16, fp32, int16, int32 + + +**共享存储版本:** + +.. c:function:: void fp_greater_s(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output, int core_mask) +.. c:function:: void hp_greater_s(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output, int core_mask) +.. c:function:: void dp_greater_s(int element_num, int optimize, int in_elements_num0, double *input1, double *input2, bool *output, int core_mask) +.. c:function:: void i32_greater_s(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output, int core_mask) +.. c:function:: void i16_greater_s(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output, int core_mask) +.. c:function:: void i8_greater_s(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input1 = (float *)0xA0000000; // Scalar input, DDR + float *input2 = (float *)0xB0000000; // Tensor input + bool *output = (bool *)0xC0000000; // Output + + // Broadcasting input1 (scalar) > input2 (tensor) + int element_num = 4096; // Size of input2 and output + int optimize = 1; // Enable broadcasting + int in_elements_num0 = 1; // Size of input1 + int core_mask = 0xff; + + fp_greater_s(element_num, optimize, in_elements_num0, input1, input2, output, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_greater_p(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output) +.. c:function:: void hp_greater_p(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output) +.. c:function:: void i32_greater_p(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output) +.. c:function:: void i16_greater_p(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output) +.. c:function:: void i8_greater_p(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input1 = (float *)0x10000000; // Tensor input, L2 + float *input2 = (float *)0x11000000; // Scalar input + bool *output = (bool *)0x12000000; // Output + + // Broadcasting input1 (tensor) > input2 (scalar) + int element_num = 1024; // Size of input1 and output + int optimize = 1; // Enable broadcasting + int in_elements_num0 = 1024; // Size of input1 + + fp_greater_p(element_num, optimize, in_elements_num0, input1, input2, output); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/greaterequal.rst.txt b/master/html/_sources/functionlib/dsplib/greaterequal.rst.txt new file mode 100644 index 0000000..2587d57 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/greaterequal.rst.txt @@ -0,0 +1,98 @@ +Greaterequal +================= + + + +逐元素比较两个输入张量,判断 `input1` 的元素是否大于或等于 `input2` 的对应元素。支持广播。 + +.. math:: + + \text{output}_i = (\text{input1}_i \ge \text{input2}_i) + +输出一个布尔张量,如果比较结果为真,则对应元素为 `true`,否则为 `false`。 + +输入: + - **element_num** - 输出张量的总元素数量。在广播情况下,等于较大输入张量的元素数量。 + - **optimize** - 广播优化标志。若为非0,则启用广播模式。 + - **in_elements_num0** - 第一个输入张量 `input1` 的元素数量。用于广播判断。 + - **input1** - 第一个输入张量的数据地址。 + - **input2** - 第二个输入张量的数据地址。 + - **output** - (输出) 输出的布尔类型张量的数据地址。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 写入比较结果的布尔张量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, fp64, int8, int16, int32 + - MT7004 支持fp16, fp32, int16, int32 + +**共享存储版本:** + +.. c:function:: void fp_greaterequal_s(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output, int core_mask) +.. c:function:: void hp_greaterequal_s(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output, int core_mask) +.. c:function:: void dp_greaterequal_s(int element_num, int optimize, int in_elements_num0, double *input1, double *input2, bool *output, int core_mask) +.. c:function:: void i8_greaterequal_s(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output, int core_mask) +.. c:function:: void i16_greaterequal_s(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output, int core_mask) +.. c:function:: void i32_greaterequal_s(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input1 = (float *)0xA0000000; // Scalar input, DDR + float *input2 = (float *)0xB0000000; // Tensor input + bool *output = (bool *)0xC0000000; // Output + + // Broadcasting input1 (scalar) >= input2 (tensor) + int element_num = 4096; // Size of input2 and output + int optimize = 1; // Enable broadcasting + int in_elements_num0 = 1; // Size of input1 + int core_mask = 0xff; + + fp_greaterequal_s(element_num, optimize, in_elements_num0, input1, input2, output, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_greaterequal_p(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output) +.. c:function:: void hp_greaterequal_p(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output) +.. c:function:: void i32_greaterequal_p(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output) +.. c:function:: void i16_greaterequal_p(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output) +.. c:function:: void i8_greaterequal_p(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input1 = (float *)0x10000000; // Tensor input, L2 + float *input2 = (float *)0x11000000; // Scalar input + bool *output = (bool *)0x12000000; // Output + + // Broadcasting input1 (tensor) >= input2 (scalar) + int element_num = 1024; // Size of input1 and output + int optimize = 1; // Enable broadcasting + int in_elements_num0 = 1024; // Size of input1 + + fp_greaterequal_p(element_num, optimize, in_elements_num0, input1, input2, output); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/groupnormfusion.rst.txt b/master/html/_sources/functionlib/dsplib/groupnormfusion.rst.txt index 0947ddc..dca4936 100644 --- a/master/html/_sources/functionlib/dsplib/groupnormfusion.rst.txt +++ b/master/html/_sources/functionlib/dsplib/groupnormfusion.rst.txt @@ -42,7 +42,7 @@ GroupNormFusion .. code-block:: c :linenos: - :emphasize-lines: 17 + :emphasize-lines: 17-19 // FT78NE 多核示例 #include @@ -76,7 +76,7 @@ GroupNormFusion .. code-block:: c :linenos: - :emphasize-lines: 16 + :emphasize-lines: 16-18 // MT7004 单核示例 #include diff --git a/master/html/_sources/functionlib/dsplib/hashtablelookup.rst.txt b/master/html/_sources/functionlib/dsplib/hashtablelookup.rst.txt new file mode 100644 index 0000000..5ff11fe --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/hashtablelookup.rst.txt @@ -0,0 +1,105 @@ +HashtableLookup +================= + +根据输入的键值在已排序的键数组中执行二分查找,返回对应的字符串值以及查找命中标记。 + +虽然名为 `HashtableLookup`,但实际使用二分查找(`bsearch`)算法,而非哈希表。要求 `keys_tensor` 必须已按升序排序。 + +.. math:: + + \forall i \in [0, input\_size), \quad + \begin{cases} + \text{if } \exists j \in [0, keys\_size) \text{ s.t. } keys[j] = input[i], & + \begin{cases} + output[i] \leftarrow values[j] \\ + hits[i] \leftarrow 1 + \end{cases} \\[10pt] + \text{else}, & + \begin{cases} + output[i] \leftarrow \text{空字符串} \\ + hits[i] \leftarrow 0 + \end{cases} + \end{cases} + +输入: + - **input_tensor** - 要查找的键数组地址(int32类型)。 + - **keys_tensor** - 已排序的键数组地址(int32类型,必须按升序排序)。 + - **values_tensor** - 值数组地址(字符串tensor,与keys_tensor一一对应)。 + - **input_size** - 输入键数组的元素个数。 + - **keys_size** - 键数组的元素个数(等于values_tensor中字符串的个数)。 + - **core_mask(可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output_tensor** - 查找结果输出地址(字符串tensor,与input_tensor长度相同)。 + - **hits_tensor** - 命中标记输出地址(uint8类型,1表示找到,0表示未找到)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int32键类型,字符串值类型 + - MT7004 支持int32键类型,字符串值类型 + - keys_tensor 必须已按升序排序,否则查找结果不正确 + - 使用二分查找算法,时间复杂度为 O(log n) + - 如果查找失败,output_tensor 对应位置为空字符串,hits_tensor 对应位置为0 + +**共享存储版本:** + +.. c:function:: void i32_hashtablelookup_s(int* input_tensor, int* keys_tensor, void* values_tensor, void* output_tensor, uint8_t* hits_tensor, int input_size, int keys_size, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15-17 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + int *input_tensor = (int *)0xA0000000; //input在DDR空间 + int *keys_tensor = (int *)0xB0000000; //已排序的键数组 + void *values_tensor = (void *)0xC0000000; //字符串值数组 + void *output_tensor = (void *)0xD0000000; //输出字符串数组 + uint8_t *hits_tensor = (uint8_t *)0xE0000000; //命中标记数组 + int input_size = 100; //输入键的个数 + int keys_size = 50; //键值对的个数 + int core_mask = 0xff; + + i32_hashtablelookup_s(input_tensor, keys_tensor, values_tensor, + output_tensor, hits_tensor, input_size, + keys_size, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i32_hashtablelookup_p(int* input_tensor, int* keys_tensor, void* values_tensor, void* output_tensor, uint8_t* hits_tensor, int input_size, int keys_size) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14-16 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + int *input_tensor = (int *)0x10810000; //input在L2空间 + int *keys_tensor = (int *)0x10820000; //已排序的键数组 + void *values_tensor = (void *)0x10830000; //字符串值数组 + void *output_tensor = (void *)0x10840000; //输出字符串数组 + uint8_t *hits_tensor = (uint8_t *)0x10850000; //命中标记数组 + int input_size = 100; //输入键的个数 + int keys_size = 50; //键值对的个数 + + i32_hashtablelookup_p(input_tensor, keys_tensor, values_tensor, + output_tensor, hits_tensor, input_size, + keys_size); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/instancenorm.rst.txt b/master/html/_sources/functionlib/dsplib/instancenorm.rst.txt new file mode 100644 index 0000000..a798a0e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/instancenorm.rst.txt @@ -0,0 +1,111 @@ +InstanceNorm +================= + + + +对输入张量按 **实例(Instance)+ 通道(Channel)** 维度执行归一化操作。 +该算子在每个样本的每个通道内,基于 ``inner_size`` 维度计算均值与方差, +并结合可学习参数 ``gamma`` 与 ``beta`` 完成缩放与偏移。 + +.. math:: + + \mu_{b,c} = \frac{1}{N} \sum_{i=1}^{N} x_{b,c,i} + + \sigma^2_{b,c} = \frac{1}{N} \sum_{i=1}^{N} x_{b,c,i}^2 - \mu_{b,c}^2 + + y_{b,c,i} = \left( \frac{x_{b,c,i} - \mu_{b,c}}{\sqrt{\sigma^2_{b,c} + \epsilon}} \right) + \cdot \gamma_c + \beta_c + +其中: + +- :math:`b` 表示 batch 维度 +- :math:`c` 表示通道维度 +- :math:`i` 表示 ``inner_size`` 维度 +- :math:`\gamma_c`、:math:`\beta_c` 为通道级缩放与偏移参数 + +输入: + - **input** - 输入数据地址,形状为 ``[batch, channel, inner_size]``。 + - **gamma** - 缩放参数地址,长度为 ``channel``。 + - **beta** - 偏移参数地址,长度为 ``channel``。 + - **batch** - batch 数。 + - **channel** - 通道数。 + - **inner_size** - 每个通道内的归一化长度。 + - **epsilon** - 数值稳定因子。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - InstanceNorm 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32`` 类型 + - MT7004 支持 ``fp16``、``fp32`` 类型 + - 归一化统计量仅在单个样本、单个通道内计算 + +**共享存储版本:** + +.. c:function:: void fp_instancenorm_s(float* input, float* gamma, float* beta, float* output, int batch, int channel, int inner_size, float epsilon, int core_mask) +.. c:function:: void hp_instancenorm_s(half* input, half* gamma, half* beta, half* output, int batch, int channel, int inner_size, float epsilon, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17-18 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + float *output = (float *)0xC0000000; + float *gamma = (float *)0xA1000000; + float *beta = (float *)0xA2000000; + + int batch = 4; + int channel = 64; + int inner_size = 256; + float epsilon = 1e-5f; + int core_mask = 0xff; + + fp_instancenorm_s(input, gamma, beta, output, + batch, channel, inner_size, epsilon, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_instancenorm_p(float* input, float* gamma, float* beta, float* output, int batch, int channel, int inner_size, float epsilon) +.. c:function:: void hp_instancenorm_p(half* input, half* gamma, half* beta, half* output, int batch, int channel, int inner_size, float epsilon) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16-17 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; // input 在 L2 空间 + float *output = (float *)0x10820000; + float *gamma = (float *)0x10830000; + float *beta = (float *)0x10840000; + + int batch = 4; + int channel = 64; + int inner_size = 256; + float epsilon = 1e-5f; + + fp_instancenorm_p(input, gamma, beta, output, + batch, channel, inner_size, epsilon); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/invertpermutation.rst.txt b/master/html/_sources/functionlib/dsplib/invertpermutation.rst.txt new file mode 100644 index 0000000..a468155 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/invertpermutation.rst.txt @@ -0,0 +1,79 @@ +InvertPermutation +================= + + + +计算输入排列的逆排列。对于输入数组 ``input``,输出数组 ``output`` 满足: + +.. math:: + + output[input_i] = i + \quad \text{for } i = 0, 1, \dots, num-1 + +输入: + - **input** - 输入排列数组地址。 + - **num** - 数组长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 逆排列数组地址,长度同输入。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32 + - MT7004 支持int16, int32 + +**共享存储版本:** + +.. c:function:: void i8_invertpermutation_s(int8_t* input, int8_t* output, int num, int core_mask) +.. c:function:: void i16_invertpermutation_s(int16_t* input, int16_t* output, int num, int core_mask) +.. c:function:: void i32_invertpermutation_s(int32_t* input, int32_t* output, int num, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main() { + int32_t *input = (int32_t *)0xA0000000; // input在DDR空间 + int32_t *output = (int32_t *)0xC0000000; + int num = 100; + int core_mask = 0xff; + + i32_invertpermutation_s(input, output, num, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_invertpermutation_p(int8_t* input, int8_t* output, int num) +.. c:function:: void i16_invertpermutation_p(int16_t* input, int16_t* output, int num) +.. c:function:: void i32_invertpermutation_p(int32_t* input, int32_t* output, int num) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main() { + int32_t *input = (int32_t *)0x10810000; // input在L2空间 + int32_t *output = (int32_t *)0x10820000; + int num = 100; + + i32_invertpermutation_p(input, output, num); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/isfinite.rst.txt b/master/html/_sources/functionlib/dsplib/isfinite.rst.txt new file mode 100644 index 0000000..a21ec92 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/isfinite.rst.txt @@ -0,0 +1,86 @@ +Isfinite +================= + + +逐元素判断输入数据是否为有限数(既不是无穷大也不是 NaN)。 + +.. math:: + + output_i = \begin{cases} + 1, & \text{if } isfinite(Input_i) \\ + 0, & \text{otherwise} + \end{cases} + +输入: + - Input - 输入数据地址。 + - length - 计算长度(对于复数类型,指复数的个数)。 + - core_mask - 核掩码(仅共享存储版本需要)。 + +输出: + - output - 计算结果地址。0表示不是有限数,1表示是有限数。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_isfinite_s(int8_t* Input, int* output, int length, int core_mask) +.. c:function:: void i16_isfinite_s(int16_t* Input, int* output, int length, int core_mask) +.. c:function:: void i32_isfinite_s(int32_t* Input, int* output, int length, int core_mask) +.. c:function:: void hp_isfinite_s(half* Input, int* output, int length, int core_mask) +.. c:function:: void fp_isfinite_s(float* Input, int* output, int length, int core_mask) +.. c:function:: void dp_isfinite_s(double* Input, int* output, int length, int core_mask) +.. c:function:: void c64_isfinite_s(float* Input, int* output, int length, int core_mask) +.. c:function:: void c128_isfinite_s(double* Input, int* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include // 假设头文件名为 isfinite.h + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + int *output = (int *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + fp_isfinite_s(input, output, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_isfinite_p(int8_t* Input, int* output, int length) +.. c:function:: void i16_isfinite_p(int16_t* Input, int* output, int length) +.. c:function:: void i32_isfinite_p(int32_t* Input, int* output, int length) +.. c:function:: void hp_isfinite_p(half* Input, int* output, int length) +.. c:function:: void fp_isfinite_p(float* Input, int* output, int length) +.. c:function:: void dp_isfinite_p(double* Input, int* output, int length) +.. c:function:: void c64_isfinite_p(float* Input, int* output, int length) +.. c:function:: void c128_isfinite_p(double* Input, int* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 8 + + //FT78NE示例 + #include + #include // 假设头文件名为 isfinite.h + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; //input在L2空间 + int *output = (int *)0x10001000; + int length = 1000; + fp_isfinite_p(input, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/l2norm.rst.txt b/master/html/_sources/functionlib/dsplib/l2norm.rst.txt new file mode 100644 index 0000000..7f1388e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/l2norm.rst.txt @@ -0,0 +1,97 @@ +L2norm +================= + + + +对输入向量进行 L2 归一化,并可选择性地应用 ReLU 或 ReLU6 激活函数。该算子首先用输入向量的 L2 范数(即欧几里得长度)去除向量中的每个元素,然后应用激活函数。 + +.. math:: + + S = \sqrt{\sum_{i=0}^{N-1} Input_i^2} + + \text{temp}_i = \frac{Input_i}{S} + + output_i = \begin{cases} + \min(6, \max(0, \text{temp}_i)), & \text{if } \text{is\_relu6} = 1 \\ + \max(0, \text{temp}_i), & \text{if } \text{is\_relu} = 1 \\ + \text{temp}_i, & \text{otherwise} + \end{cases} + +输入: + - **Input** - 输入数据地址。 + - **sqrt_sum** - 输入向量的 L2 范数(模),需要预先计算好。 + - **length** - 计算长度。 + - **is_relu** - 是否进行 ReLU 操作的标志(0或1)。 + - **is_relu6** - 是否进行 ReLU6 操作的标志(0或1)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **Output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void hp_l2norm_s(half* Input, half* output, half sqrt_sum, int length, int is_relu, int is_relu6, int core_mask) +.. c:function:: void fp_l2norm_s(float* Input, float* output, float sqrt_sum, int length, int is_relu, int is_relu6, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include + #include // 假设头文件名为 l2norm.h + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1024; + int core_mask = 0xff; + // 伪代码:用户需要自行计算L2范数 + float sum_of_squares = 0.0f; + for(int i=0; i + #include + #include // 假设头文件名为 l2norm.h + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; //input在L2空间 + float *output = (float *)0x10002000; + int length = 1024; + // 伪代码:用户需要自行计算L2范数 + float sum_of_squares = 0.0f; + for(int i=0; i + #include + + int main() { + float *src = (float *)0xA0000000; // 输入在DDR空间 + float *gamma = (float *)0xA1000000; + float *beta = (float *)0xA2000000; + float *dst = (float *)0xC0000000; + float *out_mean = (float *)0xD0000000; + float *out_variance = (float *)0xD1000000; + int param_inner_size = 16; + int param_outer_size = 1; + int norm_inner_size = 16; + int norm_outer_size = 2; + float epsilon = 1e-5; + int core_mask = 0xff; + int length = 1024; + + fp_layernormfusion_s(src, gamma, beta, dst, out_mean, out_variance, param_inner_size, param_outer_size, norm_inner_size, norm_outer_size, epsilon, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_layernormfusion_p(float* src_data, float* gamma_data, float* beta_data, float* dst_data, float* out_mean, float* out_variance, int param_inner_size, int param_outer_size, int norm_inner_size, int norm_outer_size, float epsilon) +.. c:function:: void hp_layernormfusion_p(half* src_data, half* gamma_data, half* beta_data, half* dst_data, half* out_mean, half* out_variance, int param_inner_size, int param_outer_size, int norm_inner_size, int norm_outer_size, float epsilon) +.. c:function:: void i8_layernormfusion_p(int8_t* src_data, int8_t* gamma_data, int8_t* beta_data, int8_t* dst_data, float* out_mean, float* out_variance, int param_inner_size, int param_outer_size, int norm_inner_size, int norm_outer_size, float epsilon) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + #include + #include + + int main() { + float *src = (float *)0x10810000; // 输入在L2空间 + float *gamma = (float *)0x10811000; + float *beta = (float *)0x10812000; + float *dst = (float *)0x10820000; + float *out_mean = (float *)0x10821000; + float *out_variance = (float *)0x10822000; + int param_inner_size = 16; + int param_outer_size = 1; + int norm_inner_size = 16; + int norm_outer_size = 2; + float epsilon = 1e-5; + + fp_layernormfusion_p(src, gamma, beta, dst, out_mean, out_variance, param_inner_size, param_outer_size, norm_inner_size, norm_outer_size, epsilon); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/layernormgrad.rst.txt b/master/html/_sources/functionlib/dsplib/layernormgrad.rst.txt new file mode 100644 index 0000000..f112280 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/layernormgrad.rst.txt @@ -0,0 +1,108 @@ +Layernormgrad +================= + + +计算 Layer Normalization 操作的梯度。该算子是 Layer Normalization 的反向传播部分,用于计算损失函数相对于输入 x、以及可学习参数 gamma 和 beta 的梯度。 + +.. math:: + + \text{dg}_i = \sum_{j} \text{dy}_j \cdot \frac{x_j - \mu}{\sqrt{\sigma^2 + \epsilon}} + + \text{db}_i = \sum_{j} \text{dy}_j + + \text{dx}_i = f(\text{dy}, x, \gamma, \mu, \sigma^2) + +其中 :math:\mu 是均值,:math:\sigma^2 是方差,:math:\epsilon 是一个为了防止除零而添加的极小值。dx 的计算较为复杂,它依赖于 dy、x 和 gamma。 + +输入: + - **x** - 前向传播时的输入数据地址。 + - **dy** - 后续层反向传播回来的梯度数据地址。 + - **var** - 前向传播时计算出的方差(variance)地址。 + - **mean** - 前向传播时计算出的均值(mean)地址。 + - **gamma** - 前向传播时使用的可学习缩放参数 :math:\gamma 地址。 + - **param_num** - 特征维度的大小,也是 gamma 和 beta 的大小。 + - **param_size** - 进行独立归一化的单元数量(例如批处理大小 Batch Size)。 + - **block_num** - 块的数量(通常等于 param_size)。 + - **block_size** - 每个块的大小(通常等于 param_num)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **dx** - 计算出的关于输入 x 的梯度地址。 + - **dg** - 计算出的关于参数 gamma 的梯度地址。 + - **db** - 计算出的关于参数 beta 的梯度地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void hp_layernormgrad_s(half* x, half* dy, half* var, half* mean, half* gamma, int param_num, int param_size, int block_num, int block_size, half* dx, half* dg, half* db, int core_mask) +.. c:function:: void fp_layernormgrad_s(float* x, float* dy, float* var, float* mean, float* gamma, int param_num, int param_size, int block_num, int block_size, float* dx, float* dg, float* db, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 20 + + //FT78NE示例 + #include + #include // 假设头文件名为 layernormgrad.h + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + float *x = (float *)0xA0000000; + float *dy = (float *)0xA1000000; + float *var = (float *)0xA2000000; + float *mean = (float *)0xA3000000; + float *gamma = (float *)0xA4000000; + float *dx = (float *)0xB0000000; + float *dg = (float *)0xB1000000; + float *db = (float *)0xB2000000; + + int param_num = 256; // 特征维度 + int param_size = 64; // Batch Size + int core_mask = 0xff; + + fp_layernormgrad_s(x, dy, var, mean, gamma, param_num, param_size, param_size, param_num, dx, dg, db, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_layernormgrad_p(half* x, half* dy, half* var, half* mean, half* gamma, int param_num, int param_size, int block_num, int block_size, half* dx, half* dg, half* db) +.. c:function:: void fp_layernormgrad_p(float* x, float* dy, float* var, float* mean, float* gamma, int param_num, int param_size, int block_num, int block_size, float* dx, float* dg, float* db) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 19 + + //FT78NE示例 + #include + #include // 假设头文件名为 layernormgrad.h + + int main(int argc, char* argv[]) { + // 假设在L2空间 + float *x = (float *)0x10000000; + float *dy = (float *)0x11000000; + float *var = (float *)0x12000000; + float *mean = (float *)0x13000000; + float *gamma = (float *)0x14000000; + float *dx = (float *)0x15000000; + float *dg = (float *)0x16000000; + float *db = (float *)0x17000000; + + int param_num = 256; // 特征维度 + int param_size = 64; // Batch Size + + fp_layernormgrad_p(x, dy, var, mean, gamma, param_num, param_size, param_size, param_num, dx, dg, db); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/less.rst.txt b/master/html/_sources/functionlib/dsplib/less.rst.txt new file mode 100644 index 0000000..2a33951 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/less.rst.txt @@ -0,0 +1,91 @@ +Less +================= + +逐元素计算第一个输入 (Input0) 的元素是否小于第二个输入 (Input1) 的对应元素。该算子支持广播机制。 + +.. math:: + + output_i = \begin{cases} + \text{True}, & \text{if } Input0_0 < Input1_i \text{ (Input0 is broadcasted)} \\ + \text{True}, & \text{if } Input0_i < Input1_0 \text{ (Input1 is broadcasted)} \\ + \text{True}, & \text{if } Input0_i < Input1_i \text{ (No broadcast)} \\ + \text{False}, & \text{otherwise} + \end{cases} + +输入: + - **Input0** - 第一个输入数据地址。 + - **Input1** - 第二个输入数据地址。 + - **length** - 计算的总长度。 + - **optimize** - 是否启用广播优化的标志。设置为1时启用。 + - **in_elements_num0** - Input0 的元素个数。当 optimize 启用时,用于判断哪个输入是标量。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **Output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64 + - MT7004 支持fp16, fp32, int16, int32 + +**共享存储版本:** + +.. c:function:: void i8_less_s(int8_t* Input0, int8_t* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void i16_less_s(int16_t* Input0, int16_t* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void i32_less_s(int32_t* Input0, int32_t* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void hp_less_s(half* Input0, half* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void fp_less_s(float* Input0, float* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void dp_less_s(double* Input0, double* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include // 假设头文件名为 less.h + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *input1 = (float *)0xB0000000; + bool *output = (bool *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + // 不使用广播,optimize=0, in_elements_num0 无意义 + fp_less_s(input0, input1, output, length, 0, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_less_p(int8_t* Input0, int8_t* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void i16_less_p(int16_t* Input0, int16_t* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void i32_less_p(int32_t* Input0, int32_t* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void hp_less_p(half* Input0, half* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void fp_less_p(float* Input0, float* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void dp_less_p(double* Input0, double* Input1, bool* output, int length, int optimize, int in_elements_num0) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include // 假设头文件名为 less.h + int main(int argc, char* argv[]) { + float *scalar_input = (float *)0x10000000; // 标量输入 (input0) + float *array_input = (float *)0x10001000; // 数组输入 (input1) + bool *output = (bool *)0x10002000; + int length = 1000; // 数组的长度 + scalar_input = 5.0f; // 设置标量值 + // 启用广播,比较一个标量和一个数组 + fp_less_p(scalar_input, array_input, output, length, 1, 1); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/lessequal.rst.txt b/master/html/_sources/functionlib/dsplib/lessequal.rst.txt new file mode 100644 index 0000000..4f70f7f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lessequal.rst.txt @@ -0,0 +1,91 @@ +Lessequal +================= + +逐元素计算第一个输入 (Input0) 的元素是否小于或等于第二个输入 (Input1) 的对应元素。该算子支持广播机制。 + +.. math:: + + output_i = \begin{cases} + \text{True}, & \text{if } Input0_0 \le Input1_i \text{ (Input0 is broadcasted)} \\ + \text{True}, & \text{if } Input0_i \le Input1_0 \text{ (Input1 is broadcasted)} \\ + \text{True}, & \text{if } Input0_i \le Input1_i \text{ (No broadcast)} \\ + \text{False}, & \text{otherwise} + \end{cases} + +输入: + - **Input0** - 第一个输入数据地址。 + - **Input1** - 第二个输入数据地址。 + - **length** - 计算的总长度。 + - **optimize** - 是否启用广播优化的标志。设置为1时启用。 + - **in_elements_num0** - Input0 的元素个数。当 optimize 启用时,用于判断哪个输入是标量。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **Output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64 + - MT7004 支持fp16, fp32, int16, int32 + +**共享存储版本:** + +.. c:function:: void i8_lessequal_s(int8_t* Input0, int8_t* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void i16_lessequal_s(int16_t* Input0, int16_t* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void i32_lessequal_s(int32_t* Input0, int32_t* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void hp_lessequal_s(half* Input0, half* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void fp_lessequal_s(float* Input0, float* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) +.. c:function:: void dp_lessequal_s(double* Input0, double* Input1, bool* output, int length, int optimize, int in_elements_num0, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include // 假设头文件名为 less.h + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *input1 = (float *)0xB0000000; + bool *output = (bool *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + // 不使用广播,optimize=0, in_elements_num0 无意义 + fp_lessequal_s(input0, input1, output, length, 0, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_lessequal_p(int8_t* Input0, int8_t* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void i16_lessequal_p(int16_t* Input0, int16_t* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void i32_lessequal_p(int32_t* Input0, int32_t* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void hp_lessequal_p(half* Input0, half* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void fp_lessequal_p(float* Input0, float* Input1, bool* output, int length, int optimize, int in_elements_num0) +.. c:function:: void dp_lessequal_p(double* Input0, double* Input1, bool* output, int length, int optimize, int in_elements_num0) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include // 假设头文件名为 less.h + int main(int argc, char* argv[]) { + float *scalar_input = (float *)0x10000000; // 标量输入 (input0) + float *array_input = (float *)0x10001000; // 数组输入 (input1) + bool *output = (bool *)0x10002000; + int length = 1000; // 数组的长度 + scalar_input = 5.0f; // 设置标量值 + // 启用广播,比较一个标量和一个数组 + fp_lessequal_p(scalar_input, array_input, output, length, 1, 1); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/linspace.rst.txt b/master/html/_sources/functionlib/dsplib/linspace.rst.txt index ba87351..1984dc4 100644 --- a/master/html/_sources/functionlib/dsplib/linspace.rst.txt +++ b/master/html/_sources/functionlib/dsplib/linspace.rst.txt @@ -32,7 +32,7 @@ LinSpace .. code-block:: c :linenos: - :emphasize-lines: 10 + :emphasize-lines: 9 #include @@ -54,7 +54,7 @@ LinSpace .. code-block:: c :linenos: - :emphasize-lines: 9 + :emphasize-lines: 8 #include diff --git a/master/html/_sources/functionlib/dsplib/log.rst.txt b/master/html/_sources/functionlib/dsplib/log.rst.txt new file mode 100644 index 0000000..a5a525e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/log.rst.txt @@ -0,0 +1,85 @@ +Log +================= + +逐元素计算自然对数(以 :math:`e` 为底)的对数函数。 + +.. math:: + + \text{output}_i = \ln(\text{input}_i) + +其中 :math:`\ln(\cdot)` 表示自然对数函数。 +当输入值小于等于 0 时,结果的行为依赖于具体平台和实现(可能产生 ``-inf`` 或 ``NaN``)。 + +输入: + - **input** - 输入张量的数据地址。 + - **length** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与 ``input`` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:fp32 + - MT7004 支持的数据类型:fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_log_s(float* input, float* output, int length, int core_mask) +.. c:function:: void hp_log_s(half* input, half* output, int length, int core_mask) +.. c:function:: void i16_log_s(int16_t* input, float* output, int length, int core_mask) +.. c:function:: void i32_log_s(int* input, float* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + float *output = (float *)0xB0000000; // output 在 DDR 空间 + + int length = 4096; + int core_mask = 0xff; + + fp_log_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_log_p(float* input, float* output, int length) +.. c:function:: void hp_log_p(half* input, half* output, int length) +.. c:function:: void i16_log_p(int16_t* input, float* output, int length) +.. c:function:: void i32_log_p(int* input, float* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // input 在 L2 空间 + float *output = (float *)0x11000000; // output 在 L2 空间 + + int length = 1024; + + fp_log_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/log1p.rst.txt b/master/html/_sources/functionlib/dsplib/log1p.rst.txt new file mode 100644 index 0000000..4d1eb82 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/log1p.rst.txt @@ -0,0 +1,78 @@ +Log1p +================= + +逐元素计算 log(1 + x) 的值,其中 x 是输入元素。该函数在 x 的值接近于零时,比直接计算 log(1 + x) 能够提供更高的精度。 + +.. math:: + + output_i = \ln(1 + Input_i) + +输入: + - **Input** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **Output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64 + - MT7004 支持fp16, fp32, int16, int32 + +**共享存储版本:** + +.. c:function:: void i8_log1p_s(int8_t* Input, float* output, int length, int core_mask) +.. c:function:: void i16_log1p_s(int16_t* Input, float* output, int length, int core_mask) +.. c:function:: void i32_log1p_s(int32_t* Input, float* output, int length, int core_mask) +.. c:function:: void hp_log1p_s(half* Input, half* output, int length, int core_mask) +.. c:function:: void fp_log1p_s(float* Input, float* output, int length, int core_mask) +.. c:function:: void dp_log1p_s(double* Input, double* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include // 假设头文件名为 log1p.h + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + fp_log1p_s(input, output, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_log1p_p(int8_t* Input, float* output, int length) +.. c:function:: void i16_log1p_p(int16_t* Input, float* output, int length) +.. c:function:: void i32_log1p_p(int32_t* Input, float* output, int length) +.. c:function:: void hp_log1p_p(half* Input, half* output, int length) +.. c:function:: void fp_log1p_p(float* Input, float* output, int length) +.. c:function:: void dp_log1p_p(double* Input, double* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 8 + + //FT78NE示例 + #include + #include // 假设头文件名为 log1p.h + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; //input在L2空间 + float *output = (float *)0x10001000; + int length = 1000; + fp_log1p_p(input, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/loggrad.rst.txt b/master/html/_sources/functionlib/dsplib/loggrad.rst.txt new file mode 100644 index 0000000..6b3fb7e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/loggrad.rst.txt @@ -0,0 +1,93 @@ +LogGrad +================= + + +计算对数函数(Log)的梯度。 + +该算子用于反向传播阶段,依据链式法则, +将上游梯度与 Log 输入张量进行逐元素相除, +其数学表达式如下: + +.. math:: + + \text{output}_i = \frac{\text{input0}_i}{\text{input1}_i} + +其中: + - ``input0`` 表示上游梯度(dY) + - ``input1`` 表示 Log 算子的输入(X) + + +输入: + - **input0** - 上游梯度张量的数据地址。 + - **input1** - Log 算子前向输入张量的数据地址。 + - **length** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出梯度张量的数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:fp32 + - MT7004 支持的数据类型:fp16, fp32 + - 当 ``input1`` 中存在 0 或极小值时,结果可能产生 ``±∞`` 或 ``NaN``,需由上层框架保证输入合法性 + + +**共享存储版本:** + +.. c:function:: void fp_log_grad_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void hp_log_grad_s(half* input0, half* input1, half* output, int length, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // 上游梯度 + float *input1 = (float *)0xA0100000; // Log 前向输入 + float *output = (float *)0xB0000000; // 输出梯度 + + int length = 4096; + int core_mask = 0xff; + + fp_log_grad_s(input0, input1, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_log_grad_p(float* input0, float* input1, float* output, int length) +.. c:function:: void hp_log_grad_p(half* input0, half* input1, half* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input0 = (half *)0x10000000; + half *input1 = (half *)0x10010000; + half *output = (half *)0x10020000; + + int length = 1024; + + hp_log_grad_p(input0, input1, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/logical_not.rst.txt b/master/html/_sources/functionlib/dsplib/logical_not.rst.txt new file mode 100644 index 0000000..165e8fc --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/logical_not.rst.txt @@ -0,0 +1,81 @@ +LogicalNot +================= + +对输入数据执行逐元素逻辑非(Logical NOT)运算。对于输入中的每个元素,如果元素值为 0(即为 False),则输出结果为 1(或该类型的 True 值);如果元素值为非 0(即为 True),则输出为 0。 + +.. math:: + + output_i = \neg (input_i \neq 0) + +输入: + - **input** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp) + - MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp) + - 输出结果的数据类型通常与输入数据类型保持一致。 + - 逻辑判断准则:0 值视为 False,任何非 0 值均视为 True。 + +**共享存储版本:** + +.. c:function:: void i8_logical_not_s(int8_t* input, int8_t* output, int length, int core_mask) +.. c:function:: void i16_logical_not_s(int16_t* input, int16_t* output, int length, int core_mask) +.. c:function:: void i32_logical_not_s(int32_t* input, int32_t* output, int length, int core_mask) +.. c:function:: void hp_logical_not_s(half* input, half* output, int length, int core_mask) +.. c:function:: void fp_logical_not_s(float* input, float* output, int length, int core_mask) +.. c:function:: void dp_logical_not_s(double* input, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 示例(共享存储多核并行) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *in = (float *)0xA0000000; // 输入在共享存储空间 + float *out = (float *)0xB0000000; // 输出在共享存储空间 + int length = 960001; + int core_mask = 0b1011; // 使用指定的核心掩码 + fp_logical_not_s(in, out, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_logical_not_p(int8_t* input, int8_t* output, int length) +.. c:function:: void i16_logical_not_p(int16_t* input, int16_t* output, int length) +.. c:function:: void i32_logical_not_p(int32_t* input, int32_t* output, int length) +.. c:function:: void hp_logical_not_p(half* input, half* output, int length) +.. c:function:: void fp_logical_not_p(float* input, float* output, int length) +.. c:function:: void dp_logical_not_p(double* input, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + // MT7004 示例(私有存储单核) + #include + + int main(int argc, char* argv[]) { + // 输入和输出均位于私有存储空间 + int *in = (int *)0x10000000; + int *out = (int *)0x10001000; + int length = 1024; + i32_logical_not_p(in, out, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/logical_or.rst.txt b/master/html/_sources/functionlib/dsplib/logical_or.rst.txt new file mode 100644 index 0000000..72912f0 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/logical_or.rst.txt @@ -0,0 +1,84 @@ +LogicalOr +================= + +对两个输入数据执行逐元素逻辑或(Logical OR)运算。对于输入中的每个元素,如果两个对应的输入元素中至少有一个不为 0(即为 True),则输出结果为 1(或该类型的 True 值);如果两者均为 0,则输出为 0。 + +.. math:: + + output_i = (input0_i \neq 0) \lor (input1_i \neq 0) + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp) + - MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp) + - 输出结果的数据类型通常与输入数据类型保持一致。 + - 逻辑判断准则:非 0 值视为 True,0 值视为 False。 + +**共享存储版本:** + +.. c:function:: void i8_logical_or_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_logical_or_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_logical_or_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_logical_or_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_logical_or_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_logical_or_s(double* input0, double* input1, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 示例(共享存储多核并行) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + int32_t *in0 = (int32_t *)0xA0000000; // 输入0在共享存储空间 + int32_t *in1 = (int32_t *)0xA1000000; // 输入1在共享存储空间 + int32_t *out = (int32_t *)0xB0000000; // 输出在共享存储空间 + int length = 960001; + int core_mask = 0b1011; // 指定参加计算的核心 + i32_logical_or_s(in0, in1, out, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_logical_or_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_logical_or_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_logical_or_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_logical_or_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_logical_or_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_logical_or_p(double* input0, double* input1, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // MT7004 示例(私有存储单核) + #include + + int main(int argc, char* argv[]) { + // 输入和输出均位于私有存储空间 + float *in0 = (float *)0x10000000; + float *in1 = (float *)0x10001000; + float *out = (float *)0x10002000; + int length = 1024; + fp_logical_or_p(in0, in1, out, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/logicaland.rst.txt b/master/html/_sources/functionlib/dsplib/logicaland.rst.txt new file mode 100644 index 0000000..31cbbec --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/logicaland.rst.txt @@ -0,0 +1,101 @@ +LogicalAnd +================= + + +逐元素执行逻辑与(Logical AND)运算。 + +对输入张量的对应元素先进行布尔化判断,再执行逻辑与运算,输出结果为 0 或 1, +并以与输入相同的数据类型返回。 + +.. math:: + + \text{output}_i = + \begin{cases} + 1, & \text{if } (\text{input0}_i \neq 0) \land (\text{input1}_i \neq 0) \\ + 0, & \text{otherwise} + \end{cases} + + +输入: + - **input0** - 第一个输入张量的数据地址。 + - **input1** - 第二个输入张量的数据地址。 + - **length** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与 ``input0``、``input1`` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:fp32, fp64, int8, int16, int32 + - MT7004 支持的数据类型:fp16, fp32, int16, int32 + - 输入值在参与逻辑运算前会被转换为布尔值(非 0 为 true,0 为 false) + - 输出结果仅为 0 或 1 + + +**共享存储版本:** + +.. c:function:: void fp_and_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_and_s(double* input0, double* input1, double* output, int length, int core_mask) +.. c:function:: void i8_and_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_and_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_and_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_and_s(half* input0, half* input1, half* output, int length, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // input0 在 DDR 空间 + float *input1 = (float *)0xA0010000; // input1 在 DDR 空间 + float *output = (float *)0xB0000000; // output 在 DDR 空间 + + int length = 4096; + int core_mask = 0xff; + + fp_and_s(input0, input1, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_and_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_and_p(double* input0, double* input1, double* output, int length) +.. c:function:: void i8_and_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_and_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_and_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_and_p(half* input0, half* input1, half* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input0 = (half *)0x10000000; // input0 在 L2 空间 + half *input1 = (half *)0x10002000; // input1 在 L2 空间 + half *output = (half *)0x10010000; // output 在 L2 空间 + + int length = 1024; + + hp_and_p(input0, input1, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/logsoftmax.rst.txt b/master/html/_sources/functionlib/dsplib/logsoftmax.rst.txt new file mode 100644 index 0000000..2ae0eeb --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/logsoftmax.rst.txt @@ -0,0 +1,94 @@ +LogSoftmax +================= + + + +对输入数组沿指定维度进行 Log-Softmax 计算,输出每个元素的对数概率值。 + +.. math:: + + \text{output}_{i} = \log\frac{\exp(\text{input}_{i})}{\sum_j \exp(\text{input}_j)} + \quad \text{for elements along the given axis} + +输入: + - **input_ptr** - 输入数据地址。 + - **axis** - 归一化的轴。 + - **n_dim** - 输入张量维度。 + - **inner_size** - 内部尺寸(轴之后的元素个数)。 + - **outter_size** - 外部尺寸(轴之前的元素个数)。 + - **axis_size** - 归一化轴的元素数量。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + - **sum_data** - 中间累加存储地址(用于存放指数和)。 + +输出: + - **output_ptr** - Log-Softmax 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, int8 + - MT7004 支持hp, fp + +**共享存储版本:** + +.. c:function:: void fp_logsoftmax_s(float* input_ptr, float* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask) +.. c:function:: void hp_logsoftmax_s(half* input_ptr, half* output_ptr, half* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask) +.. c:function:: void i8_logsoftmax_s(int8_t* input_ptr, int8_t* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include + + int main() { + float *input = (float *)0xA0000000; // input在DDR空间 + float *output = (float *)0xC0000000; + float *sum_data = (float *)0xD0000000; + int axis = 1; + int n_dim = 3; + int inner_size = 4; + int outter_size = 2; + int axis_size = 3; + int core_mask = 0xff; + + fp_logsoftmax_s(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_logsoftmax_p(float* input_ptr, float* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size) +.. c:function:: void hp_logsoftmax_p(half* input_ptr, half* output_ptr, half* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size) +.. c:function:: void i8_logsoftmax_p(int8_t* input_ptr, int8_t* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //FT78NE示例 + #include + #include + + int main() { + float *input = (float *)0x10810000; // input在L2空间 + float *output = (float *)0x10820000; + float *sum_data = (float *)0x10830000; + int axis = 1; + int n_dim = 3; + int inner_size = 4; + int outter_size = 2; + int axis_size = 3; + + fp_logsoftmax_p(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/lpnormalization.rst.txt b/master/html/_sources/functionlib/dsplib/lpnormalization.rst.txt new file mode 100644 index 0000000..3c415da --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lpnormalization.rst.txt @@ -0,0 +1,113 @@ +LpNormalization +================= + + + +对输入张量按 **实例(Instance)+ 通道(Channel)** 维度执行 Lp 归一化操作。 +该算子在每个样本的每个通道内,基于 ``inner_size`` 维度计算 Lp 范数, +并结合可学习参数 ``gamma`` 与 ``beta`` 完成缩放与偏移。 + +.. math:: + + \text{norm}_{b,c} = \left( \sum_{i=1}^{N} |x_{b,c,i}|^p + \epsilon \right)^{\frac{1}{p}} + + y_{b,c,i} = \frac{x_{b,c,i}}{\text{norm}_{b,c}} \cdot \gamma_c + \beta_c + +其中: + +- :math:`b` 表示 batch 维度 +- :math:`c` 表示通道维度 +- :math:`i` 表示 ``inner_size`` 维度 +- :math:`p` 为范数阶数 +- :math:`\gamma_c`、:math:`\beta_c` 为通道级缩放与偏移参数 + +输入: + - **input** - 输入数据地址,形状为 ``[batch, channel, inner_size]``。 + - **gamma** - 缩放参数地址,长度为 ``channel``。 + - **beta** - 偏移参数地址,长度为 ``channel``。 + - **p** - Lp 范数阶数。 + - **batch** - batch 数。 + - **channel** - 通道数。 + - **inner_size** - 每个通道内的归一化长度。 + - **epsilon** - 数值稳定因子。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - LpNormalization 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32`` 类型 + - MT7004 支持 ``fp16``、``fp32`` 类型 + - 归一化统计量仅在单个样本、单个通道内计算 + - ``p`` 为浮点数,可用于 L1、L2 等不同范数形式 + +**共享存储版本:** + +.. c:function:: void fp_lpnorm_s(float* input, float* gamma, float* beta, float* output, float p, int batch, int channel, int inner_size, float epsilon, int core_mask) +.. c:function:: void hp_lpnorm_s(half* input, half* gamma, half* beta, half* output, float p, int batch, int channel, int inner_size, float epsilon, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 18-19 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + float *output = (float *)0xC0000000; + float *gamma = (float *)0xA1000000; + float *beta = (float *)0xA2000000; + + int batch = 4; + int channel = 64; + int inner_size = 256; + float p = 2.0f; + float epsilon = 1e-6f; + int core_mask = 0xff; + + fp_lpnorm_s(input, gamma, beta, output, + p, batch, channel, inner_size, epsilon, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_lpnorm_p(float* input, float* gamma, float* beta, float* output, float p, int batch, int channel, int inner_size, float epsilon) +.. c:function:: void hp_lpnorm_p(half* input, half* gamma, half* beta, half* output, float p, int batch, int channel, int inner_size, float epsilon) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17-18 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; // input 在 L2 空间 + float *output = (float *)0x10820000; + float *gamma = (float *)0x10830000; + float *beta = (float *)0x10840000; + + int batch = 4; + int channel = 64; + int inner_size = 256; + float p = 2.0f; + float epsilon = 1e-6f; + + fp_lpnorm_p(input, gamma, beta, output, + p, batch, channel, inner_size, epsilon); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/lrn.rst.txt b/master/html/_sources/functionlib/dsplib/lrn.rst.txt new file mode 100644 index 0000000..4f978d7 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lrn.rst.txt @@ -0,0 +1,87 @@ +Lrn +================= + +执行局部响应归一化 (Local Response Normalization)。 + +.. math:: + + \text{output}_{i,j} = \text{input}_{i,j} \times \left( \text{bias} + \alpha \sum_{k=\max(0, j-d)}^{\min(C-1, j+d)} (\text{input}_{i,k})^2 \right)^{-\beta} + +其中,i 遍历每个 out_size 维度,j 表示当前通道,d 是 depth_radius,C 是总通道数 channel, :math:\alpha 和 :math:\beta 是缩放因子。 + +输入: + - **input** - 输入张量数据地址,其逻辑布局为 (out_size, channel)。 + - **out_size** - 空间/批处理维度的乘积 (例如 N*H*W)。 + - **channel** - 通道数 (C)。 + - **depth_radius** - 归一化窗口的半径。总的窗口大小为 2 * depth_radius + 1。 + - **alpha** - 缩放因子 :math:\alpha。 + - **beta** - 指数 :math:\beta。 + - **bias** - 偏置项。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **Output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void hp_lrn_s(half* input, half* output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias, int core_mask) +.. c:function:: void fp_lrn_s(float* input, float* output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include // 假设头文件名为 lrn.h + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xC0000000; + int out_size = 256; // 例如 N*H*W + int channel = 96; + int depth_radius = 5; + float alpha = 0.0001f; + float beta = 0.75f; + float bias = 1.0f; + int core_mask = 0xff; + fp_lrn_s(input, output, out_size, channel, depth_radius, alpha, beta, bias, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void hp_lrn_p(half* input, half* output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias) +.. c:function:: void fp_lrn_p(float* input, float* output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include // 假设头文件名为 lrn.h + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; //input在L2空间 + float *output = (float *)0x10010000; + int out_size = 256; // 例如 N*H*W + int channel = 96; + int depth_radius = 5; + float alpha = 0.0001f; + float beta = 0.75f; + float bias = 1.0f; + fp_lrn_p(input, output, out_size, channel, depth_radius, alpha, beta, bias); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/lsh_projection.rst.txt b/master/html/_sources/functionlib/dsplib/lsh_projection.rst.txt new file mode 100644 index 0000000..fae3fe1 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lsh_projection.rst.txt @@ -0,0 +1,100 @@ +LshProjection +================= + +局部敏感哈希(Locality Sensitive Hashing, LSH)投影算子。该算子通过多个哈希组(Hash Groups)对输入特征进行处理,每组生成一个指定位宽(bits_per_hash)的哈希签名(int32)。 + +内部逻辑基于 FNV1a 哈希算法和加权评分机制,将高维特征映射为低维的离散哈希值。 + +计算过程: + 1. 对于每个哈希组 $i$,循环执行 $j$ 次($j < bits\_per\_hash$)。 + 2. 每次根据特定的哈希种子 $seed_{i,j}$ 计算特征与权重的加权评分,并通过评分符号决定一个 Bit 位(0 或 1)。 + 3. 将生成的 Bit 位拼接成一个完整的 32 位整型哈希签名。 + +输入: + - **hash_seed** - 哈希种子数组地址(float 类型)。 + - **feature** - 输入特征向量地址(int32 类型)。 + - **weight** - 权重向量地址(类型随算子名而定,若为 NULL 则不加权)。 + - **hash_group_num** - 生成的哈希组数量(即输出结果的个数)。 + - **bits_per_hash** - 每组哈希签名的有效位数(通常 $\le 32$)。 + - **feature_num** - 特征向量的维度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 存储生成的哈希签名地址(int32 数组)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持权重类型:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp) + - MT7004 支持权重类型:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp) + - 特征输入(feature)在所有平台上固定为 **int32** 类型。 + - 输出结果(output)在所有平台上固定为 **int32** 类型。 + +**共享存储版本:** + +.. c:function:: void i8_lsh_projection_s(const float* hash_seed, const int32_t* feature, const int8_t* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask) +.. c:function:: void i16_lsh_projection_s(const float* hash_seed, const int32_t* feature, const int16_t* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask) +.. c:function:: void i32_lsh_projection_s(const float* hash_seed, const int32_t* feature, const int32_t* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask) +.. c:function:: void hp_lsh_projection_s(const float* hash_seed, const int32_t* feature, const half* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask) +.. c:function:: void fp_lsh_projection_s(const float* hash_seed, const int32_t* feature, const float* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask) +.. c:function:: void dp_lsh_projection_s(const float* hash_seed, const int32_t* feature, const double* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 示例:fp32 权重,多核共享存储 + #include "78NE/utils.h" + + int main() { + float* hash_seed = (float*)0xA0000000; + int32_t* feature = (int32_t*)0xA1000000; + float* weight = (float*)0xA2000000; + int32_t* output = (int32_t*)0xB0000000; + + int hash_group_num = 128; + int bits_per_hash = 16; + int feature_num = 64; + int core_mask = 0xFF; + + fp_lsh_projection_s(hash_seed, feature, weight, output, + hash_group_num, bits_per_hash, feature_num, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_lsh_projection_p(const float* hash_seed, const int32_t* feature, const int8_t* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num) +.. c:function:: void i16_lsh_projection_p(const float* hash_seed, const int32_t* feature, const int16_t* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num) +.. c:function:: void i32_lsh_projection_p(const float* hash_seed, const int32_t* feature, const int32_t* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num) +.. c:function:: void hp_lsh_projection_p(const float* hash_seed, const int32_t* feature, const half* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num) +.. c:function:: void fp_lsh_projection_p(const float* hash_seed, const int32_t* feature, const float* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num) +.. c:function:: void dp_lsh_projection_p(const float* hash_seed, const int32_t* feature, const double* weight, int32_t* output, int hash_group_num, int bits_per_hash, int feature_num) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + // MT7004 示例:fp16 (hp) 权重,私有存储单核 + #include + + int main() { + float* hash_seed = (float*)0x10000000; + int32_t* feature = (int32_t*)0x10001000; + half* weight = (half*)0x10002000; + int32_t* output = (int32_t*)0x10003000; + + int hash_group_num = 64; + int bits_per_hash = 8; + int feature_num = 32; + + hp_lsh_projection_p(hash_seed, feature, weight, output, + hash_group_num, bits_per_hash, feature_num); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/lstm.rst.txt b/master/html/_sources/functionlib/dsplib/lstm.rst.txt index f693278..117be50 100644 --- a/master/html/_sources/functionlib/dsplib/lstm.rst.txt +++ b/master/html/_sources/functionlib/dsplib/lstm.rst.txt @@ -25,13 +25,13 @@ LSTM 输入: - - **input** - 输入序列数据,形状为 :math:`(seq_len, batch, input_size)`,即每个时间步的输入特征。 - - **weight_i** - 输入到各门 :math:`(input、forget、cell、output)` 的权重矩阵,大小为 4 * hidden_size * input_size。 - - **weight_h** - 上一隐藏状态到各门的权重矩阵,大小为 :math:`4 * hidden_size * hidden_size` + - **input** - 输入序列数据,形状为 :math:`(seq\_len, batch, input\_size)`,即每个时间步的输入特征。 + - **weight_i** - 输入到各门 :math:`(input, forget, cell, output)` 的权重矩阵,大小为 4 * hidden_size * input_size。 + - **weight_h** - 上一隐藏状态到各门的权重矩阵,大小为 :math:`4 * hidden\_size * hidden\_size` - **input_bias** - 输入部分的偏置项,对应 4 个门的偏置。 - - **state_bias** - 隐藏状态部分的偏置项(也是 :math:`4 * hidden_size`),与 input_bias 一起求和形成总偏置。 - - **hidden_state** - 当前批次初始隐藏状态输入( :math:`h₀` ),执行后更新为最后时刻的隐藏状态输出( :math:`hₜ`) - - **cell_state** - 当前批次初始细胞状态输入( :math:`c₀`),执行后更新为最后时刻的细胞状态输出( :math:`cₜ`)。 + - **state_bias** - 隐藏状态部分的偏置项(也是 :math:`4 * hidden\_size`),与 input_bias 一起求和形成总偏置。 + - **hidden_state** - 当前批次初始隐藏状态输入( :math:`h_0` ),执行后更新为最后时刻的隐藏状态输出( :math:`h_t`) + - **cell_state** - 当前批次初始细胞状态输入( :math:`c_0`),执行后更新为最后时刻的细胞状态输出( :math:`c_t`)。 - **buffer** - 临时工作区指针数组(中间计算缓存,如门值、激活结果、临时矩阵等,用于优化性能)。 - **LstmParameter** - LSTM 配置参数结构体,包含输入大小、隐藏层维度、序列长度、是否双向等信息。 - **core_mask** - 核掩码(仅适用于共享存储版本)。 @@ -70,11 +70,12 @@ LSTM .. note:: - FT78NE 支持fp32 - - MT7004 支持fp32 + - MT7004 支持fp32、fp16 **共享存储版本:** .. c:function:: void fp_Lstm_s(float *output, const float *input, const float *weight_i, const float *weight_h, const float *input_bias,const float *state_bias, float *hidden_state, float *cell_state, float *buffer[9],const LstmParameter *lstm_param, int core_mask) +.. c:function:: void hp_Lstm_s(half *output, const half *input, const half *weight_i, const half *weight_h, const half *input_bias,const half *state_bias, half *hidden_state, half *cell_state, half *buffer[9],const LstmParameter *lstm_param, int core_mask) **C调用示例:** @@ -130,8 +131,9 @@ LSTM **私有存储版本:** .. c:function:: void fp_Lstm_p(float *output, const float *input, const float *weight_i, const float *weight_h, const float *input_bias, const float *state_bias, float *hidden_state, float *cell_state, float *buffer[9], const LstmParameter *lstm_param) - - **C调用示例:** +.. c:function:: void hp_Lstm_p(half *output, const half *input, const half *weight_i, const half *weight_h, const half *input_bias,const half *state_bias, half *hidden_state, half *cell_state, half *buffer[9],const LstmParameter *lstm_param) + + **C调用示例:** .. code-block:: c :linenos: diff --git a/master/html/_sources/functionlib/dsplib/lstmgrad.rst.txt b/master/html/_sources/functionlib/dsplib/lstmgrad.rst.txt new file mode 100644 index 0000000..e866c32 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lstmgrad.rst.txt @@ -0,0 +1,88 @@ +LstmGrad +================= + +计算 LSTM 网络的梯度,包括对输入、隐藏状态、细胞状态以及门控的梯度反向传播。 + +.. math:: + + dH_t &= dY_t + dH_{t+1} \\ + dC_t &= dC_{t+1} \odot f_{t+1} + dH_t \odot o_t \odot (1 - \tanh^2(C_t)) \\ + dX_t &= dA_t \cdot W^T \\ + dW &= \sum_t dA_t \cdot X_t^T \\ + dU &= \sum_t dA_t \cdot H_{t-1}^T \\ + dA_t &= dH_t \odot o_t \odot (1 - \tanh^2(C_t)) \odot g'_t + +其中: + +- \(dH_t\) 表示隐藏状态的梯度。 +- \(dC_t\) 表示细胞状态的梯度。 +- \(dX_t\) 表示输入梯度。 +- \(dW, dU\) 分别表示输入权重和隐藏状态权重的梯度。 +- \(dA_t\) 表示门控单元梯度。 +- \(f_t, o_t, C_t, g_t\) 分别为遗忘门、输出门、细胞状态、输入门的前向值。 +- \(\odot\) 表示元素逐乘。 + +输入: + - **params** - 静态参数数组,包含 LSTM 网络配置、权重、状态指针等。 + - **dynamic_params** - 动态参数数组,用于存储运行时指针及中间梯度。 + +输出: + - **dX_** - 输入梯度。 + - **dH_** - 隐藏状态梯度。 + - **dC_** - 细胞状态梯度。 + - **dA_tmp_** - 门控梯度中间缓存。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp + - MT7004 支持 hp, fp + +**共享存储版本:** + +.. c:function:: void fp_lstmgrad_s(long long *params, long long *dynamic_params, int core_mask) +.. c:function:: void hp_lstmgrad_s(long long *params, long long *dynamic_params, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + #include + + int main() { + long long params[32]; + long long dynamic_params[32]; + int core_mask = 0xff; + + // 初始化 params 和 dynamic_params + fp_lstmgrad_s(params, dynamic_params, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_lstmgrad_p(long long *params, long long *dynamic_params) +.. c:function:: void hp_lstmgrad_p(long long *params, long long *dynamic_params) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + #include + #include + + int main() { + long long params[32]; + long long dynamic_params[32]; + + // 初始化 params 和 dynamic_params + fp_lstmgrad_p(params, dynamic_params); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/lstmgraddata.rst.txt b/master/html/_sources/functionlib/dsplib/lstmgraddata.rst.txt new file mode 100644 index 0000000..dd6bb99 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lstmgraddata.rst.txt @@ -0,0 +1,88 @@ +LstmGradData +================= + +计算 LSTM 网络中每个时间步的输入梯度和隐藏状态梯度,支持单向和双向 LSTM。 + +.. math:: + + dH_t &= dY_t + dH_{t+1} \\ + dC_t &= dC_{t+1} \odot f_{t+1} + dH_t \odot o_t \odot (1 - \tanh^2(C_t)) \\ + dX_t &= dA_t \cdot W^T \\ + dW &= \sum_t dA_t \cdot X_t^T \\ + dU &= \sum_t dA_t \cdot H_{t-1}^T \\ + dA_t &= dH_t \odot o_t \odot (1 - \tanh^2(C_t)) \odot g'_t + +其中: + +- \(dH_t\) 表示隐藏状态的梯度。 +- \(dC_t\) 表示细胞状态的梯度。 +- \(dX_t\) 表示输入梯度。 +- \(dW, dU\) 分别表示输入权重和隐藏状态权重的梯度。 +- \(dA_t\) 表示门控单元梯度。 +- \(f_t, o_t, C_t, g_t\) 分别为遗忘门、输出门、细胞状态、输入门的前向值。 +- \(\odot\) 表示元素逐乘。 + +输入: + - **params** - 静态参数数组,包含 LSTM 网络配置、权重、状态指针等。 + - **dynamic_params** - 动态参数数组,用于存储运行时指针及中间梯度。 + +输出: + - **dX_** - 输入梯度。 + - **dH_** - 隐藏状态梯度。 + - **dC_** - 细胞状态梯度。 + - **dA_tmp_** - 门控梯度中间缓存。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp + - MT7004 支持 hp, fp + +**共享存储版本:** + +.. c:function:: void fp_lstmgraddata_s(long long *params, long long *dynamic_params, int core_mask) +.. c:function:: void hp_lstmgraddata_s(long long *params, long long *dynamic_params, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + #include + + int main() { + long long params[32]; + long long dynamic_params[32]; + int core_mask = 0xff; + + // 初始化 params 和 dynamic_params + fp_lstmgraddata_s(params, dynamic_params, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_lstmgraddata_p(long long *params, long long *dynamic_params) +.. c:function:: void hp_lstmgraddata_p(long long *params, long long *dynamic_params) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + #include + #include + + int main() { + long long params[32]; + long long dynamic_params[32]; + + // 初始化 params 和 dynamic_params + fp_lstmgraddata_p(params, dynamic_params); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/lstmgradweight.rst.txt b/master/html/_sources/functionlib/dsplib/lstmgradweight.rst.txt new file mode 100644 index 0000000..7a00120 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/lstmgradweight.rst.txt @@ -0,0 +1,88 @@ +LstmGradWeight +================= + +计算 LSTM 网络的权重梯度,包括输入权重 W、隐藏权重 U 和偏置 b 的反向传播梯度。 + +.. math:: + + dH_t &= dY_t + dH_{t+1} \\ + dC_t &= dC_{t+1} \odot f_{t+1} + dH_t \odot o_t \odot (1 - \tanh^2(C_t)) \\ + dX_t &= dA_t \cdot W^T \\ + dW &= \sum_t dA_t \cdot X_t^T \\ + dU &= \sum_t dA_t \cdot H_{t-1}^T \\ + dA_t &= dH_t \odot o_t \odot (1 - \tanh^2(C_t)) \odot g'_t + +其中: + +- \(dH_t\) 表示隐藏状态的梯度。 +- \(dC_t\) 表示细胞状态的梯度。 +- \(dX_t\) 表示输入梯度。 +- \(dW, dU\) 分别表示输入权重和隐藏状态权重的梯度。 +- \(dA_t\) 表示门控单元梯度。 +- \(f_t, o_t, C_t, g_t\) 分别为遗忘门、输出门、细胞状态、输入门的前向值。 +- \(\odot\) 表示元素逐乘。 + +输入: + - **params** - 静态参数数组,包含 LSTM 网络配置、权重、状态指针等。 + - **dynamic_params** - 动态参数数组,用于存储运行时指针及中间梯度。 + +输出: + - **dX_** - 输入梯度。 + - **dH_** - 隐藏状态梯度。 + - **dC_** - 细胞状态梯度。 + - **dA_tmp_** - 门控梯度中间缓存。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp + - MT7004 支持 hp, fp + +**共享存储版本:** + +.. c:function:: void fp_lstmgradweight_s(long long *params, long long *dynamic_params, int core_mask) +.. c:function:: void hp_lstmgradweight_s(long long *params, long long *dynamic_params, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + #include + #include + + int main() { + long long params[32]; + long long dynamic_params[32]; + int core_mask = 0xff; + + // 初始化 params 和 dynamic_params + fp_lstmgradweight_s(params, dynamic_params, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_lstmgradweight_p(long long *params, long long *dynamic_params) +.. c:function:: void hp_lstmgradweight_p(long long *params, long long *dynamic_params) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + #include + #include + + int main() { + long long params[32]; + long long dynamic_params[32]; + + // 初始化 params 和 dynamic_params + fp_lstmgradweight_p(params, dynamic_params); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/maximum.rst.txt b/master/html/_sources/functionlib/dsplib/maximum.rst.txt new file mode 100644 index 0000000..8b0d21d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/maximum.rst.txt @@ -0,0 +1,83 @@ +Maximum +================= + +对两个输入数据执行逐元素取最大值操作。对于每一对对应的输入元素,比较其大小,并将较大者存入输出地址。 + +.. math:: + + output_i = \max(input0_i, input1_i) + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp) + - MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp) + - 两个输入数据与输出数据的数据类型必须一致。 + +**共享存储版本:** + +.. c:function:: void i8_maximum_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_maximum_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_maximum_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_maximum_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_maximum_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_maximum_s(double* input0, double* input1, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 示例(共享存储多核并行) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *in0 = (float *)0xA0000000; // 输入0在共享存储空间 + float *in1 = (float *)0xA1000000; // 输入1在共享存储空间 + float *out = (float *)0xB0000000; // 输出在共享存储空间 + int length = 960001; + int core_mask = 0xFF; // 使用所有核心 + fp_maximum_s(in0, in1, out, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_maximum_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_maximum_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_maximum_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_maximum_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_maximum_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_maximum_p(double* input0, double* input1, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // MT7004 示例(私有存储单核) + #include + + int main(int argc, char* argv[]) { + // 输入与输出均位于芯片私有存储空间 + int *in0 = (int *)0x10000000; + int *in1 = (int *)0x10001000; + int *out = (int *)0x10002000; + int length = 1024; + i32_maximum_p(in0, in1, out, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/maximumgrad.rst.txt b/master/html/_sources/functionlib/dsplib/maximumgrad.rst.txt new file mode 100644 index 0000000..93f1684 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/maximumgrad.rst.txt @@ -0,0 +1,98 @@ +Maximumgrad +================= + +计算逐元素 Maximum 操作的梯度。该算子是 Maximum 算子的反向传播部分。梯度 dy 将被路由到在前向传播中值较大的那个输入。 + +.. math:: + + \text{dx0}_i = \begin{cases} + \text{dy}_i, & \text{if } \text{Input0}_i > \text{Input1}_i \\ + 0, & \text{otherwise} + \end{cases} + + \text{dx1}_i = \begin{cases} + \text{dy}_i, & \text{if } \text{Input1}_i \ge \text{Input0}_i \\ + 0, & \text{otherwise} + \end{cases} + +输入: + - **Input0** - 前向传播时的第一个输入数据地址。 + - **Input1** - 前向传播时的第二个输入数据地址。 + - **dy** - 后续层反向传播回来的梯度数据地址。 + - **Input0_dims** - Input0 的维度信息数组。 + - **Input1_dims** - Input1 的维度信息数组。 + - **num_dims** - 输入张量的维度数量。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **dx0** - 计算出的关于 Input0 的梯度地址。 + - **dx1** - 计算出的关于 Input1 的梯度地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void hp_maximumgrad_s(half* Input0, half* Input1, half* dy, int* Input0_dims, int* Input1_dims, int num_dims, half* dx0, half* dx1, int core_mask) +.. c:function:: void fp_maximumgrad_s(float* Input0, float* Input1, float* dy, int* Input0_dims, int* Input1_dims, int num_dims, float* dx0, float* dx1, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + //FT78NE示例 + #include + #include // 假设头文件名为 maximumgrad.h + + int main(int argc, char* argv[]) { + // 假设在DDR空间,且形状相同 + float *input0 = (float *)0xA0000000; + float *input1 = (float *)0xA1000000; + float *dy = (float *)0xA2000000; + float *dx0 = (float *)0xB0000000; + float *dx1 = (float *)0xB1000000; + + int dims[] = {4, 256}; + int num_dims = 2; + int core_mask = 0xff; + + fp_maximumgrad_s(input0, input1, dy, dims, dims, num_dims, dx0, dx1, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void hp_maximumgrad_p(half* Input0, half* Input1, half* dy, int* Input0_dims, int* Input1_dims, int num_dims, half* dx0, half* dx1) +.. c:function:: void fp_maximumgrad_p(float* Input0, float* Input1, float* dy, int* Input0_dims, int* Input1_dims, int num_dims, float* dx0, float* dx1) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include // 假设头文件名为 maximumgrad.h + + int main(int argc, char* argv[]) { + // 假设在L2空间,且形状相同 + float *input0 = (float *)0x10000000; + float *input1 = (float *)0x11000000; + float *dy = (float *)0x12000000; + float *dx0 = (float *)0x13000000; + float *dx1 = (float *)0x14000000; + + int dims[] = {4, 256}; + int num_dims = 2; + + fp_maximumgrad_p(input0, input1, dy, dims, dims, num_dims, dx0, dx1); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/maxpoolfusion.rst.txt b/master/html/_sources/functionlib/dsplib/maxpoolfusion.rst.txt new file mode 100644 index 0000000..9597ce6 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/maxpoolfusion.rst.txt @@ -0,0 +1,235 @@ +MaxPoolFusion +================= + + + 对输入张量执行最大池化操作 + + .. math:: + + \text{output}_{b, h_o, w_o, c} = + \operatorname{clip}\Bigg( + \max_{h_i, w_i \in \mathcal{W}(h_o, w_o)} + \Big( + \text{input}_{b,\; h_i,\; w_i,\; c} + \Big),\; + \text{minf},\; \text{maxf} + \Bigg) + + 其中,窗口区域 :math:`\mathcal{W}(h_o, w_o)` 的定义如下: + + .. math:: + + h_i = h_o \cdot \text{stride}_h - \text{pad}_u + \Delta h + + w_i = w_o \cdot \text{stride}_w - \text{pad}_l + \Delta w + + \Delta h \in [0,\ \text{win}_h - 1],\quad + \Delta w \in [0,\ \text{win}_w - 1] + + 有效窗口点满足: + + .. math:: + + 0 \le h_i < \text{in}_h,\qquad + 0 \le w_i < \text{in}_w + + 原始窗口起点定义为: + + .. math:: + + h_{\text{start}} = h_o \cdot \text{stride}_h - \text{pad}_u + + .. math:: + + w_{\text{start}} = w_o \cdot \text{stride}_w - \text{pad}_l + + 合法采样范围为: + + .. math:: + + \Delta h \in + \Big[ + \max(0,\ -h_{\text{start}}),\; + \min(\text{win}_h,\ \text{in}_h - h_{\text{start}}) + \Big) + + .. math:: + + \Delta w \in + \Big[ + \max(0,\ -w_{\text{start}}),\; + \min(\text{win}_w,\ \text{in}_w - w_{\text{start}}) + \Big) + + 最大池化计算: + + .. math:: + + v_{\max} = + \max_{\Delta h,\ \Delta w}\; + \text{input}_{b,\; + h_{\text{start}} + \Delta h,\; + w_{\text{start}} + \Delta w,\; + c} + + 最终输出: + + .. math:: + + \text{output}_{b, h_o, w_o, c} = + \min\big(\max(v_{\max},\ \text{minf}),\ \text{maxf}\big) + + + + 输入: + - **input** - 输入张量指针,采用 **NHWC 格式**,形状为 :math:`[batch,\ in\_h,\ in\_w,\ channel]` + - **in_w** - 输入张量的宽度 (W) + - **in_h** - 输入张量的高度 (H) + - **win_w** - 池化窗口的宽度,即窗口在 W 方向的大小 + - **win_h** - 池化窗口的高度,即窗口在 H 方向的大小 + - **output_w** - 输出特征图的宽度 + - **output_h** - 输出特征图的高度 + - **batch** - 批次大小,即输入中的 batch 数 + - **channel** - 通道数 C ,每个池化位置都分别对 C 个通道独立执行最大池化与裁剪 + - **stride_w** - 池化窗口在 W 方向的步长 + - **stride_h** - 池化窗口在 H 方向的步长 + - **pad_l** - 输入特征图左侧的填充大小 + - **pad_u** - 输入特征图上侧的填充大小 + - **minf** - 输出结果的下界值。池化结果会执行 :math:`\max(v,\ \text{minf})` + - **maxf** - 输出结果的上界值。池化结果会执行 :math:`\min(v,\ \text{maxf})` + - **core_mask** - 核心掩码,指定使用的计算核心 + + 输出: + - **output** - 输出张量指针,采用 **NHWC 格式**,形状为 :math:`[batch,\ output\_h,\ output\_w,\ channel]`。 + + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持fp32, fp64 + - MT7004 支持fp16, fp32 + - 调用时将除 core_mask 外的参数打包通过 long long params 数组传入,顺序为: + input, output, in_w, in_h, win_w, win_h, output_w, output_h, batch, channel, + stride_w, stride_h, pad_l, pad_u, minf, maxf + +**共享存储版本:** + +.. c:function:: void hp_maxpool_fusion_s(long long *params, int core_mask) +.. c:function:: void fp_maxpool_fusion_s(long long *params, int core_mask) +.. c:function:: void dp_maxpool_fusion_s(long long *params, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 45 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float* input_ptr = (float*)0xA0000000; + float* output_ptr = (float*)0xB0000000; + float* check_ptr = (float*)0xC0000000; + int in_w = 32; + int in_h = 32; + int win_w = 6; + int win_h = 6; + int batch = 4; + int channel = 2; + int stride_w = 4; + int stride_h = 4; + int pad_l = 0; + int pad_u = 0; + float minf = 0.0f; + float maxf = 50.0f; + + // 根据标准公式计算输出尺寸 + int dividor = in_w + pad_l * 2 - win_w; + int output_w = (dividor + stride_w - 1) / stride_w + 1; + int dividor2 = in_h + pad_u * 2 - win_h; + int output_h = (dividor2 + stride_h - 1) / stride_h + 1; + + long long params[16]; + params[0] = (long long)input_ptr; + params[1] = (long long)output_ptr; + params[2] = (long long)in_w; + params[3] = (long long)in_h; + params[4] = (long long)win_w; + params[5] = (long long)win_h; + params[6] = (long long)output_w; + params[7] = (long long)output_h; + params[8] = (long long)batch; + params[9] = (long long)channel; + params[10] = (long long)stride_w; + params[11] = (long long)stride_h; + params[12] = (long long)pad_l; + params[13] = (long long)pad_u; + params[14] = (long long)&minf; + params[15] = (long long)&maxf; + int core_mask = 0x0f; + fp_maxpool_fusion_s(params, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_maxpool_fusion_p(long long *params) +.. c:function:: void fp_maxpool_fusion_p(long long *params) +.. c:function:: void dp_maxpool_fusion_p(long long *params) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 44 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float* input_ptr = (float*)0xA0000000; + float* output_ptr = (float*)0xB0000000; + float* check_ptr = (float*)0xC0000000; + int in_w = 32; + int in_h = 32; + int win_w = 6; + int win_h = 6; + int batch = 4; + int channel = 2; + int stride_w = 4; + int stride_h = 4; + int pad_l = 0; + int pad_u = 0; + float minf = 0.0f; + float maxf = 50.0f; + + // 根据标准公式计算输出尺寸 + int dividor = in_w + pad_l * 2 - win_w; + int output_w = (dividor + stride_w - 1) / stride_w + 1; + int dividor2 = in_h + pad_u * 2 - win_h; + int output_h = (dividor2 + stride_h - 1) / stride_h + 1; + + long long params[16]; + params[0] = (long long)input_ptr; + params[1] = (long long)output_ptr; + params[2] = (long long)in_w; + params[3] = (long long)in_h; + params[4] = (long long)win_w; + params[5] = (long long)win_h; + params[6] = (long long)output_w; + params[7] = (long long)output_h; + params[8] = (long long)batch; + params[9] = (long long)channel; + params[10] = (long long)stride_w; + params[11] = (long long)stride_h; + params[12] = (long long)pad_l; + params[13] = (long long)pad_u; + params[14] = (long long)&minf; + params[15] = (long long)&maxf; + fp_maxpool_fusion_p(params); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/maxpoolgrad.rst.txt b/master/html/_sources/functionlib/dsplib/maxpoolgrad.rst.txt new file mode 100644 index 0000000..6841365 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/maxpoolgrad.rst.txt @@ -0,0 +1,209 @@ +MaxPoolGrad +================= + + 描述 MaxPool 的反向传播(梯度)计算。该算子将上游梯度(dy)只回传到前向最大池化过程中被选为最大值的位置;其它位置的梯度为 0。 + + 数学定义: + + .. math:: + + \text{output}_{b,\ h_i,\ w_i,\ c} = + \begin{cases} + \text{dy}_{b,\ h_o,\ w_o,\ c}, & + \text{if } (h_i,\ w_i) + = \displaystyle \arg\max_{(h,w)\in\mathcal{W}(h_o,w_o)} + \text{input}_{b,\ h,\ w,\ c}, \\ + 0, & \text{otherwise}. + \end{cases} + + 其中,:math:`\mathcal{W}(h_o, w_o)` 表示输出位置 :math:`(h_o, w_o)` 对应的池化窗口区域。窗口像素位置 :math:`(h, w)` 可表示为: + + .. math:: + + h = h_o \cdot \text{stride}_h - \text{pad}_u + \Delta h + + .. math:: + + w = w_o \cdot \text{stride}_w - \text{pad}_l + \Delta w + + .. math:: + + \Delta h \in [0,\ \text{win}_h - 1], \qquad + \Delta w \in [0,\ \text{win}_w - 1] + + 并且仅当采样点落在输入有效范围内时会被考虑: + + .. math:: + + 0 \le h < \text{in}_h, \qquad 0 \le w < \text{in}_w. + + 实现细节说明: + - 前向池化使用窗口 :math:`\text{win}_h \times \text{win}_w`,步长为 :math:`\text{stride}_h`, :math:`\text{stride}_w`,并且在边界处使用 pad(pad\_u, pad\_l)。 + + - 反向传播时,输出梯度 tensor(即需要写入的输入梯度)在每个 batch 开始前先被初始化为 0(代码中有一次整体清零)。 + + - 对于每个输出像素 :math:`(h_o,w_o)` 以及每个通道 c: + + - 在对应的输入窗口中找到前向最大值的位置 :math:`(h^*,w^*)`; + + - 将上游梯度 :math:`\text{dy}_{b,h_o,w_o,c}` 累加到该位置::math:`\text{output}_{b,h^*,w^*,c} \mathrel{+}= \text{dy}_{b,h_o,w_o,c}`。 + + - 其他位置梯度保持 0。 + + 输入: + - **input** - 输入张量指针,采用 **NHWC 格式**,形状为 :math:`[batch,\ in\_h,\ in\_w,\ channel]` + - **dy** - 上游梯度张量指针,采用 **NHWC 格式**,形状为 :math:`[batch,\ output\_h,\ output\_w,\ channel]` + - **in_w** - 输入张量的宽度 (W) + - **in_h** - 输入张量的高度 (H) + - **win_w** - 池化窗口的宽度,即窗口在 W 方向的大小 + - **win_h** - 池化窗口的高度,即窗口在 H 方向的大小 + - **output_w** - 输出特征图的宽度 + - **output_h** - 输出特征图的高度 + - **batch** - 批次大小,即输入中的 batch 数 + - **channel** - 通道数 C ,每个池化位置都分别对 C 个通道独立执行最大池化与裁剪 + - **stride_w** - 池化窗口在 W 方向的步长 + - **stride_h** - 池化窗口在 H 方向的步长 + - **pad_l** - 输入特征图左侧的填充大小 + - **pad_u** - 输入特征图上侧的填充大小 + - **minf** - 输出结果的下界值。池化结果会执行 :math:`\max(v,\ \text{minf})` + - **maxf** - 输出结果的上界值。池化结果会执行 :math:`\min(v,\ \text{maxf})` + - **core_mask** - 核心掩码,指定使用的计算核心 + + 输出: + - **output** - 输出张量指针,采用 **NHWC 格式**,形状为 :math:`[batch,\ in\_h,\ in\_w,\ channel]`。 + + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持fp32, fp64 + - MT7004 支持fp16, fp32 + - 调用时将除 core_mask 外的参数打包通过 long long params 数组传入,顺序为: + input, dy, output, in_w, in_h, win_w, win_h, output_w, output_h, batch, channel, + stride_w, stride_h, pad_l, pad_u, minf, maxf + +**共享存储版本:** + +.. c:function:: void hp_maxpool_grad_s(long long *params, int core_mask) +.. c:function:: void fp_maxpool_grad_s(long long *params, int core_mask) +.. c:function:: void dp_maxpool_grad_s(long long *params, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 47 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + double* input_ptr = (double*)0xA0000000; + double* dy_ptr = (double*)0xB0000000; + double* output_ptr = (double*)0xC0000000; + double* check_ptr = (double*)0xD0000000; + int in_w = gin_w; + int in_h = gin_h; + int win_w = 6; + int win_h = 6; + int batch = gbatch; + int channel = 2; + int stride_w = 4; + int stride_h = 4; + int pad_l = 1; + int pad_u = 1; + double minf = 0.0f; + double maxf = 50.0f; + + // 根据标准公式计算输出尺寸 + int dividor = in_w + pad_l*2 - win_w; + int output_w = (dividor + stride_w - 1) / stride_w + 1; + int dividor2 = in_h + pad_u*2 - win_h; + int output_h = (dividor2 + stride_h - 1) / stride_h + 1; + + long long params[17]; + params[0] = (long long)input_ptr; + params[1] = (long long)dy_ptr; + params[2] = (long long)output_ptr; + params[3] = (long long)in_w; + params[4] = (long long)in_h; + params[5] = (long long)win_w; + params[6] = (long long)win_h; + params[7] = (long long)output_w; + params[8] = (long long)output_h; + params[9] = (long long)batch; + params[10] = (long long)channel; + params[11] = (long long)stride_w; + params[12] = (long long)stride_h; + params[13] = (long long)pad_l; + params[14] = (long long)pad_u; + params[15] = (long long)&minf; //注意这里传指针,不能直接强制转换成long long + params[16] = (long long)&maxf; + int core_mask = 0x0f; + fp_maxpool_grad_s(params, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void hp_maxpool_grad_p(long long *params) +.. c:function:: void fp_maxpool_grad_p(long long *params) +.. c:function:: void dp_maxpool_grad_p(long long *params) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 46 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + double* input_ptr = (double*)0xA0000000; + double* dy_ptr = (double*)0xB0000000; + double* output_ptr = (double*)0xC0000000; + double* check_ptr = (double*)0xD0000000; + int in_w = gin_w; + int in_h = gin_h; + int win_w = 6; + int win_h = 6; + int batch = gbatch; + int channel = 2; + int stride_w = 4; + int stride_h = 4; + int pad_l = 1; + int pad_u = 1; + double minf = 0.0f; + double maxf = 50.0f; + + // 根据标准公式计算输出尺寸 + int dividor = in_w + pad_l*2 - win_w; + int output_w = (dividor + stride_w - 1) / stride_w + 1; + int dividor2 = in_h + pad_u*2 - win_h; + int output_h = (dividor2 + stride_h - 1) / stride_h + 1; + + long long params[17]; + params[0] = (long long)input_ptr; + params[1] = (long long)dy_ptr; + params[2] = (long long)output_ptr; + params[3] = (long long)in_w; + params[4] = (long long)in_h; + params[5] = (long long)win_w; + params[6] = (long long)win_h; + params[7] = (long long)output_w; + params[8] = (long long)output_h; + params[9] = (long long)batch; + params[10] = (long long)channel; + params[11] = (long long)stride_w; + params[12] = (long long)stride_h; + params[13] = (long long)pad_l; + params[14] = (long long)pad_u; + params[15] = (long long)&minf; //注意这里传指针,不能直接强制转换成long long + params[16] = (long long)&maxf; + fp_maxpool_grad_p(params); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/mfcc.rst.txt b/master/html/_sources/functionlib/dsplib/mfcc.rst.txt new file mode 100644 index 0000000..bd9a562 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/mfcc.rst.txt @@ -0,0 +1,222 @@ +MFCC +====== + +计算梅尔频率倒谱系数(Mel-Frequency Cepstral Coefficients)。 + +.. math:: + + MFCC_i = f(Input_i) + +输入: + - **mfcc\_params** - MFCC 计算相关的配置参数,如采样率、梅尔滤波器数量等。 + - **spec\_params** - 频谱计算参数,定义了FFT窗口大小、重叠长度等。 + - **mfcc\_workspace** - 工作空间参数,包括计算所需的中间缓冲区。 + - **spec\_workspace** - 频谱计算的工作空间,包括FFT的窗口、输入数据等。 + - **core\_mask** (可选) - 核掩码(仅适用于多核版本)。 + +输出: + - **output(mfcc\_params中)** - 计算结果地址,存储MFCC特征。 + - **output_shape(mfcc\_params中)** - 输出数据的形状,描述MFCC特征的维度。 + +**结构体定义:** + .. code-block:: c + :linenos: + + typedef struct { + // 输入/输出 + float* input; // 输入数据地址 + int* input_shape; // 输入数据形状 + float* output; // 输出数据地址 + int* output_shape; // 输出数据形状 + + // 配置参数 + int sample_rate; // 采样率 + int n_mfcc; // MFCC系数数量 + int dct_type; // DCT变换类型 + bool log_mels; // 是否对梅尔频谱取对数 + float f_min; // 最小频率 + float f_max; // 最大频率 + int n_mels; // 梅尔滤波器数量 + NormType norm; // 归一化类型 + NormMode norm_M; // 归一化模式 + MelType mel_scale; // 梅尔尺度类型 + } MfccParam; + + typedef struct { + // 输入 + float* input; // 输入数据地址 + int input_len; // 输入数据长度 + + // 输出 + float* output; // 输出数据地址 + int* output_shape; // 输出数据形状 + + // 配置参数 + int pad; // 填充大小 + WindowType window_type; // 窗函数类型 + int n_fft; // FFT点数 + int hop_length; // 帧移长度 + int win_length; // 窗长度 + float power; // 功率值 + bool normalized; // 是否归一化 + bool center; // 是否中心化 + BorderType pad_mode; // 填充模式 + bool onesided; // 是否单边频谱 + } SpectrogramParam; + + typedef struct { + // 滤波器组计算缓冲区 + float *all_freqs; // 所有频率点 + float *m_pts; // 梅尔频率点 + float *f_pts; // 线性频率点 + float *fb; // 滤波器组 + + // 三角滤波器计算缓冲区 + float *f_diff; // 频率差值 + float *slopes; // 斜率 + float *down_slopes; // 下降斜率 + float *up_slopes; // 上升斜率 + + // DCT变换缓冲区 + float* dct_mat; // DCT矩阵 + + // 中间结果缓冲区 + float* spectrogram_output; // 频谱图输出 + float* mel_spectrogram_output; // 梅尔频谱输出 + } MfccWorkspaceParam; + + typedef struct { + float* fft_window; // FFT窗函数缓冲区 + float* fft_window_later; // 后续FFT窗缓冲区 + float* input_data_pad; // 填充后的输入数据 + float* input_data; // 输入数据缓冲区 + float* input_win; // 加窗输入数据 + float* exp_complex; // 复数指数缓冲区 + float* spec_f; // 频谱频率缓冲区 + float* output_onsided; // 单边输出缓冲区 + } WorkspaceParam; +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp32, fp16 + +**共享存储版本:** + +.. c:function:: void fp_mfcc_s(const MfccParam* mfcc_params, SpectrogramParam* spec_params, MfccWorkspaceParam* mfcc_workspace, WorkspaceParam* spec_workspace, int core_mask) +.. c:function:: void hp_mfcc_s(const MfccParam* mfcc_params, SpectrogramParam* spec_params, MfccWorkspaceParam* mfcc_workspace, WorkspaceParam* spec_workspace, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 45 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float* input_data = (float*)0x81000000; + float* output_mfcc = (float*)0x82000000; + int input_shape[2] = {1, 4000}; + int output_shape[2] = {0, 0, 0}; + MfccParam* mfcc_params = (MfccParam*)0x83100000; + SpectrogramParam* spec_params = (SpectrogramParam*)0x83200000; + MfccWorkspaceParam* mfcc_workspace = (MfccWorkspaceParam*)0x83300000; + WorkspaceParam* spec_workspace = (WorkspaceParam*)0x83400000; + + mfcc_params->input = input_data; + mfcc_params->input_shape = input_shape; + mfcc_params->output = output_mfcc; + mfcc_params->output_shape = output_shape; + mfcc_params->sample_rate = 800; + mfcc_params->n_mfcc = 6; + mfcc_params->dct_type = 2; + mfcc_params->log_mels = true; + mfcc_params->f_min = 0.0f; + mfcc_params->f_max = 400.0f; + mfcc_params->n_mels = 8; + mfcc_params->norm = NORM_SLANEY; + mfcc_params->norm_M = NORM_MODE_ORTHO; + mfcc_params->mel_scale = MEL_HTK; + + // 结构体 2: SpectrogramParam + spec_params->pad = 0; + spec_params->window_type = kHann; + spec_params->n_fft = 32; + spec_params->hop_length = 16; + spec_params->win_length = 32; + spec_params->power = 2.0f; + spec_params->normalized = false; + spec_params->center = false; + spec_params->pad_mode = kConstant; + spec_params->onesided = true; + + int core_mask = 0xff; + //为mfcc_workspace和spec_workspace里每个中间缓冲区指针分配地址 + //... + + fp_mfcc_s(mfcc_params, spec_params, mfcc_workspace, spec_workspace, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_mfcc_p(const MfccParam* mfcc_params, SpectrogramParam* spec_params, MfccWorkspaceParam* mfcc_workspace, WorkspaceParam* spec_workspace) +.. c:function:: void hp_mfcc_p(const MfccParam* mfcc_params, SpectrogramParam* spec_params, MfccWorkspaceParam* mfcc_workspace, WorkspaceParam* spec_workspace) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 45 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + MfccParam mfcc_params; + SpectrogramParam spec_params; + MfccWorkspaceParam mfcc_workspace; + WorkspaceParam spec_workspace; + + float* input_data = (float*)0x10810000; + float* output_mfcc = (float*)0x10820000; + int input_shape[2] = {1, 4000}; + int output_shape[2] = {0, 0, 0}; + + mfcc_params.input = input_data; + mfcc_params.input_shape = input_shape; + mfcc_params.output = output_mfcc; + mfcc_params.output_shape = output_shape; + mfcc_params.sample_rate = 800; + mfcc_params.n_mfcc = 6; + mfcc_params.dct_type = 2; + mfcc_params.log_mels = true; + mfcc_params.f_min = 0.0f; + mfcc_params.f_max = 400.0f; + mfcc_params.n_mels = 8; + mfcc_params.norm = NORM_SLANEY; + mfcc_params.norm_M = NORM_MODE_ORTHO; + mfcc_params.mel_scale = MEL_HTK; + + // 结构体 2: SpectrogramParam + spec_params.pad = 0; + spec_params.window_type = kHann; + spec_params.n_fft = 32; + spec_params.hop_length = 16; + spec_params.win_length = 32; + spec_params.power = 2.0f; + spec_params.normalized = false; + spec_params.center = false; + spec_params.pad_mode = kConstant; + spec_params.onesided = true; + + //为mfcc_workspace和spec_workspace里每个中间缓冲区指针分配地址 + //... + + fp_mfcc_p(&mfcc_params, &spec_params, &mfcc_workspace, &spec_workspace); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/minimum.rst.txt b/master/html/_sources/functionlib/dsplib/minimum.rst.txt new file mode 100644 index 0000000..4b1be03 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/minimum.rst.txt @@ -0,0 +1,83 @@ +Minimum +================= + +对两个输入数据执行逐元素取最小值操作。对于每一对对应的输入元素,比较其大小,并将较小者存入输出地址。 + +.. math:: + + output_i = \min(input0_i, input1_i) + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp) + - MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp) + - 两个输入数据与输出数据的数据类型必须一致。 + +**共享存储版本:** + +.. c:function:: void i8_minimum_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_minimum_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_minimum_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_minimum_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_minimum_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_minimum_s(double* input0, double* input1, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 示例(共享存储多核并行) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + double *in0 = (double *)0xA0000000; // 输入0在共享存储空间 + double *in1 = (double *)0xA1000000; // 输入1在共享存储空间 + double *out = (double *)0xB0000000; // 输出在共享存储空间 + int length = 960001; + int core_mask = 0xFF; // 使用所有核心并行计算 + dp_minimum_s(in0, in1, out, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_minimum_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_minimum_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_minimum_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_minimum_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_minimum_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_minimum_p(double* input0, double* input1, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // MT7004 示例(私有存储单核) + #include + + int main(int argc, char* argv[]) { + // 输入与输出均位于芯片私有存储空间 + half *in0 = (half *)0x10000000; + half *in1 = (half *)0x10001000; + half *out = (half *)0x10002000; + int length = 1024; + hp_minimum_p(in0, in1, out, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/minimumgrad.rst.txt b/master/html/_sources/functionlib/dsplib/minimumgrad.rst.txt new file mode 100644 index 0000000..121d5b9 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/minimumgrad.rst.txt @@ -0,0 +1,98 @@ +Minimumgrad +================= + +计算逐元素 Minimum 操作的梯度。该算子是 Minimum 算子的反向传播部分。梯度 dy 将被路由到在前向传播中值较小的那个输入。 + +.. math:: + + \text{dx0}_i = \begin{cases} + \text{dy}_i, & \text{if } \text{Input0}_i < \text{Input1}_i \\ + 0, & \text{otherwise} + \end{cases} + + \text{dx1}_i = \begin{cases} + \text{dy}_i, & \text{if } \text{Input1}_i \le \text{Input0}_i \\ + 0, & \text{otherwise} + \end{cases} + +输入: + - **Input0** - 前向传播时的第一个输入数据地址。 + - **Input1** - 前向传播时的第二个输入数据地址。 + - **dy** - 后续层反向传播回来的梯度数据地址。 + - **Input0_dims** - Input0 的维度信息数组。 + - **Input1_dims** - Input1 的维度信息数组。 + - **num_dims** - 输入张量的维度数量。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **dx0** - 计算出的关于 Input0 的梯度地址。 + - **dx1** - 计算出的关于 Input1 的梯度地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void hp_minimumgrad_s(half* Input0, half* Input1, half* dy, int* Input0_dims, int* Input1_dims, int num_dims, half* dx0, half* dx1, int core_mask) +.. c:function:: void fp_minimumgrad_s(float* Input0, float* Input1, float* dy, int* Input0_dims, int* Input1_dims, int num_dims, float* dx0, float* dx1, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + //FT78NE示例 + #include + #include // 假设头文件名为 minimumgrad.h + + int main(int argc, char* argv[]) { + // 假设在DDR空间,且形状相同 + float *input0 = (float *)0xA0000000; + float *input1 = (float *)0xA1000000; + float *dy = (float *)0xA2000000; + float *dx0 = (float *)0xB0000000; + float *dx1 = (float *)0xB1000000; + + int dims[] = {4, 256}; + int num_dims = 2; + int core_mask = 0xff; + + fp_minimumgrad_s(input0, input1, dy, dims, dims, num_dims, dx0, dx1, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void hp_minimumgrad_p(half* Input0, half* Input1, half* dy, int* Input0_dims, int* Input1_dims, int num_dims, half* dx0, half* dx1) +.. c:function:: void fp_minimumgrad_p(float* Input0, float* Input1, float* dy, int* Input0_dims, int* Input1_dims, int num_dims, float* dx0, float* dx1) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include // 假设头文件名为 minimumgrad.h + + int main(int argc, char* argv[]) { + // 假设在L2空间,且形状相同 + float *input0 = (float *)0x10000000; + float *input1 = (float *)0x11000000; + float *dy = (float *)0x12000000; + float *dx0 = (float *)0x13000000; + float *dx1 = (float *)0x14000000; + + int dims[] = {4, 256}; + int num_dims = 2; + + fp_minimumgrad_p(input0, input1, dy, dims, dims, num_dims, dx0, dx1); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/mod.rst.txt b/master/html/_sources/functionlib/dsplib/mod.rst.txt new file mode 100644 index 0000000..74169d6 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/mod.rst.txt @@ -0,0 +1,84 @@ +Mod +================= + +对两个输入数据执行逐元素取模(取余)操作。对于整数类型,执行标准的 C 语言 ``%`` 运算;对于浮点类型,执行类似于 ``fmod`` 的余数运算。 + +.. math:: + + output_i = input0_i \pmod{input1_i} + +输入: + - **input0** - 第一个输入数据地址(被除数)。 + - **input1** - 第二个输入数据地址(除数)。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp) + - MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp) + - 浮点数取模运算遵循标准 C 库函数 ``fmod`` 的行为。 + - 若除数元素为 0,结果为未定义行为,需由上层逻辑保证除数非 0。 + +**共享存储版本:** + +.. c:function:: void i8_mod_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_mod_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_mod_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_mod_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_mod_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_mod_s(double* input0, double* input1, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 示例(共享存储多核并行) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + int32_t *in0 = (int32_t *)0xA0000000; // 输入0在共享存储空间 + int32_t *in1 = (int32_t *)0xA1000000; // 输入1在共享存储空间 + int32_t *out = (int32_t *)0xB0000000; // 输出在共享存储空间 + int length = 10000; + int core_mask = 0xFF; // 使用所有核心进行并行取模计算 + i32_mod_s(in0, in1, out, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_mod_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_mod_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_mod_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_mod_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_mod_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_mod_p(double* input0, double* input1, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // MT7004 示例(私有存储单核) + #include + + int main(int argc, char* argv[]) { + // 输入与输出均位于私有存储空间 + float *in0 = (float *)0x10000000; + float *in1 = (float *)0x10001000; + float *out = (float *)0x10002000; + int length = 512; + fp_mod_p(in0, in1, out, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/mul.rst.txt b/master/html/_sources/functionlib/dsplib/mul.rst.txt new file mode 100644 index 0000000..3db9cb3 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/mul.rst.txt @@ -0,0 +1,91 @@ +Mul +================= + +对两个输入数据执行逐元素乘法运算。支持实数类型和复数类型。 + +.. math:: + + \text{对于实数类型:}\quad output_i = input0_i \times input1_i + +.. math:: + + \text{对于复数类型:}\quad (a+bi) \times (c+di) = (ac - bd) + (ad + bc)i + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **length** - 计算长度(对于复数,指复数的个数)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp), cplx64 (c64), cplx128 (c128) + - MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp), cplx64 (c64) + - 复数类型(cplx64/cplx128)在内存中以实部、虚部交替存储,运算遵循复数乘法公式。 + +**共享存储版本:** + +.. c:function:: void i8_mul_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_mul_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_mul_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_mul_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_mul_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_mul_s(double* input0, double* input1, double* output, int length, int core_mask) +.. c:function:: void c64_mul_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void c128_mul_s(double* input0, double* input1, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 示例:复数类型 cplx64 共享存储多核计算 + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *in0 = (float *)0xA0000000; // 输入0 (包含 real, imag) + float *in1 = (float *)0xA1000000; // 输入1 + float *out = (float *)0xB0000000; // 输出 + int num_complex = 480000; // 复数个数 + int core_mask = 0xFF; + c64_mul_s(in0, in1, out, num_complex, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_mul_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_mul_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_mul_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_mul_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_mul_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_mul_p(double* input0, double* input1, double* output, int length) +.. c:function:: void c64_mul_p(float* input0, float* input1, float* output, int length) +.. c:function:: void c128_mul_p(double* input0, double* input1, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // MT7004 示例:fp16 (half) 类型私有存储单核计算 + #include + + int main(int argc, char* argv[]) { + // 输入与输出均位于私有存储空间 + half *in0 = (half *)0x10000000; + half *in1 = (half *)0x10001000; + half *out = (half *)0x10002000; + int length = 1024; + hp_mul_p(in0, in1, out, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/mulgrad.rst.txt b/master/html/_sources/functionlib/dsplib/mulgrad.rst.txt new file mode 100644 index 0000000..1f6c304 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/mulgrad.rst.txt @@ -0,0 +1,158 @@ +Mulgrad +================= + + +计算逐元素乘法 (Mul) 操作的梯度。该算子是 Mul 算子的反向传播(backward pass)部分。梯度的计算遵循链式法则。 + +.. math:: + + \text{dx0}_i = \text{dy}_i \times \text{Input1}_i + + \text{dx1}_i = \text{dy}_i \times \text{Input0}_i + +其中 dx0 和 dx1 分别是损失函数对前向输入 Input0 和 Input1 的梯度。 + +Gradmul1L版本专门用于 `x1` 张量维度大于或等于 `x2` 张量的广播场景。Gradmul2l版本专门用于 `x2` 张量维度大于或等于 `x1` 张量的广播场景。 + +输入: + - **dy** - 来自后一层的上游梯度张量。 + - **x1** - 前向传播时的第一个输入张量(被除数)。 + - **x2** - 前向传播时的第二个输入张量(除数)。 + - **large_shape** - `x1` 和 `x2` 中维度较大的张量的形状。 + - **small_shape** - `x1` 和 `x2` 中维度较小的张量的形状。 + - **out_shape** - 输出张量 `dx1` 和 `dx2` 的形状。 + - **ndims** - 张量的维度数。 + - **large_strides** - 维度较大张量的步长信息。 + - **small_strides** - 维度较小张量的步长信息。 + - **out_strides** - 输出张量的步长信息。 + - **large_multiples** - 维度较大张量的广播倍数。 + - **small_multiples** - 维度较小张量的广播倍数。 + - **tile_data0** - 临时工作空间地址。 + - **tile_data1** - 临时工作空间地址。 + - **indices** - 用于广播计算的临时索引空间地址。 + - **core_mask** - 核掩码。 + +输出: + - **dx1** - 写入计算出的对 `x1` 的梯度。 + - **dx2** - 写入计算出的对 `x2` 的梯度。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_gradmul_s(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, int* indices, int core_mask) +.. c:function:: void hp_gradmul_s(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, int* indices, int core_mask) +.. c:function:: void fp_gradmul1l_s(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, int* indices, int core_mask) +.. c:function:: void hp_gradmul1l_s(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, int* indices, int core_mask) +.. c:function:: void fp_gradmul2l_s(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, int* indices, int core_mask) +.. c:function:: void hp_gradmul2l_s(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, int* indices, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 38 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0xA1000000; + float *dx1 = (float *)0xA2000000; + float *dx2 = (float *)0xA3000000; + float *x1_data = (float *)0xA4000000; + float *x2_data = (float *)0xA5000000; + float *tile_data0 = (float *)0xA6000000; + float *tile_data1 = (float *)0xA7000000; + + long long ndims = 4; + long long dy_size; + long long x1_size; + long long x2_size; + + int *large_strides = (int *)0xAB000000; + int *small_strides = (int *)0xAB100000; + int *out_strides = (int *)0xAB200000; + int *large_multiples = (int *)0xAB300000; + int *small_multiples = (int *)0xAB400000; + int *indices = (int *)0xAB500000; + int *large_shape = (int *)0xAB600000; + int *small_shape = (int *)0xAB700000; + int *out_shape = (int *)0xAB800000; + + large_shape[0] = 12; large_shape[1] = 14; large_shape[2] = 3; large_shape[3] = 5; + small_shape[0] = 12; small_shape[1] = 14; small_shape[2] = 3; small_shape[3] = 5; + out_shape[0] = 12; out_shape[1] = 14; out_shape[2] = 3; out_shape[3] = 5; + + int core_mask = 0xff; + + dy_size = out_shape[0] * out_shape[1] * out_shape[2] * out_shape[3]; + x1_size = large_shape[0] * large_shape[1] * large_shape[2] * large_shape[3]; + x2_size = small_shape[0] * small_shape[1] * small_shape[2] * small_shape[3]; + + fp_gradmul_s(dy, x1, x2, large_shape, small_shape, out_shape, ndims, large_strides, small_strides, out_strides, large_multiples, small_multiples, dx1, dx2, tile_data0, tile_data1, indices, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_gradmul_p(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, int* indices) +.. c:function:: void hp_gradmul_p(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1) +.. c:function:: void fp_gradmul1l_p(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, int* indices) +.. c:function:: void hp_gradmul1l_p(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, int* indices) +.. c:function:: void fp_gradmul2l_p(float* dy, float* x1, float* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, float* dx1, float* dx2, float* tile_data0, float* tile_data1, int* indices) +.. c:function:: void hp_gradmul2l_p(half* dy, half* x1, half* x2, int* large_shape, int* small_shape, int* out_shape, int ndims, int* large_strides, int* small_strides, int* out_strides, int* large_multiples, int* small_multiples, half* dx1, half* dx2, half* tile_data0, half* tile_data1, int* indices) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 36 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *dy = (float *)0x10000000; + float *dx1 = (float *)0x12000000; + float *dx2 = (float *)0x13000000; + float *x1_data = (float *)0x14000000; + float *x2_data = (float *)0x15000000; + float *tile_data0 = (float *)0x16000000; + float *tile_data1 = (float *)0x17000000; + + long long ndims = 4; + long long dy_size; + long long x1_size; + long long x2_size; + + int *large_strides = (int *)0x1B000000; + int *small_strides = (int *)0x1B100000; + int *out_strides = (int *)0x1B200000; + int *large_multiples = (int *)0x1B300000; + int *small_multiples = (int *)0x1B400000; + int *indices = (int *)0x1B500000; + int *large_shape = (int *)0x1B600000; + int *small_shape = (int *)0x1B700000; + int *out_shape = (int *)0x1B800000; + + large_shape[0] = 12; large_shape[1] = 14; large_shape[2] = 3; large_shape[3] = 5; + small_shape[0] = 12; small_shape[1] = 14; small_shape[2] = 3; small_shape[3] = 5; + out_shape[0] = 12; out_shape[1] = 14; out_shape[2] = 3; out_shape[3] = 5; + + dy_size = out_shape[0] * out_shape[1] * out_shape[2] * out_shape[3]; + x1_size = large_shape[0] * large_shape[1] * large_shape[2] * large_shape[3]; + x2_size = small_shape[0] * small_shape[1] * small_shape[2] * small_shape[3]; + + fp_gradmul_p(dy, x1, x2, large_shape, small_shape, out_shape, ndims, large_strides, small_strides, out_strides, large_multiples, small_multiples, dx1, dx2, tile_data0, tile_data1, indices); + return 0; + } + + \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/neg.rst.txt b/master/html/_sources/functionlib/dsplib/neg.rst.txt new file mode 100644 index 0000000..2165f5f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/neg.rst.txt @@ -0,0 +1,81 @@ +Neg +================= +逐元素计算输入数据的取负结果。 + +.. math:: + + output_i = -Input_i + +输入: + - **Input** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_neg_s(int8_t* Input, int8_t* output, int length, int core_mask) +.. c:function:: void i16_neg_s(int16_t* Input, int16_t* output, int length, int core_mask) +.. c:function:: void i32_neg_s(int32_t* Input, int32_t* output, int length, int core_mask) +.. c:function:: void hp_neg_s(half* Input, half* output, int length, int core_mask) +.. c:function:: void fp_neg_s(float* Input, float* output, int length, int core_mask) +.. c:function:: void dp_neg_s(double* Input, double* output, int length, int core_mask) +.. c:function:: void c64_neg_s(float* Input, float* output, int length, int core_mask) +.. c:function:: void c128_neg_s(double* Input, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xB0000000; + int length = 1000; + int core_mask = 0xff; + fp_neg_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_neg_p(int8_t* Input, int8_t* output, int length) +.. c:function:: void i16_neg_p(int16_t* Input, int16_t* output, int length) +.. c:function:: void i32_neg_p(int32_t* Input, int32_t* output, int length) +.. c:function:: void hp_neg_p(half* Input, half* output, int length) +.. c:function:: void fp_neg_p(float* Input, float* output, int length) +.. c:function:: void dp_neg_p(double* Input, double* output, int length) +.. c:function:: void c64_neg_p(float* Input, float* output, int length) +.. c:function:: void c128_neg_p(double* Input, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + //MT7004示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; //input在L2空间 + float *output = (float *)0x10001000; + int length = 1000; + fp_neg_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/neg_grad.rst.txt b/master/html/_sources/functionlib/dsplib/neg_grad.rst.txt new file mode 100644 index 0000000..c8bba1a --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/neg_grad.rst.txt @@ -0,0 +1,83 @@ +NegGrad +================= +逐元素计算输入数据的取负梯度结果。 + +.. math:: + + output_i = -Input_i + +输入: + - **Input** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + + +**共享存储版本:** + +.. c:function:: void i8_neg_grad_s(int8_t* Input, int8_t* output, int length, int core_mask) +.. c:function:: void i16_neg_grad_s(int16_t* Input, int16_t* output, int length, int core_mask) +.. c:function:: void i32_neg_grad_s(int32_t* Input, int32_t* output, int length, int core_mask) +.. c:function:: void hp_neg_grad_s(half* Input, half* output, int length, int core_mask) +.. c:function:: void fp_neg_grad_s(float* Input, float* output, int length, int core_mask) +.. c:function:: void dp_neg_grad_s(double* Input, double* output, int length, int core_mask) +.. c:function:: void c64_neg_grad_s(float* Input, float* output, int length, int core_mask) +.. c:function:: void c128_neg_grad_s(double* Input, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xB0000000; + int length = 1000; + int core_mask = 0xff; + fp_neg_grad_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_neg_grad_p(int8_t* Input, int8_t* output, int length) +.. c:function:: void i16_neg_grad_p(int16_t* Input, int16_t* output, int length) +.. c:function:: void i32_neg_grad_p(int32_t* Input, int32_t* output, int length) +.. c:function:: void hp_neg_grad_p(half* Input, half* output, int length) +.. c:function:: void fp_neg_grad_p(float* Input, float* output, int length) +.. c:function:: void dp_neg_grad_p(double* Input, double* output, int length) +.. c:function:: void c64_neg_grad_p(float* Input, float* output, int length) +.. c:function:: void c128_neg_grad_p(double* Input, double* output, int length) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + //MT7004示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; //input在L2空间 + float *output = (float *)0x10001000; + int length = 1000; + fp_neg_grad_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/nllloss.rst.txt b/master/html/_sources/functionlib/dsplib/nllloss.rst.txt new file mode 100644 index 0000000..e6b17b7 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/nllloss.rst.txt @@ -0,0 +1,111 @@ +NLLLoss +================= + + + +对输入的对数概率(log-probabilities)和目标标签计算负对数似然损失(Negative Log Likelihood Loss)。 + +该算子通常用于分类任务中,输入为已经取对数的概率值(例如 LogSoftmax 的输出)。 + +.. math:: + + \text{loss}_i = - \log p_{i, y_i} \cdot w_{y_i} + +其中: + +- :math:`p_{i, y_i}` 表示第 :math:`i` 个样本在真实类别 :math:`y_i` 上的预测概率 +- :math:`w_{y_i}` 表示对应类别的权重 + +根据 ``reduction_type`` 的不同,对 batch 维度的损失进行不同方式的归约。 + +输入: + - **log_probs** - 输入的对数概率数据地址,形状为 ``[batch_size, class_num]``。 + - **labels** - 真实标签索引地址,形状为 ``[batch_size]``。 + - **weight** - 类别权重数组地址,形状为 ``[class_num]``。 + - **batch_size** - batch 大小。 + - **class_num** - 类别数量。 + - **reduction_type** - 损失归约方式: + - ``0``:None,不做归约,逐样本输出 + - ``1``:Sum,对 batch 内损失求和 + - ``2``:Mean,对 batch 内损失按权重和求平均 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **loss** - 损失输出地址: + - 当 ``reduction_type = 0`` 时,输出长度为 ``batch_size`` + - 当 ``reduction_type = 1`` 或 ``2`` 时,仅使用 ``loss[0]`` + - **total_weight** - 所有样本权重之和地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp, int8 + - MT7004 支持 hp, fp + - 输入 ``log_probs`` 应已是对数概率值 + - ``labels`` 中的索引需满足 ``0 <= label < class_num`` + +**共享存储版本:** + +.. c:function:: void i8_nllloss_s(const int8_t* log_probs, const int* labels, const int8_t* weight, int32_t* loss, int32_t* total_weight, int batch_size, int class_num, int reduction_type, int core_mask) +.. c:function:: void hp_nllloss_s(const half* log_probs, const int* labels, const half* weight, half* loss, half* total_weight, int batch_size, int class_num, int reduction_type, int core_mask) +.. c:function:: void fp_nllloss_s(const float* log_probs, const int* labels, const float* weight, float* loss, float* total_weight, int batch_size, int class_num, int reduction_type, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16-17 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *log_probs = (float *)0xA0000000; // [batch_size, class_num] + int *labels = (int *)0xA0001000; // [batch_size] + float *weight = (float *)0xA0002000; // [class_num] + float *loss = (float *)0xC0000000; + float *total_w = (float *)0xC0001000; + int batch_size = 32; + int class_num = 1000; + int reduction_type = 2; // Mean + int core_mask = 0xff; + + fp_nllloss_s(log_probs, labels, weight, loss, total_w, + batch_size, class_num, reduction_type, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_nllloss_p(const int8_t* log_probs, const int* labels, const int8_t* weight, int32_t* loss, int32_t* total_weight, int batch_size, int class_num, int reduction_type) +.. c:function:: void hp_nllloss_p(const half* log_probs, const int* labels, const half* weight, half* loss, half* total_weight, int batch_size, int class_num, int reduction_type) +.. c:function:: void fp_nllloss_p(const float* log_probs, const int* labels, const float* weight, float* loss, float* total_weight, int batch_size, int class_num, int reduction_type) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15-16 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *log_probs = (float *)0x10810000; // L2空间 + int *labels = (int *)0x10820000; + float *weight = (float *)0x10830000; + float *loss = (float *)0x10840000; + float *total_w = (float *)0x10850000; + int batch_size = 32; + int class_num = 1000; + int reduction_type = 1; // Sum + + fp_nllloss_p(log_probs, labels, weight, loss, total_w, + batch_size, class_num, reduction_type); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/nlllossgrad.rst.txt b/master/html/_sources/functionlib/dsplib/nlllossgrad.rst.txt new file mode 100644 index 0000000..77834fe --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/nlllossgrad.rst.txt @@ -0,0 +1,120 @@ +NLLLossGrad +================= + + + +对 NLLLoss(Negative Log Likelihood Loss)算子的反向传播过程进行计算,得到输入 logits 的梯度。 + +该算子根据前向 NLLLoss 的 ``reduction_type``,对上游梯度进行对应方式的反向分发,仅在真实类别索引处产生非零梯度,其余位置梯度为 0。 + +.. math:: + + \frac{\partial L}{\partial x_{i,j}} = + \begin{cases} + - g_i \cdot w_{y_i}, & j = y_i,\ \text{reduction = none} \\ + - g \cdot w_{y_i}, & j = y_i,\ \text{reduction = sum} \\ + - g \cdot \dfrac{w_{y_i}}{\sum w}, & j = y_i,\ \text{reduction = mean} \\ + 0, & j \neq y_i + \end{cases} + +其中: + +- :math:`g_i` 表示逐样本损失梯度 +- :math:`g` 表示归约后的标量损失梯度 +- :math:`y_i` 表示第 :math:`i` 个样本的真实类别索引 +- :math:`w_{y_i}` 表示对应类别权重 + +输入: + - **logits** - 前向输入 logits 地址,形状为 ``[batch, class_num]`` (仅用于尺寸信息)。 + - **loss_grad** - 上游损失梯度地址: + - ``reduction_type = 0 (none)`` 时,形状为 ``[batch]`` + - ``reduction_type = 1 / 2 (sum / mean)`` 时,仅使用 ``loss_grad[0]`` + - **labels** - 真实标签索引地址,形状为 ``[batch]``。 + - **weight** - 类别权重数组地址,形状为 ``[class_num]``。 + - **total_weight** - 权重和地址(仅 ``reduction_type = mean`` 时使用)。 + - **batch** - batch 大小。 + - **class_num** - 类别数量。 + - **reduction_type** - 损失归约方式: + - ``0``:Sum + - ``1``:Mean + - ``2``:None + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **logits_grad** - logits 的梯度输出地址,形状为 ``[batch, class_num]``。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 仅支持 fp 类型 + - MT7004 支持 hp, fp 类型 + - 输出梯度在非真实类别位置恒为 0 + - ``labels`` 中索引需满足 ``0 <= label < class_num`` + +**共享存储版本:** + +.. c:function:: void hp_nlllossgrad_s(half* logits, half* loss_grad, int* labels, half* weight, half* total_weight, half* logits_grad, int batch, int class_num, int reduction_type, int core_mask) +.. c:function:: void fp_nlllossgrad_s(float* logits, float* loss_grad, int* labels, float* weight, float* total_weight, float* logits_grad, int batch, int class_num, int reduction_type, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 18-19 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *logits = (float *)0xA0000000; // [batch, class_num] + float *loss_grad = (float *)0xA0001000; // 上游梯度 + int *labels = (int *)0xA0002000; // [batch] + float *weight = (float *)0xA0003000; // [class_num] + float *total_weight = (float *)0xA0004000; + float *logits_grad = (float *)0xC0000000; + + int batch = 32; + int class_num = 1000; + int reduction_type = 1; // Mean + int core_mask = 0xff; + + fp_nlllossgrad_s(logits, loss_grad, labels, weight, total_weight, + logits_grad, batch, class_num, reduction_type, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void hp_nlllossgrad_p(half* logits, half* loss_grad, int* labels, half* weight, half* total_weight, half* logits_grad, int batch, int class_num, int reduction_type) +.. c:function:: void fp_nlllossgrad_p(float* logits, float* loss_grad, int* labels, float* weight, float* total_weight, float* logits_grad, int batch, int class_num, int reduction_type) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17-18 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *logits = (float *)0x10810000; // L2空间 + float *loss_grad = (float *)0x10820000; + int *labels = (int *)0x10830000; + float *weight = (float *)0x10840000; + float *total_weight = (float *)0x10850000; + float *logits_grad = (float *)0x10860000; + + int batch = 32; + int class_num = 1000; + int reduction_type = 2; // None + + fp_nlllossgrad_p(logits, loss_grad, labels, weight, total_weight, + logits_grad, batch, class_num, reduction_type); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/non_max_suppression.rst.txt b/master/html/_sources/functionlib/dsplib/non_max_suppression.rst.txt new file mode 100644 index 0000000..f7a84e0 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/non_max_suppression.rst.txt @@ -0,0 +1,151 @@ +NonMaxSuppression +=================== +对检测框进行非极大值抑制(Non-Maximum Suppression, NMS),用于在目标检测中选取得分较高且重叠度低的边界框,并去除那些重叠度较高的低得分框。其目的是减少冗余边界框,提高目标检测的准确性。 + +.. math:: + + \text{对于每个类别 } c, \text{在同一批次下保留置信度较高且 IoU 小于阈值的框:} + + \text{若 } \mathrm{IoU}(box_i, box_j) > \text{iou\_threshold} \Rightarrow \text{抑制 } box_j, + +.. math:: + IoU=\frac{intersec\_area}{(cand.get\_area() + box.get\_area() - intersec\_area)} + +输入: + - **NmsParam** - 输入参数结构体。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + + **结构体定义:** + + .. code-block:: c + :linenos: + + typedef struct { + float* boxes; // 形状为 (batch, num_boxes, 4) + float* score; // 形状为 (batch, class, num_boxes) + int* output; // 输出缓冲区 + float* candidates; // 候选框临时缓冲区 + int batch_num; // 批次数量 + int class_num; // 类别数量 + int box_num; // 每个类别的框数量 + int center_point_box; // 0: corner box, 1: center point box + bool simple_out; // true: 仅输出index, false: 输出 batch, class, index + int max_output_per_class; // 每个类别最大输出框数 + float iou_threshold; // IoU 阈值 + float score_threshold; // 分数阈值 + } NmsParamFp32; + typedef struct { + int8_t* boxes; // 形状为 (batch, num_boxes, 4) + int8_t* score; // 形状为 (batch, class, num_boxes) + int* output; // 输出缓冲区 + float* candidates; // 候选框临时缓冲区 + int batch_num; // 批次数量 + int class_num; // 类别数量 + int box_num; // 每个类别的框数量 + int center_point_box; // 0: corner box, 1: center point box + bool simple_out; // true: 仅输出index, false: 输出 batch, class, index + int max_output_per_class; // 每个类别最大输出框数 + float iou_threshold; // IoU 阈值 + int8_t score_threshold; // 分数阈值 + } NmsParamInt8; + typedef struct { + float16* boxes; + float16* score; + int* output; + float16* candidates; + uint64_t batch_num; + uint64_t class_num; + uint64_t box_num; + uint64_t center_point_box; + uint64_t simple_out; + uint64_t max_output_per_class; + float iou_threshold_bits; + float score_threshold_bits; + } NmsParamFp16; +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32, int8 + - MT7004 支持 fp16, fp32 + - MT7004中使用结构体作为参数传入。 + +**共享存储版本:** + +.. c:function:: void i8_non_max_suppression_s(NmsParamInt8* param, int core_mask) +.. c:function:: void fp_non_max_suppression_s(NmsParamFp32* param, int core_mask) +.. c:function:: void hp_non_max_suppression_s(NmsParamFp16* param, int core_mask) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 24 + + //MT7004示例 + #include + + int main(int argc, char* argv[]) { + float *boxes = (float *)0xA0000000; + float *score = (float *)0xA1000000; + int *output = (int *)0xA2000000; + NmsParamFp32* param = (NmsParamFp32*)0xA3000000; + + param->boxes = boxes; + param->score = score; + param->output = output; + param->candidates = candidates; + param->batch_num = 1; + param->class_num = 3; + param->box_num = 100; + param->center_point_box = 0; + param->simple_out = 1; + param->max_output_per_class = 50; + param->iou_threshold_bits = 0.5; + param->score_threshold_bits = 0.1; + + int core_mask = 0xff; + fp_non_max_suppression_s(param, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_non_max_suppression_p(NmsParamInt8* param) +.. c:function:: void fp_non_max_suppression_p(NmsParamFp32* param) +.. c:function:: void hp_non_max_suppression_p(NmsParamFp16* param) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 23 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *boxes = (float *)0x1000000; + float *score = (float *)0xA1000000; + int *output = (int *)0xA2000000; + NmsParamFp32* param = (NmsParamFp32*)0xA3000000; + + param->boxes = boxes; + param->score = score; + param->output = output; + param->candidates = candidates; + param->batch_num = 1; + param->class_num = 3; + param->box_num = 100; + param->center_point_box = 0; + param->simple_out = 1; + param->max_output_per_class = 50; + param->iou_threshold_bits = 0.5; + param->score_threshold_bits = 0.1; + + fp_non_max_suppression_s(param); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/nonzero.rst.txt b/master/html/_sources/functionlib/dsplib/nonzero.rst.txt new file mode 100644 index 0000000..c0b9bda --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/nonzero.rst.txt @@ -0,0 +1,119 @@ +NonZero +================= + + + +查找输入张量中 **非零元素的位置索引**。 +该算子按线性顺序遍历输入张量,对于满足“非零”判定条件的元素, +计算其在各维度上的索引,并顺序写入输出索引张量。 + +对于浮点类型,非零的判定标准为: + +.. math:: + + |x| > \epsilon + +其中 :math:`\epsilon` 为平台定义的极小阈值(如 ``FLOAT_EPS``), +用于避免数值误差导致的误判。 + +输入: + - **input** - 输入数据地址,按一维连续内存存储。 + - **dim_strides** - 各维度步长数组,长度为 ``input_rank``, + 用于将线性索引映射为多维索引。 + - **shape** - 输入张量的形状数组,长度为 ``input_rank`` (当前实现中主要用于描述维度信息)。 + - **input_rank** - 输入张量的维度数。 + - **length** - 输入张量的元素总数。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 非零元素索引输出地址, + 按 ``[non_zero_num, input_rank]`` 形式顺序存储。 + - **non_zero_num** - 非零元素个数指针,函数执行结束后保存最终非零元素数量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32``、``fp64``、``int8``、``int16``、``int32``、``cplx64``、``cplx128`` 类型 + - MT7004 支持 ``fp16``、``fp32``、``int16``、``int32``、``cplx64`` 类型 + - 输出索引按照输入线性遍历顺序排列 + - ``non_zero_num`` 在调用前需初始化为 0 + +**共享存储版本:** + +.. c:function:: void fp_nonzero_s(float* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length, int core_mask) +.. c:function:: void dp_nonzero_s(double* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length, int core_mask) +.. c:function:: void i8_nonzero_s(int8_t* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length, int core_mask) +.. c:function:: void i16_nonzero_s(int16_t* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length, int core_mask) +.. c:function:: void i32_nonzero_s(int32_t* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length, int core_mask) +.. c:function:: void c64_nonzero_s(cplx64* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length, int core_mask) +.. c:function:: void c128_nonzero_s(cplx128* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 18-20 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + int *output = (int *)0xC0000000; + int *non_zero_num = (int *)0xA1000000; + int *dim_strides = (int *)0xA2000000; + int *shape = (int *)0xA3000000; + + int input_rank = 3; + int length = 1024; + int core_mask = 0xff; + + non_zero_num[0] = 0; + + fp_nonzero_s(input, output, non_zero_num, + dim_strides, shape, input_rank, + length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_nonzero_p(float* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length) +.. c:function:: void dp_nonzero_p(double* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length) +.. c:function:: void i8_nonzero_p(int8_t* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length) +.. c:function:: void i16_nonzero_p(int16_t* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length) +.. c:function:: void i32_nonzero_p(int32_t* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length) +.. c:function:: void c64_nonzero_p(cplx64* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length) +.. c:function:: void c128_nonzero_p(cplx128* input, int* output, int* non_zero_num, int* dim_strides, int* shape, int input_rank, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17-18 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; // L2 空间 + int *output = (int *)0x10820000; + int *non_zero_num = (int *)0x10830000; + int *dim_strides = (int *)0x10840000; + int *shape = (int *)0x10850000; + + int input_rank = 3; + int length = 1024; + + non_zero_num[0] = 0; + + fp_nonzero_p(input, output, non_zero_num, + dim_strides, shape, input_rank, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/not_equal.rst.txt b/master/html/_sources/functionlib/dsplib/not_equal.rst.txt new file mode 100644 index 0000000..f21db52 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/not_equal.rst.txt @@ -0,0 +1,88 @@ +NotEqual +================= + +逐元素计算两个输入是否不相等 + +.. math:: + + output_i = \begin{cases} + \text{True}, & \text{if } Input0_i \neq Input1_i \\ + \text{False}, & \text{if } Input0_i = Input1_i + \end{cases} + +输入: + - **Input0** - 第一个输入数据地址。 + - **Input1** - 第二个输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_not_equal_s(int8_t* Input0, int8_t* Input1, bool* output, int length, int core_mask) +.. c:function:: void i16_not_equal_s(int16_t* Input0, int16_t* Input1, bool* output, int length, int core_mask) +.. c:function:: void i32_not_equal_s(int* Input0, int* Input1, bool* output, int length, int core_mask) +.. c:function:: void hp_not_equal_s(half* Input0, half* Input1, bool* output, int length, int core_mask) +.. c:function:: void fp_not_equal_s(float* Input0, float* Input1, bool* output, int length, int core_mask) +.. c:function:: void dp_not_equal_s(double* Input0, double* Input1, bool* output, int length, int core_mask) +.. c:function:: void c64_not_equal_s(float* Input0, float* Input1, bool* output, int length, int core_mask) +.. c:function:: void c128_not_equal_s(double* Input0, double* Input1, bool* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input在DDR空间 + float *input1 = (float *)0xB0000000; + bool *output = (bool *)0xC0000000; + int length = 1000; + int core_mask = 0xff; + fp_not_equal_s(input0, input1, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_not_equal_p(int8_t* Input0, int8_t* Input1, bool* output, int length) +.. c:function:: void i16_not_equal_p(int16_t* Input0, int16_t* Input1, bool* output, int length) +.. c:function:: void i32_not_equal_p(int32_t* Input0, int32_t* Input1, bool* output, int length) +.. c:function:: void hp_not_equal_p(half* Input0, half* Input1, bool* output, int length) +.. c:function:: void fp_not_equal_p(float* Input0, float* Input1, bool* output, int length) +.. c:function:: void dp_not_equal_p(double* Input0, double* Input1, bool* output, int length) +.. c:function:: void c64_not_equal_p(float* Input0, float* Input1, bool* output, int length) +.. c:function:: void c128_not_equal_p(double* Input0, double* Input1, bool* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //MT7004示例 + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; + float *input1 = (float *)0x10001000; + bool *output = (bool *)0xC0000000; + int length = 1000; + fp_not_equal_p(input0, input1, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/onehot.rst.txt b/master/html/_sources/functionlib/dsplib/onehot.rst.txt new file mode 100644 index 0000000..e9cd886 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/onehot.rst.txt @@ -0,0 +1,115 @@ +OneHot +================= + + 根据输入的索引值,在指定轴上生成独热编码(one-hot encoding)。 + + .. math:: + + output_{i_1, i_2, \dots, i_{axis}, \dots, i_n} = + \begin{cases} + on\_value, & \text{if } i_{axis} = indices_{j} \\ + off\_value, & \text{otherwise} + \end{cases} + + 其中,:math:`indices_{j}` 为输入索引张量的元素,索引 :math:`j` 与输出下标 + :math:`(i_1, \dots, i_{axis-1}, i_{axis+1}, \dots, i_n)` 一一对应。 + :math:`depth` 表示 one-hot 向量的长度,:math:`axis` 表示新维度插入的位置。 + + 当启用 ``support_neg_index`` 时,若 :math:`indices_{j} < 0`,则执行如下修正: + + .. math:: + + indices_{j} = indices_{j} + depth + + 输入: + - **on_off** - 包含**on_value**和**off_value**,表示独热编码中“热”和“冷”位置的值。 + - **axis** - 指定生成独热编码的轴。 + - **depth** - 独热编码的深度。 + - **support_neg_index** - 是否支持负索引。 + - **indices_shape_size** - 索引形状的维度大小。 + - **indices_shape** - 索引的形状数组地址。 + - **indices** - 索引值地址。 + - **core_mask** - 核心掩码,指定使用的计算核心。 + + 输出: + - **output** - 生成的独热编码结果地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + - MT6678 调用时将 axis/depth/support_neg_index/indices_shape_size打包通过 int_param 数组传入 + +**共享存储版本:** + +.. c:function:: void i8_onehot_s(int8_t *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, int8_t *output, int core_mask) +.. c:function:: void i16_onehot_s(int16_t *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, int16_t *output, int core_mask) +.. c:function:: void i32_onehot_s(int *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, int *output, int core_mask) +.. c:function:: void hp_onehot_s(half *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, half *output, int core_mask) +.. c:function:: void fp_onehot_s(float *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, float *output, int core_mask) +.. c:function:: void dp_onehot_s(double *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, double *output, int core_mask) +.. c:function:: void c64_onehot_s(float* on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, float *output, int core_mask) +.. c:function:: void c128_onehot_s(double* on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, double *output, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float on_off[2] = {1.0, 0.0}; + int axis=1, depth=4, indices_shape_size=2; + int support_neg_index=1; + int int_param[4] = {axis, depth, support_neg_index, indices_shape_size}; + int indices_shape[2] = {3, 3}; + int *indices = (int *)0xA0000000; // indices在DDR空间 + float *output = (float*)0xB0000000; // indices_shape[0] * depth * indices_shape[1] + int core_mask = 0xff; + fp_onehot_p(on_off, int_param, indices_shape, indices, output, core_mask); + // 7004调用方式: + // fp_onehot_p(on_off, axis, depth, support_neg_index, indices_shape_size, indices_shape, indices, output, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_onehot_p(int8_t *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, int8_t *output) +.. c:function:: void i16_onehot_p(int16_t *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, int16_t *output) +.. c:function:: void i32_onehot_p(int *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, int *output) +.. c:function:: void hp_onehot_p(half *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, half *output) +.. c:function:: void fp_onehot_p(float *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, float *output) +.. c:function:: void dp_onehot_p(double *on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, double *output) +.. c:function:: void c64_onehot_p(float* on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, float *output) +.. c:function:: void c128_onehot_p(double* on_off, int axis, int depth, int support_neg_index, int indices_shape_size, int *indices_shape, int *indices, double *output) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float on_off[2] = {1.0, 0.0}; + int axis=1, depth=4, indices_shape_size=2; + int support_neg_index=1; + int int_param[4] = {axis, depth, support_neg_index, indices_shape_size}; + int indices_shape[2] = {3, 3}; + int *indices = (int *)0x10000000; // indices在L2空间 + float *output = (float*)0x10001000; // indices_shape[0] * depth * indices_shape[1] + fp_onehot_p(on_off, int_param, indices_shape, indices, output); + // 7004调用方式: + // fp_onehot_p(on_value, off_value, axis, depth, support_neg_index, indices_shape_size, indices_shape, indices, output); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/ones_like.rst.txt b/master/html/_sources/functionlib/dsplib/ones_like.rst.txt new file mode 100644 index 0000000..25236e9 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/ones_like.rst.txt @@ -0,0 +1,87 @@ +OnesLike +================= + +对输出数组的所有元素填充数值 1。对于复数类型,实部填充 1.0,虚部填充 0.0。 + +.. math:: + + \text{对于实数类型:}\quad output_i = 1 + +.. math:: + + \text{对于复数类型:}\quad output_i = 1 + 0i + +输入: + - **output** - 待填充的目标内存地址。 + - **length** - 处理的元素个数(对于复数类型,指复数的个数)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 填充完成后的结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 int16, int32, fp16, fp32, cplx64 + - 复数类型(cplx64 / cplx128)在内存中以连续的 [实部, 虚部] 形式存储,OnesLike 将其设置为 [1.0, 0.0]。 + - 共享存储版本在内部使用 DMA 加速和“加倍拷贝法”以提高填充效率。 + +**共享存储版本:** + +.. c:function:: void i8_ones_like_s(int8_t* output, int length, int core_mask) +.. c:function:: void i16_ones_like_s(int16_t* output, int length, int core_mask) +.. c:function:: void i32_ones_like_s(int32_t* output, int length, int core_mask) +.. c:function:: void hp_ones_like_s(half* output, int length, int core_mask) +.. c:function:: void fp_ones_like_s(float* output, int length, int core_mask) +.. c:function:: void dp_ones_like_s(double* output, int length, int core_mask) +.. c:function:: void c64_ones_like_s(float* output, int length, int core_mask) +.. c:function:: void c128_ones_like_s(double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例(共享存储) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *output = (float *)0xB0000000; // 输出在 DDR 空间 + int length = 960001; + int core_mask = 0x0B; // 使用核心 0, 1, 3 + fp_ones_like_s(output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_ones_like_p(int8_t* output, int length) +.. c:function:: void i16_ones_like_p(int16_t* output, int length) +.. c:function:: void i32_ones_like_p(int32_t* output, int length) +.. c:function:: void hp_ones_like_p(half* output, int length) +.. c:function:: void fp_ones_like_p(float* output, int length) +.. c:function:: void dp_ones_like_p(double* output, int length) +.. c:function:: void c64_ones_like_p(float* output, int length) +.. c:function:: void c128_ones_like_p(double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 7 + + //MT7004 示例(私有存储) + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0x10002000; + int length = 1024; + fp_ones_like_p(output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/padfusion.rst.txt b/master/html/_sources/functionlib/dsplib/padfusion.rst.txt new file mode 100644 index 0000000..2260e03 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/padfusion.rst.txt @@ -0,0 +1,188 @@ +PadFusion +================= +通过指定的填充模式和大小对输入张量进行填充,支持常数填充、反射填充和对称填充三种模式。 + +输入: + - **input** - 输入张量地址。 + - **input_shape** - 输入张量形状数组(不超过4维,不足时从后往前补0)。 + - **output_shape** - 输出张量形状数组(不超过4维,不足时从后往前补0)。 + - **paddings** - 填充大小数组(长度为 (2 × 输入张量维数),不足时从后往前补0)。 + - **padding_mode** - 填充模式:Constant=0, Reflect=1, Symmetric=2。 + - **constant_value** - 常量填充时的填充值(仅Constant模式使用)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 输出数据地址。 + - **formated_input_shape** - 格式化后的输入形状数组地址。 + - **formated_output_shape** - 格式化后的输出形状数组地址。 + - **formated_paddings** - 格式化后的填充数组地址。 + - **in_strides** - 输入张量步长数组地址。 + - **out_strides** - 输出张量步长数组地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持的数据类型:fp16, fp32, int16, int32, cplx64 + +**填充模式说明:** + - ``kConstant = 0`` - 常数填充,使用指定的 constant_value 进行填充 + - ``kReflect = 1`` - 反射填充,使用张量边缘的值(不包含边界值)填充输入张量。例如,向 [1, 2, 3, 4] 的两边分别填充2个元素,结果为 [3, 2, 1, 2, 3, 4, 3, 2]。 + - ``kSymmetric = 2`` - 对称填充,使用张量边缘的值(包含边界值)填充输入张量。例如,向 [1, 2, 3, 4] 的两边分别填充2个元素,结果为 [2, 1, 1, 2, 3, 4, 4, 3]。 + +**共享存储版本:** + +.. c:function:: void i8_padfusion_s(long long* params, int core_mask) +.. c:function:: void i16_padfusion_s(long long* params, int core_mask) +.. c:function:: void i32_padfusion_s(long long* params, int core_mask) +.. c:function:: void fp_padfusion_s(long long* params, int core_mask) +.. c:function:: void dp_padfusion_s(long long* params, int core_mask) +.. c:function:: void c64_padfusion_s(long long* params, int core_mask) +.. c:function:: void c128_padfusion_s(long long* params, int core_mask) +.. c:function:: void hp_padfusion_s(long long* params, int core_mask) + +**参数数组结构:** + +.. code-block:: c + :linenos: + + long long params[12]; + params[0] = (long long)input; // 输入数据地址 + params[1] = (long long)output; // 输出数据地址 + params[2] = (long long)input_shape; // 输入形状数组 + params[3] = (long long)output_shape; // 输出形状数组 + params[4] = (long long)paddings; // 填充数组 + params[5] = (long long)padding_mode; // 填充模式 + params[6] = (long long)constant_value; // 常数填充值的*地址* + params[7] = (long long)formated_input_shape; // 格式化输入形状 + params[8] = (long long)formated_output_shape; // 格式化输出形状 + params[9] = (long long)formated_paddings; // 格式化填充数组 + params[10] = (long long)in_strides; // 输入步长数组 + params[11] = (long long)out_strides; // 输出步长数组 + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 49 + + // FT78NE 多核示例 + #include + #include + + int main(void) { + srand(time(0)); + + // 输入参数设置 + int input_shape[4] = {23, 31, 29, 28}; + int paddings[8] = {3, 2, 3, 2, 1, 2, 0, 2}; + int padding_mode = 2; // kSymmetric + double constant_value[2] = {1.0, 0.0}; + int core_mask = 0xff; + + // 内存分配(DDR空间) + double* input = (double*)0x81000000; + double* output = (double*)0x82000000; + int* formated_input_shape = (int*)0x84000000; + int* formated_output_shape = (int*)0x85000000; + int* formated_paddings = (int*)0x86000000; + int* in_strides = (int*)0x87000000; + int* out_strides = (int*)0x88000000; + int* output_shape = (int*)0x89000000; + + // 计算输出形状 + for (int i = 0; i < 4; ++i) { + output_shape[i] = input_shape[i] + paddings[i * 2] + paddings[i * 2 + 1]; + } + + // 初始化input + // ... 省略初始化input数据代码 ... + + // 准备参数数组 + long long params[12]; + params[0] = (long long)input; + params[1] = (long long)output; + params[2] = (long long)input_shape; + params[3] = (long long)output_shape; + params[4] = (long long)paddings; + params[5] = (long long)padding_mode; + params[6] = (long long)constant_value; + params[7] = (long long)formated_input_shape; + params[8] = (long long)formated_output_shape; + params[9] = (long long)formated_paddings; + params[10] = (long long)in_strides; + params[11] = (long long)out_strides; + + // 执行 PadFusion 操作 + c128_padfusion_s(params, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_padfusion_p(long long* params) +.. c:function:: void i16_padfusion_p(long long* params) +.. c:function:: void i32_padfusion_p(long long* params) +.. c:function:: void fp_padfusion_p(long long* params) +.. c:function:: void dp_padfusion_p(long long* params) +.. c:function:: void c64_padfusion_p(long long* params) +.. c:function:: void c128_padfusion_p(long long* params) +.. c:function:: void hp_padfusion_p(long long* params) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 47 + + // MT7004 单核示例 + #include + #include + + + int main(void) { + // 输入参数设置 + int input_shape[4] = {4, 8, 4, 8}; + int paddings[8] = {1, 1, 1, 1, 1, 1, 1, 1}; + int padding_mode = 0; // kConstant + float constant_value = 0.0f; + + // 内存分配(L2空间) + float* input = (float*)0x10000000; + float* output = (float*)0x10100000; + int* output_shape = (int*)0x10200000; + int* formated_input_shape = (int*)0x10300000; + int* formated_output_shape = (int*)0x10400000; + int* formated_paddings = (int*)0x10500000; + int* in_strides = (int*)0x10600000; + int* out_strides = (int*)0x10700000; + + // 计算输出形状 + for (int i = 0; i < 4; ++i) { + output_shape[i] = input_shape[i] + paddings[i * 2] + paddings[i * 2 + 1]; + } + + // 初始化input + // ... 省略初始化input数据代码 ... + + // 准备参数数组 + long long params[12]; + params[0] = (long long)input; + params[1] = (long long)output; + params[2] = (long long)input_shape; + params[3] = (long long)output_shape; + params[4] = (long long)paddings; + params[5] = (long long)padding_mode; + params[6] = (long long)&constant_value; + params[7] = (long long)formated_input_shape; + params[8] = (long long)formated_output_shape; + params[9] = (long long)formated_paddings; + params[10] = (long long)in_strides; + params[11] = (long long)out_strides; + + // 执行 PadFusion 操作 + fp_padfusion_p(params); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/pow_fusion.rst.txt b/master/html/_sources/functionlib/dsplib/pow_fusion.rst.txt new file mode 100644 index 0000000..8def4ed --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/pow_fusion.rst.txt @@ -0,0 +1,95 @@ +PowFusion +================= +先对输入按线性变换,然后逐元素计算幂运算,支持指数广播。 + +.. math:: + + \text{if broadcast: } output_i = (scale \times Input_i + shift)^{exponent_0} \\ + \text{else: } output_i = (scale \times Input_i + shift)^{exponent_i} + +输入: + - **Input** - 输入数据地址。 + - **exponent** - 指数数据地址;当 **broadcast** 为 True 时读取 `exponent[0]` 作为标量。 + - **length_in** - 输入长度。 + - **scale** - 线性变换比例系数。 + - **shift** - 线性变换偏移值。 + - **broadcast** - 是否将指数作为标量广播。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32, fp64, int8, int16, int32 + - MT7004 支持 fp32, fp16, int16, int32 + - 当指数为整数(即 fabs(exp - (int)exp) < 1e-6)时,内部使用优化的整数次幂实现以提高性能。 + - 对于负底数与非整数指数或其它非法值(如 0^negative),结果可能为未定义或产生 NaN/Inf,上层应负责必要的数值检查与处理。 + +**共享存储版本:** + +.. c:function:: void i8_pow_fusion_s(int8_t* Input, int8_t* exponent, int8_t* output, int length_in, int8_t scale, int8_t shift, bool broadcast, int core_mask) +.. c:function:: void i16_pow_fusion_s(int16_t* Input, int16_t* exponent, int16_t* output, int length_in, int scale, int shift, bool broadcast, int core_mask) +.. c:function:: void i32_pow_fusion_s(int32_t* Input, int32_t* exponent, int32_t* output, int length_in, int scale, int shift, bool broadcast, int core_mask) +.. c:function:: void hp_pow_fusion_s(half* Input, half* exponent, half* output, int length_in, float scale, float shift, bool broadcast, int core_mask) +.. c:function:: void fp_pow_fusion_s(float* Input, float* exponent, float* output, int length_in, float scale, float shift, bool broadcast, int core_mask) +.. c:function:: void dp_pow_fusion_s(double* Input, double* exponent, double* output, int length_in, double scale, double shift, bool broadcast, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + // FT78NE 共享存储示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + float *exponent = (float *)0xA0100000; // exponent 在 DDR 空间(或标量) + float *output = (float *)0xB0000000; + int length_in = 1024; + float scale = 1.0f; + float shift = 0.0f; + bool broadcast = false; + int core_mask = 0xff; + fp_pow_fusion_s(input, exponent, output, length_in, scale, shift, broadcast, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_pow_fusion_p(int8_t* Input, int8_t* exponent, int8_t* output, int length_in, int8_t scale, int8_t shift, bool broadcast) +.. c:function:: void i16_pow_fusion_p(int16_t* Input, int16_t* exponent, int16_t* output, int length_in, int scale, int shift, bool broadcast) +.. c:function:: void i32_pow_fusion_p(int32_t* Input, int32_t* exponent, int32_t* output, int length_in, int scale, int shift, bool broadcast) +.. c:function:: void hp_pow_fusion_p(half* Input, half* exponent, half* output, int length_in, float scale, float shift, bool broadcast) +.. c:function:: void fp_pow_fusion_p(float* Input, float* exponent, float* output, int length_in, float scale, float shift, bool broadcast) +.. c:function:: void dp_pow_fusion_p(double* Input, double* exponent, double* output, int length_in, double scale, double shift, bool broadcast) + + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + // MT7004 私有存储示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; + float *exponent = (float *)0x10001000; + float *output = (float *)0x10002000; + int length_in = 1024; + float scale = 1.0f; + float shift = 0.0f; + bool broadcast = false; + fp_pow_fusion_p(input, exponent, output, length_in, scale, shift, broadcast); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/power_grad.rst.txt b/master/html/_sources/functionlib/dsplib/power_grad.rst.txt new file mode 100644 index 0000000..dd76241 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/power_grad.rst.txt @@ -0,0 +1,79 @@ +PowerGrad +================= +计算 Power的梯度。 +对每个输入元素执行线性变换,并根据公式计算梯度结果。 + +.. math:: + + output_i = power \times scale \times Input1_i \times (scale \times Input2_i + shift)^{(power - 1)} + +输入: + - **Input1** - 第一个输入数据地址,对应前向传播的输入。 + - **Input2** - 第二个输入数据地址,对应前向传播的输入。 + - **length** - 输入长度。 + - **power** - 幂指数。 + - **scale** - 缩放系数。 + - **shift** - 偏移值。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 梯度计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp32, fp16 + - 若 **scale** 为 0,则输出恒为 0。 + +**共享存储版本:** + +.. c:function:: void fp_power_grad_s(float* Input1, float* Input2, float* output, int length, float power, float scale, float shift, int core_mask) +.. c:function:: void hp_power_grad_s(half* Input1, half* Input2, half* output, int length, float power, float scale, float shift, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input1 = (float *)0xA0000000; // input1 在 DDR 空间 + float *input2 = (float *)0xA1000000; // input2 在 DDR 空间 + float *output = (float *)0xB0000000; // 输出结果在 DDR 空间 + int length = 1024; + float power = 2.0f, scale = 0.5f, shift = 1.0f; + int core_mask = 0xff; + fp_power_grad_s(input1, input2, output, length, power, scale, shift, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_power_grad_p(float* Input1, float* Input2, float* output, int length, float power, float scale, float shift) +.. c:function:: void hp_power_grad_p(half* Input1, half* Input2, half* output, int length, float power, float scale, float shift) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //MT7004示例 + #include + + int main(int argc, char* argv[]) { + float *input1 = (float *)0x10000000; + float *input2 = (float *)0x10001000; + float *output = (float *)0x10002000; + int length = 1024; + float power = 2.0f, scale = 0.5f, shift = 1.0f; + fp_power_grad_p(input1, input2, output, length, power, scale, shift); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/prelufusion.rst.txt b/master/html/_sources/functionlib/dsplib/prelufusion.rst.txt new file mode 100644 index 0000000..91c7b2d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/prelufusion.rst.txt @@ -0,0 +1,84 @@ +PReLUFusion +================= + +对输入数组逐元素执行 PReLU 激活函数。对于每个元素,若小于等于 0,则乘以斜率参数 slope,否则保持原值。 + +.. math:: + + \text{dst}_i = + \begin{cases} + \text{src}_i \cdot \text{slope}, & \text{if } \text{src}_i \le 0 \\ + \text{src}_i, & \text{otherwise} + \end{cases} + +输入: + - **src_data** - 输入数据地址。 + - **slope** - PReLU 斜率参数。 + - **start** - 起始索引。 + - **end** - 结束索引。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst_data** - 输出数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, int8 + - MT7004 支持hp, fp + +**共享存储版本:** + +.. c:function:: void fp_prelufusion_s(float* src_data, float* dst_data, float slope, int start, int end, int core_mask) +.. c:function:: void hp_prelufusion_s(half* src_data, half* dst_data, half slope, int start, int end, int core_mask) +.. c:function:: void i8_prelufusion_s(int8_t* src_data, int8_t* dst_data, float slope, int start, int end, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + #include + + int main() { + float *src = (float *)0xA0000000; // 输入在DDR空间 + float *dst = (float *)0xC0000000; + float slope = 0.25; + int start = 0; + int end = 999; + int core_mask = 0xff; + + fp_prelufusion_s(src, dst, slope, start, end, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_prelufusion_p(float* src_data, float* dst_data, float slope, int start, int end) +.. c:function:: void hp_prelufusion_p(half* src_data, half* dst_data, half slope, int start, int end) +.. c:function:: void i8_prelufusion_p(int8_t* src_data, int8_t* dst_data, float slope, int start, int end) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main() { + float *src = (float *)0x10810000; // 输入在L2空间 + float *dst = (float *)0x10820000; + float slope = 0.25; + int start = 0; + int end = 999; + + fp_prelufusion_p(src, dst, slope, start, end); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/priorbox.rst.txt b/master/html/_sources/functionlib/dsplib/priorbox.rst.txt new file mode 100644 index 0000000..8207c0f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/priorbox.rst.txt @@ -0,0 +1,100 @@ +Priorbox +================= + +根据输入的特征图尺寸、步长和预设的候选框配置,逐元素计算先验框(Prior Boxes),通常用于目标检测模型。 + +该算子遍历特征图的每一个网格中心点,并围绕该中心点生成一系列具有不同尺寸和宽高比的候选框。生成的坐标 ``[xmin, ymin, xmax, ymax]`` 经过归一化处理。 + +输入: + - **fmap_h** - 当前特征图的高度。 + - **fmap_w** - 当前特征图的宽度。 + - **step_h** - 计算先验框在原图上坐标时的垂直缩放比例。 + - **step_w** - 计算先验框在原图上坐标时的水平缩放比例。 + - **offset** - 中心点偏移量,通常为0.5,表示取网格的正中心。 + - **min_sizes** - 定义基础先验框尺寸的浮点数数组。 + - **min_sizes_size** - ``min_sizes`` 数组的长度。 + - **max_sizes** - 可选的浮点数数组,用于生成额外的正方形先验框。 + - **max_sizes_size** - ``max_sizes`` 数组的长度。 + - **different_aspect_ratios** - 定义不同宽高比的浮点数数组。 + - **different_aspect_ratios_size** - ``different_aspect_ratios`` 数组的长度。 + - **core_mask** - 核掩码 (仅共享存储版本需要)。 + +输出: + - **output** - 用于存储最终生成的先验框坐标的数组。 + - **output_size** - 指向一个整数的指针,函数执行后,该整数将记录写入 ``output`` 数组的浮点数总数。 + +支持平台: + ``6678`` + ``7004`` + +.. note:: + - 6678支持fp32类型。 + - 7004支持fp16和fp32类型。 + +**共享存储版本:** + +.. c:function:: int fp_priorbox_s(int fmap_h, int fmap_w, float step_h, float step_w, float offset, float *min_sizes, int min_sizes_size, float *max_sizes, int max_sizes_size, float *different_aspect_ratios, int different_aspect_ratios_size, float *output, int *output_size, int core_mask) +.. c:function:: int hp_priorbox_s(int fmap_h, int fmap_w, half step_h, half step_w, half offset, half *min_sizes, int min_sizes_size, half *max_sizes, int max_sizes_size, half *different_aspect_ratios, int different_aspect_ratios_size, half *output, int *output_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 22 + + #include "priorbox.h" + + int main() { + // DDR memory pointers + float *min_sizes = (float *)0xA0000000; + float *max_sizes = (float *)0xA0001000; + float *ratios = (float *)0xA0002000; + float *output = (float *)0xB0000000; + int output_size = 0; + + // Parameters + int fmap_h = 19, fmap_w = 19; + float step_h = 16.0f, step_w = 16.0f; + float offset = 0.5f; + int min_sizes_size = 1; + int max_sizes_size = 1; + int ratios_size = 2; + int core_mask = 0xff; + + // (假设 min_sizes, max_sizes, ratios 已在DDR中填充数据) + + fp_priorbox_s(fmap_h, fmap_w, step_h, step_w, offset, min_sizes, min_sizes_size, max_sizes, max_sizes_size, ratios, ratios_size, output, &output_size, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: int fp_priorbox_p(int fmap_h, int fmap_w, float step_h, float step_w, float offset, float *min_sizes, int min_sizes_size, float *max_sizes, int max_sizes_size, float *different_aspect_ratios, int different_aspect_ratios_size, float *output, int *output_size) +.. c:function:: int hp_priorbox_p(int fmap_h, int fmap_w, half step_h, half step_w, half offset, half *min_sizes, int min_sizes_size, half *max_sizes, int max_sizes_size, half *different_aspect_ratios, int different_aspect_ratios_size, half *output, int *output_size) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + #include "priorbox.h" + + int main() { + // L2 memory arrays + float min_sizes_data[] = {30.0f}; + float max_sizes_data[] = {60.0f}; + float ratios_data[] = {2.0f, 0.5f}; + float output_data[4 * 19 * 19 * (1 + 1 + 2)]; // 足够大的空间 + int output_size = 0; + + // Parameters + int fmap_h = 19, fmap_w = 19; + float step_h = 16.0f, step_w = 16.0f; + float offset = 0.5f; + + fp_priorbox_p(fmap_h, fmap_w, step_h, step_w, offset, min_sizes_data, 1, max_sizes_data, 1, ratios_data, 2, output_data, &output_size); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/quantdtypecast.rst.txt b/master/html/_sources/functionlib/dsplib/quantdtypecast.rst.txt new file mode 100644 index 0000000..e1a072f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/quantdtypecast.rst.txt @@ -0,0 +1,113 @@ +QuantDTypeCast +================= + + +逐元素执行量化与反量化的数据类型转换操作,用于在浮点数与整型量化表示之间进行转换。 + +- **Quantize**:将 fp32/fp16 数据量化为 int8 +- **Dequantize**:将 int8 数据反量化为 fp32/fp16 + + +.. math:: + + \text{Quantize:}\quad + q_i = \operatorname{clip}\Bigl(\operatorname{round}\bigl(\frac{x_i}{\text{scale}} + \text{zp}\bigr),\ q_{\min},\ q_{\max}\Bigr) + +.. math:: + + \text{Dequantize:}\quad + x_i = (q_i - \text{zp}) \times \text{scale} + +其中: + +- :math:`x_i` 为输入浮点值 +- :math:`q_i` 为量化后的整型值 +- :math:`\text{scale}` 为量化比例因子 +- :math:`\text{zp}` 为零点(zero point) +- :math:`q_{\min}, q_{\max}` 为量化数据类型的取值范围(int8 对应 -128 到 127) + + +输入: + - **input** - 输入数据地址。 + - **scale** - 量化或反量化所使用的比例因子。 + - **zp** - 零点(zero point)。 + - **length** - 输入数据的元素个数。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出数据地址,其大小与 ``input`` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型: + - Quantize:fp32 → int8 + - Dequantize:int8 → fp32 + - MT7004 支持的数据类型: + - Quantize:fp32/fp16 → int8 + - Dequantize:int8 → fp32/fp16 + - 当输入值为 ``+∞`` 或 ``-∞`` 时,Quantize 结果分别饱和到 ``q_max`` 或 ``q_min`` + + +**共享存储版本:** + +.. c:function:: void fp_to_i8_quant_s(float* input, int8_t* output, float scale, int zp, int length, int core_mask) +.. c:function:: void i8_to_fp_dequant_s(int8_t* input, float* output, float scale, int zp, int length, int core_mask) +.. c:function:: void hp_to_i8_quant_s(half* input, int8_t* output, half scale, int zp, int length, int core_mask) +.. c:function:: void i8_to_hp_dequant_s(int8_t* input, half* output, half scale, int zp, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + // FT78NE 多核示例(Quantize) + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + int8_t *output = (int8_t *)0xB0000000; // output 在 DDR 空间 + + float scale = 0.05f; + int zp = 0; + int length = 4096; + int core_mask = 0xff; + + fp_to_i8_quant_s(input, output, scale, zp, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_to_i8_quant_p(float* input, int8_t* output, float scale, int zp, int length) +.. c:function:: void i8_to_fp_dequant_p(int8_t* input, float* output, float scale, int zp, int length) +.. c:function:: void hp_to_i8_quant_p(half* input, int8_t* output, half scale, int zp, int length) +.. c:function:: void i8_to_hp_dequant_p(int8_t* input, half* output, half scale, int zp, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + // MT7004 单核示例(Dequantize) + #include + #include + + int main(int argc, char* argv[]) { + int8_t *input = (int8_t *)0x10000000; // input 在 L2 空间 + float *output = (float *)0x11000000; // output 在 L2 空间 + + float scale = 0.05f; + int zp = 0; + int length = 1024; + + i8_to_fp_dequant_p(input, output, scale, zp, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/random_normal.rst.txt b/master/html/_sources/functionlib/dsplib/random_normal.rst.txt new file mode 100644 index 0000000..7ffa3f2 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/random_normal.rst.txt @@ -0,0 +1,79 @@ +RandomNormal +================= + +生成服从指定均值和标准差的正态分布随机数序列。 + +.. math:: + + output_i \sim \mathcal{N}(\text{mean}, \text{scale}^2) + +输入: + - **length** - 输出数据长度。 + - **mean** - 正态分布的均值。 + - **scale** - 正态分布的标准差。 + - **seed** - 随机数种子,用于控制生成序列的随机性。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 生成的随机数结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp32, fp16 + - 输出服从 :math:`\mathcal{N}(\text{mean}, \text{scale}^2)` 分布。 + - 相同的随机种子会生成确定性一致的输出序列。 + + +**共享存储版本:** + +.. c:function:: void fp_random_normal_s(float* output, int length, float mean, float scale, unsigned int seed, int core_mask) +.. c:function:: void hp_random_normal_s(half* output, int length, float mean, float scale, unsigned int seed, int core_mask) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 示例 + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0xA0000000; // output 在 DDR 空间 + int length = 1024; + float mean = 0.0f; + float scale = 1.0f; + unsigned int seed = 1234; + int core_mask = 0xff; + fp_random_normal_s(output, length, mean, scale, seed, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_random_normal_p(float* output, int length, float mean, float scale, unsigned int seed) +.. c:function:: void hp_random_normal_p(half* output, int length, float mean, float scale, unsigned int seed) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // MT7004 示例 + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0x10010000; + int length = 1024; + float mean = 0.0f; + float scale = 1.0f; + unsigned int seed = 1234; + fp_random_normal_p(output, length, mean, scale, seed); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/random_standard_normal.rst.txt b/master/html/_sources/functionlib/dsplib/random_standard_normal.rst.txt new file mode 100644 index 0000000..7f2bf77 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/random_standard_normal.rst.txt @@ -0,0 +1,72 @@ +RandomStandardNormal +====================== + +生成服从标准正态分布 :math:`\mathcal{N}(0, 1)` 的随机数序列。 + +.. math:: + + output_i \sim \mathcal{N}(0, 1) + +输入: + - **length** - 输出数据长度。 + - **seed** - 随机数种子,用于控制生成序列的随机性。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 生成的标准正态分布随机数地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp32, fp16 + - 相同的种子可生成相同的随机数序列(确定性行为)。 + - 随机数生成使用 Box–Muller 算法实现。 + +**共享存储版本:** + +.. c:function:: void fp_random_standard_normal_s(float* output, int length, unsigned int seed, int core_mask) +.. c:function:: void hp_random_standard_normal_s(half* output, int length, unsigned int seed, int core_mask) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + // FT78NE 示例 + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0xA0000000; // output 在 DDR 空间 + int length = 1024; + unsigned int seed = 1234; + int core_mask = 0xff; + fp_random_standard_normal_s(output, length, seed, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_random_standard_normal_p(float* output, int length, unsigned int seed) +.. c:function:: void hp_random_standard_normal_p(half* output, int length, unsigned int seed) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + // MT7004 示例 + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0x10010000; + int length = 1024; + unsigned int seed = 1234; + fp_random_standard_normal_p(output, length, seed); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/range.rst.txt b/master/html/_sources/functionlib/dsplib/range.rst.txt index f5e7db7..75c0d94 100644 --- a/master/html/_sources/functionlib/dsplib/range.rst.txt +++ b/master/html/_sources/functionlib/dsplib/range.rst.txt @@ -36,7 +36,7 @@ Range .. code-block:: c :linenos: - :emphasize-lines: 11 + :emphasize-lines: 10 #include #include @@ -63,7 +63,7 @@ Range .. code-block:: c :linenos: - :emphasize-lines: 10 + :emphasize-lines: 9 #include #include diff --git a/master/html/_sources/functionlib/dsplib/rank.rst.txt b/master/html/_sources/functionlib/dsplib/rank.rst.txt new file mode 100644 index 0000000..3a6005b --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/rank.rst.txt @@ -0,0 +1,64 @@ +Rank +================= + + 获取输入张量的秩(维数),并将该值写入输出地址。 + + 输入: + - **output** - 输出数据的地址,用于存储秩的结果。 + - **n** - 输入张量的秩(维数)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 存储秩数值的地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - 由于该算子对于不同数据类型的具体实现一致,因此统一使用 ``rank_s`` 和 ``rank_p`` 命名,不再区分数据类型前缀(如 ``fp_``, ``i8_`` 等)。 + - 支持的数据类型包括:int8, int16, int32, fp32, fp64, cplx64, cplx128。 + +**共享存储版本:** + +.. c:function:: void rank_s(int* output, int n, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + #include + + int main(int argc, char* argv[]) { + int n = 4; // 假设张量的秩为4 + int *output = (int *)0xA0000000; + int core_mask = 0xff; + + rank_s(output, n, core_mask); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void rank_p(int* output, int n) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 7 + + #include + + int main(int argc, char* argv[]) { + int n = 3; // 假设张量的秩为3 + int *output = (int *)0x10810000; + + rank_p(output, n); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/real_div.rst.txt b/master/html/_sources/functionlib/dsplib/real_div.rst.txt new file mode 100644 index 0000000..c0bf170 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/real_div.rst.txt @@ -0,0 +1,85 @@ +RealDiv +================= +逐元素计算两个输入的除法。 + +.. math:: + + output_i = \frac{input0_i}{input1_i} + +输入: + - **input0** - 被除数输入数据地址。 + - **input1** - 除数输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 若除数元素为 0,则输出结果为无穷大或未定义值,需由上层逻辑处理。 + +**共享存储版本:** + +.. c:function:: void i8_real_div_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_real_div_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_real_div_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_real_div_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_real_div_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_real_div_s(double* input0, double* input1, double* output, int length, int core_mask) +.. c:function:: void c64_real_div_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void c128_real_div_s(double* input0, double* input1, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // input0 在 DDR 空间 + float *input1 = (float *)0xA1000000; // input1 在 DDR 空间 + float *output = (float *)0xB0000000; // 输出结果在 DDR 空间 + int length = 1024; + int core_mask = 0xff; + fp_real_div_s(input0, input1, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_real_div_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_real_div_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_real_div_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_real_div_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_real_div_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_real_div_p(double* input0, double* input1, double* output, int length) +.. c:function:: void c64_real_div_p(float* input0, float* input1, float* output, int length) +.. c:function:: void c128_real_div_p(double* input0, double* input1, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //MT7004 示例 + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; + float *input1 = (float *)0x10001000; + float *output = (float *)0x10002000; + int length = 1024; + fp_real_div_p(input0, input1, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/reciprocal.rst.txt b/master/html/_sources/functionlib/dsplib/reciprocal.rst.txt new file mode 100644 index 0000000..d1cd809 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/reciprocal.rst.txt @@ -0,0 +1,82 @@ +Reciprocal +================= +逐元素计算输入数据的倒数。 + +.. math:: + + output_i = \frac{1}{Input_i} + +输入: + - **Input** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 当输入为 0 时,输出结果为无穷大或未定义值,需由上层逻辑处理。 + +**共享存储版本:** + +.. c:function:: void i8_reciprocal_s(int8_t* Input, float* output, int length, int core_mask) +.. c:function:: void i16_reciprocal_s(int16_t* Input, float* output, int length, int core_mask) +.. c:function:: void i32_reciprocal_s(int32_t* Input, float* output, int length, int core_mask) +.. c:function:: void hp_reciprocal_s(half* Input, half* output, int length, int core_mask) +.. c:function:: void fp_reciprocal_s(float* Input, float* output, int length, int core_mask) +.. c:function:: void dp_reciprocal_s(double* Input, double* output, int length, int core_mask) +.. c:function:: void c64_reciprocal_s(float* Input, float* output, int length, int core_mask) +.. c:function:: void c128_reciprocal_s(double* Input, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xB0000000; + int length = 1000; + int core_mask = 0xff; + fp_reciprocal_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_reciprocal_p(int8_t* Input, float* output, int length) +.. c:function:: void i16_reciprocal_p(int16_t* Input, float* output, int length) +.. c:function:: void i32_reciprocal_p(int32_t* Input, float* output, int length) +.. c:function:: void hp_reciprocal_p(half* Input, half* output, int length) +.. c:function:: void fp_reciprocal_p(float* Input, float* output, int length) +.. c:function:: void dp_reciprocal_p(double* Input, double* output, int length) +.. c:function:: void c64_reciprocal_p(float* Input, float* output, int length) +.. c:function:: void c128_reciprocal_p(double* Input, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + //MT7004示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; + float *output = (float *)0x10001000; + int length = 1000; + fp_reciprocal_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/reducescatter.rst.txt b/master/html/_sources/functionlib/dsplib/reducescatter.rst.txt new file mode 100644 index 0000000..16d2287 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/reducescatter.rst.txt @@ -0,0 +1,86 @@ +ReduceScatter +================= + + + +对输入数组进行指定类型的分布式归约操作(Reduce),并将结果分散到各输出位置。支持 ReduceSum、ReduceMean、ReduceMax 和 ReduceMin。 + +输入: + - **input_data** - 输入数据地址。 + - **data_size** - 数据长度。 + - **reduce_type** - 归约类型: + - 0: ReduceSum + - 1: ReduceMean + - 2: ReduceMax + - 3: ReduceMin + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output_data** - 输出数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, dp, int8, int16, int32 + - MT7004 支持hp, fp, i16, i32 + +**共享存储版本:** + +.. c:function:: void fp_reducescatter_s(float* input_data, float* output_data, int data_size, int reduce_type, int core_mask) +.. c:function:: void hp_reducescatter_s(half* input_data, half* output_data, int data_size, int reduce_type, int core_mask) +.. c:function:: void dp_reducescatter_s(double* input_data, double* output_data, int data_size, int reduce_type, int core_mask) +.. c:function:: void i8_reducescatter_s(int8_t* input_data, int8_t* output_data, int data_size, int reduce_type, int core_mask) +.. c:function:: void i16_reducescatter_s(int16_t* input_data, int16_t* output_data, int data_size, int reduce_type, int core_mask) +.. c:function:: void i32_reducescatter_s(int* input_data, int* output_data, int data_size, int reduce_type, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main() { + float *input = (float *)0xA0000000; // 输入在DDR空间 + float *output = (float *)0xC0000000; + int data_size = 1024; + int reduce_type = 0; // ReduceSum = 4; + int core_mask = 0xff; + + fp_reducescatter_s(input, output, data_size, reduce_type, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_reducescatter_p(float* input_data, float* output_data, int data_size, int reduce_type) +.. c:function:: void hp_reducescatter_p(half* input_data, half* output_data, int data_size, int reduce_type) +.. c:function:: void dp_reducescatter_p(double* input_data, double* output_data, int data_size, int reduce_type) +.. c:function:: void i8_reducescatter_p(int8_t* input_data, int8_t* output_data, int data_size, int reduce_type) +.. c:function:: void i16_reducescatter_p(int16_t* input_data, int16_t* output_data, int data_size, int reduce_type) +.. c:function:: void i32_reducescatter_p(int* input_data, int* output_data, int data_size, int reduce_type) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main() { + float *input = (float *)0x10810000; // 输入在L2空间 + float *output = (float *)0x10820000; + int data_size = 1024; + int reduce_type = 0; // ReduceSum + = 4; + + fp_reducescatter_p(input, output, data_size, reduce_type); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/reshape.rst.txt b/master/html/_sources/functionlib/dsplib/reshape.rst.txt new file mode 100644 index 0000000..c70cd2f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/reshape.rst.txt @@ -0,0 +1,84 @@ +Reshape +================= + +对输入张量进行重塑操作。在底层实现上,由于张量形状仅由参数定义,该算子执行从输入地址到输出地址的连续数据拷贝。 + +.. math:: + + output_i = input_i + +输入: + - **input** - 输入数据起始地址。 + - **length** - 需拷贝的元素个数。对于复数类型,length 表示复数的个数。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 输出数据起始地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, c64, c128 + - MT7004 支持 fp16, fp32, int16, int32, c64 + - 对于复数类型(c64 / c128),算子会自动处理实部和虚部的连续拷贝,其迁移字节数是普通类型的两倍。 + +**共享存储版本:** + +.. c:function:: void i8_reshape_s(int8_t* input, int8_t* output, int length, int core_mask) +.. c:function:: void i16_reshape_s(int16_t* input, int16_t* output, int length, int core_mask) +.. c:function:: void i32_reshape_s(int* input, int* output, int length, int core_mask) +.. c:function:: void hp_reshape_s(half* input, half* output, int length, int core_mask) +.. c:function:: void fp_reshape_s(float* input, float* output, int length, int core_mask) +.. c:function:: void dp_reshape_s(double* input, double* output, int length, int core_mask) +.. c:function:: void c64_reshape_s(float* input, float* output, int length, int core_mask) +.. c:function:: void c128_reshape_s(double* input, double* output, int length, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例(共享存储,多核并行拷贝) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // DDR 空间 + float *output = (float *)0xB0000000; // DDR 空间 + int length = 960001; + int core_mask = 0x0B; // 使用逻辑核 0, 1, 2 + fp_reshape_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_reshape_p(int8_t* input, int8_t* output, int length) +.. c:function:: void i16_reshape_p(int16_t* input, int16_t* output, int length) +.. c:function:: void i32_reshape_p(int32_t* input, int32_t* output, int length) +.. c:function:: void hp_reshape_p(half* input, half* output, int length) +.. c:function:: void fp_reshape_p(float* input, float* output, int length) +.. c:function:: void dp_reshape_p(double* input, double* output, int length) +.. c:function:: void c64_reshape_p(float* input, float* output, int length) +.. c:function:: void c128_reshape_p(double* input, double* output, int length) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 8 + + //MT7004 示例(私有存储,单核拷贝) + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; + float *output = (float *)0x10820000; + int length = 1024; + fp_reshape_p(input, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/resizegrad.rst.txt b/master/html/_sources/functionlib/dsplib/resizegrad.rst.txt new file mode 100644 index 0000000..747df01 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/resizegrad.rst.txt @@ -0,0 +1,205 @@ +Resizegrad +================= + +计算 Resize 操作的梯度。该算子包含两种实现:最近邻插值梯度(ResizeNearestNeighborGrad)和双线性插值梯度(ResizeBiLinearGrad)。用于将上游梯度从输出尺寸反向传播到输入尺寸。 + +最近邻插值梯度(ResizeNearestNeighborGrad) + + +对于最近邻插值,输入梯度通过最近邻映射反向传播到输出梯度。 + +.. math:: + + out_y = + \begin{cases} + \mathrm{round}(in_y \cdot height\_scale), & \text{if } align\_corners \\ + \lfloor in_y \cdot height\_scale \rfloor, & \text{otherwise} + \end{cases} + +.. math:: + + out_x = + \begin{cases} + \mathrm{round}(in_x \cdot width\_scale), & \text{if } align\_corners \\ + \lfloor in_x \cdot width\_scale \rfloor, & \text{otherwise} + \end{cases} + +.. math:: + + out\_addr[out\_offset] \mathrel{+}= in\_addr[in\_offset] + +其中 `in_addr` 是上游梯度(输出尺寸),`out_addr` 是输入梯度(输入尺寸)。注意使用累加操作(+=),因为多个输入位置可能映射到同一个输出位置。 + +双线性插值梯度(ResizeBiLinearGrad) + + +对于双线性插值,输入梯度通过双线性插值的反向传播分配到4个相邻的输出位置。 + +.. math:: + + in_y = h \cdot height\_scale + +.. math:: + + top_y = \max(\lfloor in_y \rfloor, 0) + +.. math:: + + bottom_y = \min(\lceil in_y \rceil, out\_height - 1) + +.. math:: + + y_{\mathrm{lerp}} = in_y - \lfloor in_y \rfloor + +.. math:: + + inverse\_y\_lerp = 1.0 - y_{\mathrm{lerp}} + +对于 x 维度有类似的公式: + +.. math:: + + in_x = w \cdot width\_scale + +.. math:: + + left_x = \max(\lfloor in_x \rfloor, 0) + +.. math:: + + right_x = \min(\lceil in_x \rceil, out\_width - 1) + +.. math:: + + x_{\mathrm{lerp}} = in_x - \lfloor in_x \rfloor + +.. math:: + + inverse\_x\_lerp = 1.0 - x_{\mathrm{lerp}} + +输入梯度按权重分配到4个相邻位置: + +.. math:: + + out\_addr[top_y, left_x] \mathrel{+}= in\_addr[h, w] \cdot + (inverse\_y\_lerp \cdot inverse\_x\_lerp) + +.. math:: + + out\_addr[top_y, right_x] \mathrel{+}= in\_addr[h, w] \cdot + (inverse\_y\_lerp \cdot x_{\mathrm{lerp}}) + +.. math:: + + out\_addr[bottom_y, left_x] \mathrel{+}= in\_addr[h, w] \cdot + (y_{\mathrm{lerp}} \cdot inverse\_x\_lerp) + +.. math:: + + out\_addr[bottom_y, right_x] \mathrel{+}= in\_addr[h, w] \cdot + (y_{\mathrm{lerp}} \cdot x_{\mathrm{lerp}}) + +输入: + - **in_addr** - 指向上游梯度数据的指针(输出尺寸的梯度)。 + - **out_addr** - 指向输出梯度数据的指针(输入尺寸的梯度),需要初始化为0。 + - **batch_size** - 批次大小。 + - **channel** - 通道数。 + - **format** - 数据格式,0 表示 NHWC,1 表示 NCHW。 + - **align_corners** - 是否对齐角点标志,0 表示不对齐,1 表示对齐。 + - **in_height** - 输入高度。 + - **in_width** - 输入宽度。 + - **out_height** - 输出高度。 + - **out_width** - 输出宽度。 + - **height_scale** - 高度缩放因子,通常为 (out_height - 1) / (in_height - 1) 或 out_height / in_height。 + - **width_scale** - 宽度缩放因子,通常为 (out_width - 1) / (in_width - 1) 或 out_width / in_width。 + +输出: + - **out_addr** - 计算后的输入梯度(输入尺寸的梯度)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp16, fp32 + + +**共享存储版本:** + +.. c:function:: void fp_resizenearestneighborgrad_s(float* in_addr, float* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale, int core_mask) +.. c:function:: void hp_resizenearestneighborgrad_s(half* in_addr, half* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale, int core_mask) +.. c:function:: void fp_resizebilineargrad_s(float* in_addr, float* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale, int core_mask) +.. c:function:: void hp_resizebilineargrad_s(half* in_addr, half* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale, int core_mask) + +**C调用示例(最近邻插值梯度):** + +.. code-block:: c + :linenos: + :emphasize-lines: 22-24 + + #include + #include + + int main(int argc, char* argv[]) { + float *in_addr = (float *)0x10010000; + float *out_addr = (float *)0x10020000; + + int batch_size = 4; + int channel = 3; + int format = 1; + int align_corners = 1; + int in_height = 4, in_width = 4; + int out_height = 6, out_width = 6; + int core_mask = 0xff; + + float height_scale = (float)(out_height - 1) / (in_height - 1); + float width_scale = (float)(out_width - 1) / (in_width - 1); + + int output_size = in_height * in_width * channel * batch_size; + memset(out_addr, 0, output_size * sizeof(float)); + + fp_resizenearestneighborgrad_s(in_addr, out_addr, batch_size, channel, + format, align_corners, in_height, in_width, + out_height, out_width, height_scale, width_scale, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_resizenearestneighborgrad_p(float* in_addr, float* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale) +.. c:function:: void hp_resizenearestneighborgrad_p(half* in_addr, half* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale) +.. c:function:: void fp_resizebilineargrad_p(float* in_addr, float* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale) +.. c:function:: void hp_resizebilineargrad_p(half* in_addr, half* out_addr, int batch_size, int channel, int format, int align_corners, int in_height, int in_width, int out_height, int out_width, float height_scale, float width_scale) + +**C调用示例(双线性插值梯度):** + +.. code-block:: c + :linenos: + :emphasize-lines: 21-23 + + #include + #include + + int main(int argc, char* argv[]) { + float *in_addr = (float *)0x10010000; + float *out_addr = (float *)0x10020000; + + int batch_size = 4; + int channel = 3; + int format = 0; + int align_corners = 0; + int in_height = 4, in_width = 4; + int out_height = 6, out_width = 6; + + float height_scale = (float)out_height / in_height; + float width_scale = (float)out_width / in_width; + + int output_size = in_height * in_width * channel * batch_size; + memset(out_addr, 0, output_size * sizeof(float)); + + fp_resizebilineargrad_p(in_addr, out_addr, batch_size, channel, + format, align_corners, in_height, in_width, + out_height, out_width, height_scale, width_scale); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/rfft.rst.txt b/master/html/_sources/functionlib/dsplib/rfft.rst.txt new file mode 100644 index 0000000..40688c6 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/rfft.rst.txt @@ -0,0 +1,158 @@ +RFFT +========================= + +对输入复数序列执行一维快速傅里叶变换(FFT)或逆变换(IFFT)。 +内部基于分治 FFT 算法, +并使用预计算旋转因子(twiddle factor)以提升性能。 +傅里叶变换,可以对参数进行调整,以实现FFT/IFFT/RFFT/IRFFT。 + +数学定义如下: + +.. math:: + + X(k) = \sum_{n=0}^{N-1} x(n)\,e^{-j 2\pi kn / N} \quad (\text{Forward FFT}) + +.. math:: + + x(n) = \sum_{k=0}^{N-1} X(k)\,e^{j 2\pi kn / N} \quad (\text{Inverse FFT}) + +其中 :math:`N` 为 FFT 点数。 + +输入: + - **input** - 输入复数数据地址。 + - **fft_size** - FFT 点数。 + - **dir** - 变换方向: + - ``FFT_FORWARD``:正向 FFT + - ``FFT_INVERSE``:反向 FFT + - **scratch_ptr** - 临时缓冲区地址,用于存放旋转因子及中间计算结果。 + - **twiddle** - 旋转因子地址(仅共享存储版本使用)。 + - **fft_size1** - 第一阶段 FFT 点数(仅共享存储版本使用)。 + - **fft_size2** - 第二阶段 FFT 点数(仅共享存储版本使用)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 输出复数序列地址。 + +内部核心计算公式如下: +对于FFT,它计算以下表达式: + +.. math:: + X[\omega_1, \dots, \omega_d] = + \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{-j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}}, + +其中, :math:`d` = `signal_ndim` 是信号的维度,:math:`N_i` 则是信号第 :math:`i` 个维度的大小。 + +对于IFFT,它计算以下表达式: + +.. math:: + X[\omega_1, \dots, \omega_d] = + \frac{1}{\prod_{i=1}^d N_i} \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{\ j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}}, + +其中, :math:`d` = `signal_ndim` 是信号的维度,:math:`N_i` 则是信号第 :math:`i` 维的大小。 + +.. note:: + - FFT/IFFT要求complex64或complex128类型的输入,返回complex64或complex128类型的输出。 + - RFFT要求bool, uint8, int8, int16, int32, int64, float32或float64类型的输入, + 返回complex64或complex128类型的输出。 + - IRFFT要求complex64或complex128类型的输入,返回float32或float64类型的输出。 + +参数: + - **signal_ndim** (int) - 表示每个信号中的维数,控制着傅里叶变换的维数,其值只能为1、2或3。 + - **inverse** (bool) - 表示该操作是否为逆变换,用以选择FFT 和 RFFT 或 IFFT 和 IRFFT。 + + - 如果为 ``True`` ,则为IFFT 和 IRFFT。 + - 如果为 ``False`` ,FFT 和 RFFT。 + + - **real** (bool) - 表示该操作是否为实变换,与 `inverse` 共同决定具体的变换模式: + + - `inverse` 为 ``False`` , `real` 为 ``False`` :对应FFT模式。 + - `inverse` 为 ``True`` , `real` 为 ``False`` :对应IFFT模式。 + - `inverse` 为 ``False`` , `real` 为 ``True`` :对应RFFT模式。 + - `inverse` 为 ``True`` , `real` 为 ``True`` :对应IRFFT模式。 + + - **norm** (str,可选) - 表示该操作的规范化方式,可选值:[ ``"backward"`` , ``"forward"`` , ``"ortho"`` ]。默认值: ``"backward"`` 。 + + - "backward",正向变换不缩放,逆变换按 :math:`1/n` 缩放,其中 `n` 表示输入 `x` 的元素数量。。 + - "ortho",正向变换与逆变换均按 :math:`1/\sqrt n` 缩放。 + - "forward",正向变换按 :math:`1/n` 缩放,逆变换不缩放。 + + - **onesided** (bool,可选) - 控制输入是否减半以避免冗余。默认值: ``True`` 。 + - **signal_sizes** (tuple,可选) - 原始信号的大小(RFFT变换之前的信号,不包含batch这一维),只有在IRFFT模式下和设置 `onesided` 为True时需要该参数,需要满足 + 以下条件。默认值: ``()`` 。 + + - `signal_sizes` 的长度等于IRFFT的 `signal_ndim` : :math:`len(signal\_sizes)=signal\_ndim` 。 + - `signal_sizes` 的最后一个维度除以2等于IRFFT输入的最后一个维度: :math:`signal\_size[-1]/2+1=x.shape[-1]` 。 + - 除了最后一个维度外, `signal_sizes` 的维度与输入shape完全相同: :math:`signal\_sizes[:-1]=x.shape[:-1]` 。 + +异常: + - **TypeError** - 如果FFT/IFFT/IRFF的输入类型不是以下类型之一:complex64、complex128。 + - **TypeError** - 如果输入的类型不是Tensor。 + - **ValueError** - 如果输入 `x` 的维度小于 `signal_ndim` 。 + - **ValueError** - 如果 `signal_ndim` 大于3或小于1。 + - **ValueError** - 如果 `norm` 取值不是"backward"、"forward"或"ortho"。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +类型支持: + - FT78NE:``cplx64``、``cplx128`` + - MT7004:``cplx64`` + +**共享存储版本:** + +.. c:function:: void c64_rfft_s(fftw_complex* input, fftw_complex* output, fftw_complex* scratch_ptr, fftw_complex* twiddle, int fft_size1, int fft_size2, char dir, int core_mask) +.. c:function:: void c128_rfft_s(fftw_complex* input, fftw_complex* output, fftw_complex* scratch_ptr, fftw_complex* twiddle, int fft_size1, int fft_size2, char dir, int core_mask) + +**C 调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15-16 + + // FT78NE 示例(共享存储版本) + #include + #include + + int main(int argc, char* argv[]) { + fftw_complex *input = (fftw_complex *)0xA0000000; + fftw_complex *output = (fftw_complex *)0xC0000000; + fftw_complex *scratch = (fftw_complex *)0xA1000000; + fftw_complex *twiddle = (fftw_complex *)0xA2000000; + + int fft_size1 = 32; + int fft_size2 = 32; + int core_mask = 0xff; + + c64_rfft_s(input, output, scratch, twiddle, + fft_size1, fft_size2, FFT_FORWARD, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void c64_rfft_p(fftw_complex* input, fftw_complex* output, fftw_complex* scratch_ptr, int fft_size, char dir) +.. c:function:: void c128_rfft_p(fftw_complex* input, fftw_complex* output, fftw_complex* scratch_ptr, int fft_size, char dir) + +**C 调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12-13 + + // FT78NE 示例(私有存储版本) + #include + #include + + int main(int argc, char* argv[]) { + fftw_complex *input = (fftw_complex *)0x10810000; + fftw_complex *output = (fftw_complex *)0x10820000; + fftw_complex *scratch = (fftw_complex *)0x10830000; + + int fft_size = 1024; + + c64_rfft_p(input, output, scratch, fft_size, FFT_FORWARD); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/roipooling.rst.txt b/master/html/_sources/functionlib/dsplib/roipooling.rst.txt new file mode 100644 index 0000000..b18de4c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/roipooling.rst.txt @@ -0,0 +1,98 @@ +ROIPooling +================= + + + +对输入特征图按指定 ROI (Region of Interest) 进行池化操作,将每个 ROI 区域划分为固定大小的池化单元,并取每个通道的最大值输出。 + +.. math:: + + \text{pooled\_height} = \left\lceil \frac{\text{roi\_end\_h} - \text{roi\_start\_h} + 1}{\text{pooled\_height}} \right\rceil + + \text{pooled\_width} = \left\lceil \frac{\text{roi\_end\_w} - \text{roi\_start\_w} + 1}{\text{pooled\_width}} \right\rceil + + \text{pooled\_output}[i,j,c] = \max_{h,w \in \text{bin}(i,j)} \text{input}[\text{roi\_start\_h}+h, \text{roi\_start\_w}+w, c] + +其中: + +- \( \text{bin}(i,j) \) 表示第 \(i\) 行、第 \(j\) 列池化单元对应的输入特征图区域。 +- \(c\) 表示通道索引。 + +输入: + - **in_ptr** - 输入特征图地址。 + - **input_n** - 输入批大小。 + - **input_h** - 输入高度。 + - **input_w** - 输入宽度。 + - **input_c** - 输入通道数。 + - **num_rois** - ROI 数量。 + - **scale** - ROI 缩放因子。 + - **pooled_height** - 池化输出高度。 + - **pooled_width** - 池化输出宽度。 + - **roi** - ROI 坐标数组,格式为 [batch_index, x1, y1, x2, y2]。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **out_ptr** - 池化输出地址。 + - **max_c** - 每通道最大值缓冲区,用于计算池化结果。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, int8 + - MT7004 支持hp, fp + +**共享存储版本:** + +.. c:function:: void fp_roipooling_s(int input_n,int input_h,int input_w,int input_c,int num_rois,float scale,int pooled_height,int pooled_width,const float *in_ptr,float *out_ptr,const float *roi,float *max_c,int core_mask) +.. c:function:: void hp_roipooling_s(int input_n,int input_h,int input_w,int input_c,int num_rois,float scale,int pooled_height,int pooled_width,const half *in_ptr,half *out_ptr,const half *roi,half *max_c,int core_mask) +.. c:function:: void i8_roipooling_s(int input_n,int input_h,int input_w,int input_c,int num_rois,float scale,int pooled_height,int pooled_width,const int8_t *in_ptr,int8_t *out_ptr,const float *roi,int8_t *max_c,int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + #include + + int main() { + float *input = (float *)0xA0000000; // 输入在DDR空间 + float *output = (float *)0xC0000000; + float max_c[64]; // 通道数假设为64 + float roi[10*5]; // 10个ROI示例 + int num_rois = 10; + int core_mask = 0xff; + + fp_roipooling_s(1, 32, 32, 64, num_rois, 1.0f, 7, 7, input, output, roi, max_c, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_roipooling_p(int input_n,int input_h,int input_w,int input_c,int num_rois,float scale,int pooled_height,int pooled_width,const float *in_ptr,float *out_ptr,const float *roi,float *max_c) +.. c:function:: void hp_roipooling_p(int input_n,int input_h,int input_w,int input_c,int num_rois,float scale,int pooled_height,int pooled_width,const half *in_ptr,half *out_ptr,const half *roi,half *max_c) +.. c:function:: void i8_roipooling_p(int input_n,int input_h,int input_w,int input_c,int num_rois,float scale,int pooled_height,int pooled_width,const int8_t *in_ptr,int8_t *out_ptr,const float *roi,int8_t *max_c) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main() { + float *input = (float *)0x10810000; // 输入在L2空间 + float *output = (float *)0x10820000; + float max_c[64]; + float roi[10*5]; + int num_rois = 10; + + fp_roipooling_p(1, 32, 32, 64, num_rois, 1.0f, 7, 7, input, output, roi, max_c); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/round.rst.txt b/master/html/_sources/functionlib/dsplib/round.rst.txt new file mode 100644 index 0000000..0b2646a --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/round.rst.txt @@ -0,0 +1,88 @@ +Round +================= + + +逐元素执行四舍五入(Round to Nearest Integer)运算。 + +该算子对输入张量的每个元素执行就近取整, +当输入值的小数部分恰好为 0.5 时,采用“远离 0”方向取整, +其行为与 C 标准库中的 ``round`` / ``roundf`` 函数一致。 + +.. math:: + + \text{output}_i = \operatorname{round}(\text{input}_i) + + +输入: + - **input** - 输入张量的数据地址。 + - **length** - 输入张量的总元素数量。 + - **core_mask** - 核掩码。 + +输出: + - **output** - 输出张量的数据地址,其大小与 ``input`` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:fp32, fp64 + - MT7004 支持的数据类型:fp16, fp32 + - 当输入为 ``±∞`` 或 ``NaN`` 时,输出结果遵循对应平台数学库的处理规则 + + +**共享存储版本:** + +.. c:function:: void fp_round_s(float* input, float* output, int length, int core_mask) +.. c:function:: void dp_round_s(double* input, double* output, int length, int core_mask) +.. c:function:: void hp_round_s(half* input, half* output, int length, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + float *output = (float *)0xB0000000; // output 在 DDR 空间 + + int length = 4096; + int core_mask = 0xff; + + fp_round_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_round_p(float* input, float* output, int length) +.. c:function:: void dp_round_p(double* input, double* output, int length) +.. c:function:: void hp_round_p(half* input, half* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input = (half *)0x10000000; // input 在 L2 空间 + half *output = (half *)0x10010000; // output 在 L2 空间 + + int length = 1024; + + hp_round_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/rsqrt.rst.txt b/master/html/_sources/functionlib/dsplib/rsqrt.rst.txt new file mode 100644 index 0000000..358fa51 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/rsqrt.rst.txt @@ -0,0 +1,86 @@ +Rsqrt +================= + +逐元素计算输入数据的平方根倒数(Reciprocal Square Root)。 + +.. math:: + + \text{dst}_i = \frac{1}{\sqrt{\text{src}_i}} + +对于输入 `src` 中的每个元素,计算其平方根的倒数。 + +输入: + - **src** - 输入数据地址。 + - **length** - 计算长度(对于复数类型,指复数的个数)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **dst** - 计算结果地址,其大小与 `src` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_rsqrt_s(int8_t* src, float* dst, int length, int core_mask) +.. c:function:: void i16_rsqrt_s(int16_t* src, float* dst, int length, int core_mask) +.. c:function:: void i32_rsqrt_s(int32_t* src, float* dst, int length, int core_mask) +.. c:function:: void hp_rsqrt_s(half* src, half* dst, int length, int core_mask) +.. c:function:: void fp_rsqrt_s(float* src, float* dst, int length, int core_mask) +.. c:function:: void dp_rsqrt_s(double* src, double* dst, int length, int core_mask) +.. c:function:: void c64_rsqrt_s(float* src, float* dst, int length, int core_mask) +.. c:function:: void c128_rsqrt_s(double* src, double* dst, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *src = (float *)0xA0000000; // input在DDR空间 + float *dst = (float *)0xB0000000; // output + int length = 1000; + int core_mask = 0xff; + fp_rsqrt_s(src, dst, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_rsqrt_p(int8_t* src, float* dst, int length) +.. c:function:: void i16_rsqrt_p(int16_t* src, float* dst, int length) +.. c:function:: void i32_rsqrt_p(int32_t* src, float* dst, int length) +.. c:function:: void hp_rsqrt_p(half* src, half* dst, int length) +.. c:function:: void fp_rsqrt_p(float* src, float* dst, int length) +.. c:function:: void dp_rsqrt_p(double* src, double* dst, int length) +.. c:function:: void c64_rsqrt_p(float* src, float* dst, int length) +.. c:function:: void c128_rsqrt_p(double* src, double* dst, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *src = (float *)0x10000000; // input在L2空间 + float *dst = (float *)0x10001000; // output + int length = 1000; + fp_rsqrt_p(src, dst, length); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/rsqrtgrad.rst.txt b/master/html/_sources/functionlib/dsplib/rsqrtgrad.rst.txt new file mode 100644 index 0000000..452aadd --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/rsqrtgrad.rst.txt @@ -0,0 +1,81 @@ +Rsqrtgrad +================= + +计算 Rsqrt(平方根倒数)操作的梯度。该算子是 Rsqrt 算子的反向传播(backward pass)部分。 + +.. math:: + + \text{output}_i = -\frac{1}{2} \times \text{input2}_i \times \text{input1}_i^3 + +其中 `input1` 是前向传播时 Rsqrt 的输出(即 :math:`y = \frac{1}{\sqrt{x}}`),`input2` 是来自后一层的上游梯度 :math:`dy`,`output` 是对原始输入 :math:`x` 的梯度 :math:`dx`。 + +输入: + - **input1** - 前向传播时 Rsqrt 的输出数据地址(即 :math:`y = \frac{1}{\sqrt{x}}`)。 + - **input2** - 来自后一层的上游梯度数据地址(即 :math:`dy`)。 + - **size** - 计算长度。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 计算出的对原始输入的梯度数据地址(即 :math:`dx`)。 + +支持平台: + ``MT7004`` + +.. note:: + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_rsqrtgrad_s(float* input1, float* input2, float* output, int size, int core_mask) +.. c:function:: void hp_rsqrtgrad_s(half* input1, half* input2, half* output, int size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + float *input1 = (float *)0xA0000000; // Rsqrt的输出 y = 1/sqrt(x) + float *input2 = (float *)0xA1000000; // 上游梯度 dy + float *output = (float *)0xB0000000; // 输出梯度 dx + + int size = 1000; + int core_mask = 0xff; + + fp_rsqrtgrad_s(input1, input2, output, size, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_rsqrtgrad_p(float* input1, float* input2, float* output, int size) +.. c:function:: void hp_rsqrtgrad_p(half* input1, half* input2, half* output, int size) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + float *input1 = (float *)0x10000000; // Rsqrt的输出 y = 1/sqrt(x) + float *input2 = (float *)0x10001000; // 上游梯度 dy + float *output = (float *)0x10002000; // 输出梯度 dx + + int size = 1000; + + fp_rsqrtgrad_p(input1, input2, output, size); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/scatter_nd.rst.txt b/master/html/_sources/functionlib/dsplib/scatter_nd.rst.txt new file mode 100644 index 0000000..228f257 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/scatter_nd.rst.txt @@ -0,0 +1,100 @@ +ScatterNd +================= + +依据指定的索引 ``indices``,将更新值 ``updates`` 累加到输出张量 ``output`` 的对应位置。 + +.. math:: + + output[indices_i] = output[indices_i] + updates_i + +输入: + - **output** - 输出张量的起始地址(计算前作为基础值,计算后存储结果)。 + - **output_shape** - 输出张量的形状数组。 + - **output_ndim** - 输出张量的维度数。 + - **indices** - 索引数据地址,其形状通常为 ``(num_slices, indices_depth)``。 + - **indices_shape** - 索引张量的形状数组。 + - **indices_ndim** - 索引张量的维度数。 + - **updates** - 更新数据地址。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 该算子在多核实现中由于涉及随机访存写,通常直接在 DDR 空间操作。 + - 张量维度最大支持 8 维。 + +**共享存储版本:** + +.. c:function:: void i8_scatter_nd_s(int8_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int8_t* updates, int core_mask) +.. c:function:: void i16_scatter_nd_s(int16_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int16_t* updates, int core_mask) +.. c:function:: void i32_scatter_nd_s(int32_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int32_t* updates, int core_mask) +.. c:function:: void hp_scatter_nd_s(half* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, half* updates, int core_mask) +.. c:function:: void fp_scatter_nd_s(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates, int core_mask) +.. c:function:: void dp_scatter_nd_s(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates, int core_mask) +.. c:function:: void c64_scatter_nd_s(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates, int core_mask) +.. c:function:: void c128_scatter_nd_s(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例(共享存储) + #include + #include "78NE/utils.h" + + int main() { + float *output = (float *)0xA0000000; // 基础输出张量在 DDR + float *updates = (float *)0xB0000000; // 更新值在 DDR + int *indices = (int *)0xC0000000; // 索引在 DDR + int out_shape[] = {4, 4, 4}; + int ind_shape[] = {5, 2}; + int out_ndim = 3; + int ind_ndim = 2; + int core_mask = 0x0B; + + fp_scatter_nd_s(output, out_shape, out_ndim, indices, ind_shape, ind_ndim, updates, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_scatter_nd_p(int8_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int8_t* updates) +.. c:function:: void i16_scatter_nd_p(int16_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int16_t* updates) +.. c:function:: void i32_scatter_nd_p(int32_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int32_t* updates) +.. c:function:: void hp_scatter_nd_p(half* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, half* updates) +.. c:function:: void fp_scatter_nd_p(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates) +.. c:function:: void dp_scatter_nd_p(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates) +.. c:function:: void c64_scatter_nd_p(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates) +.. c:function:: void c128_scatter_nd_p(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + //MT7004 示例 + #include + + int main() { + float *output = (float *)0x10810000; + float *updates = (float *)0x10820000; + int *indices = (int *)0x10830000; + int out_shape[] = {4, 4, 4}; + int ind_shape[] = {5, 2}; + int out_ndim = 3; + int ind_ndim = 2; + + fp_scatter_nd_p(output, out_shape, out_ndim, indices, ind_shape, ind_ndim, updates); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/scatter_nd_update.rst.txt b/master/html/_sources/functionlib/dsplib/scatter_nd_update.rst.txt new file mode 100644 index 0000000..24c8574 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/scatter_nd_update.rst.txt @@ -0,0 +1,103 @@ +ScatterNdUpdate +================= + +根据索引(indices)将更新值(updates)散布并更新到输出张量(output)的指定切片中。 + +.. math:: + + output[indices[i]] = updates[i] + +该算子通过 ``indices`` 指定的坐标,定位到 ``output`` 中的特定子部分(切片),并使用 ``updates`` 中对应的值进行覆盖更新。 + +输入: + - **output** - 待更新的输出张量地址(输入/输出)。 + - **output_shape** - 输出张量的形状数组地址。 + - **output_ndim** - 输出张量的维度数。 + - **indices** - 索引张量数据地址,其最后一个维度代表索引深度。 + - **indices_shape** - 索引张量的形状数组地址。 + - **indices_ndim** - 索引张量的维度数。 + - **updates** - 更新数据源地址,其形状必须与索引定位出的切片形状一致。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 更新后的结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - indices 数组的类型固定为 int32。 + - 算子支持张量维度最大为 8 维。 + - 共享存储版本内部使用 DMA 传输加速,直接在 DDR 空间进行切片覆盖。 + +**共享存储版本:** + +.. c:function:: void i8_scatter_nd_update_s(int8_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int8_t* updates, int core_mask) +.. c:function:: void i16_scatter_nd_update_s(int16_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int16_t* updates, int core_mask) +.. c:function:: void i32_scatter_nd_update_s(int* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int* updates, int core_mask) +.. c:function:: void hp_scatter_nd_update_s(half* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, half* updates, int core_mask) +.. c:function:: void fp_scatter_nd_update_s(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates, int core_mask) +.. c:function:: void dp_scatter_nd_update_s(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates, int core_mask) +.. c:function:: void c64_scatter_nd_update_s(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates, int core_mask) +.. c:function:: void c128_scatter_nd_update_s(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 16 + + // FT78NE 示例(共享存储) + #include + #include "78NE/utils.h" + + int main() { + float *output = (float *)0xA0000000; // 原始张量在 DDR + float *updates = (float *)0xB0000000; // 更新值在 DDR + int *indices = (int *)0xC0000000; // 索引在 DDR + + int out_shape[] = {4, 4, 4}; + int ind_shape[] = {5, 2}; // 更新5个切片,每个索引深度为2 + int out_ndim = 3; + int ind_ndim = 2; + int core_mask = 0xFF; // 使用8核并行 + + fp_scatter_nd_update_s(output, out_shape, out_ndim, indices, ind_shape, ind_ndim, updates, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_scatter_nd_update_p(int8_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int8_t* updates) +.. c:function:: void i16_scatter_nd_update_p(int16_t* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int16_t* updates) +.. c:function:: void i32_scatter_nd_update_p(int* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, int* updates) +.. c:function:: void hp_scatter_nd_update_p(half* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, half* updates) +.. c:function:: void fp_scatter_nd_update_p(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates) +.. c:function:: void dp_scatter_nd_update_p(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates) +.. c:function:: void c64_scatter_nd_update_p(float* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, float* updates) +.. c:function:: void c128_scatter_nd_update_p(double* output, int* output_shape, int output_ndim, int* indices, int* indices_shape, int indices_ndim, double* updates) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + + int main() { + float *output = (float *)0x10810000; + float *updates = (float *)0x10820000; + int *indices = (int *)0x10830000; + + int out_shape[] = {4, 4, 4}; + int ind_shape[] = {5, 2}; + + // 调用单核版本 + fp_scatter_nd_update_p(output, out_shape, 3, indices, ind_shape, 2, updates); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/select.rst.txt b/master/html/_sources/functionlib/dsplib/select.rst.txt new file mode 100644 index 0000000..04c3458 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/select.rst.txt @@ -0,0 +1,188 @@ +Select +================= + +根据条件张量逐元素选择输入值。对于每个输出位置,如果条件为真(True),则选择 `input0` 的值;否则选择 `input1` 的值。该算子支持广播机制。 + +.. math:: + + \text{output}_i = \begin{cases} + \text{input0}[idx2], & \text{if } \text{condition}[idx1] = \text{True} \\ + \text{input1}[idx3], & \text{if } \text{condition}[idx1] = \text{False} + \end{cases} + +其中,当不需要广播时(`is_broadcast = 0`),`idx1 = idx2 = idx3 = i`;当需要广播时(`is_broadcast = 1`),使用索引映射 `index_list1`、`index_list2`、`index_list3` 来确定各个输入张量的索引。 + +输入: + - **input0** - 第一个输入数据地址。当条件为真时选择此值。 + - **input1** - 第二个输入数据地址。当条件为假时选择此值。 + - **condition** - 条件数据地址(bool类型)。决定选择哪个输入的值。 + - **output_dims** - 输出张量的维度信息数组。 + - **output_dims_num** - 输出张量的维度数。 + - **index_list1** - 条件张量的索引映射数组,用于广播场景。大小为输出总元素数。 + - **index_list2** - input0 的索引映射数组,用于广播场景。大小为输出总元素数。 + - **index_list3** - input1 的索引映射数组,用于广播场景。大小为输出总元素数。 + - **is_broadcast** - 是否需要广播的标志。0 表示不需要广播,1 表示需要广播。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 输出数据地址,其形状由 `output_dims` 和 `output_dims_num` 确定。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_select_s(int8_t* input0, int8_t* input1, bool* condition, int8_t* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) +.. c:function:: void i16_select_s(int16_t* input0, int16_t* input1, bool* condition, int16_t* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) +.. c:function:: void i32_select_s(int32_t* input0, int32_t* input1, bool* condition, int32_t* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) +.. c:function:: void hp_select_s(half* input0, half* input1, bool* condition, half* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) +.. c:function:: void fp_select_s(float* input0, float* input1, bool* condition, float* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) +.. c:function:: void dp_select_s(double* input0, double* input1, bool* condition, double* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) +.. c:function:: void c64_select_s(float* input0, float* input1, bool* condition, float* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) +.. c:function:: void c128_select_s(double* input0, double* input1, bool* condition, double* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast, int core_mask) + +**C调用示例(无广播):** + +.. code-block:: c + :linenos: + :emphasize-lines: 34-35 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + float *input0 = (float *)0xA0000000; + float *input1 = (float *)0xA1000000; + bool *condition = (bool *)0xA2000000; + float *output = (float *)0xB0000000; + + // 输出形状 [2, 3, 4] + unsigned long long output_dims[] = {2, 3, 4}; + unsigned long long output_dims_num = 3; + + // 计算总元素数 + unsigned long long total_elements = 2 * 3 * 4; // 24 + + // 索引映射数组(无广播时可以为NULL或与输出索引相同) + unsigned long long *index_list1 = (unsigned long long *)0xC0000000; + unsigned long long *index_list2 = (unsigned long long *)0xC0100000; + unsigned long long *index_list3 = (unsigned long long *)0xC0200000; + + // 初始化索引映射(无广播时直接使用顺序索引) + for (unsigned long long i = 0; i < total_elements; i++) { + index_list1[i] = i; + index_list2[i] = i; + index_list3[i] = i; + } + + long long is_broadcast = 0; // 不需要广播 + int core_mask = 0xff; + + fp_select_s(input0, input1, condition, output, output_dims, output_dims_num, + index_list1, index_list2, index_list3, is_broadcast, core_mask); + return 0; + } + +**C调用示例(有广播):** + +.. code-block:: c + :linenos: + :emphasize-lines: 35-36 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + float *input0 = (float *)0xA0000000; // 形状 [3, 4] + float *input1 = (float *)0xA1000000; // 标量或形状 [1] + bool *condition = (bool *)0xA2000000; // 形状 [2, 3, 4] + float *output = (float *)0xB0000000; // 形状 [2, 3, 4] + + // 输出形状 [2, 3, 4] + unsigned long long output_dims[] = {2, 3, 4}; + unsigned long long output_dims_num = 3; + + // 计算总元素数 + unsigned long long total_elements = 2 * 3 * 4; // 24 + + // 索引映射数组(需要根据广播规则预先计算) + unsigned long long *index_list1 = (unsigned long long *)0xC0000000; + unsigned long long *index_list2 = (unsigned long long *)0xC0100000; + unsigned long long *index_list3 = (unsigned long long *)0xC0200000; + + // 初始化索引映射(示例:需要根据实际广播规则计算) + // 这里假设 condition 和 output 形状相同,input0 需要广播 + for (unsigned long long i = 0; i < total_elements; i++) { + index_list1[i] = i; // condition 索引 + index_list2[i] = i % 12; // input0 索引(假设需要广播) + index_list3[i] = 0; // input1 索引(标量) + } + + long long is_broadcast = 1; // 需要广播 + int core_mask = 0xff; + + fp_select_s(input0, input1, condition, output, output_dims, output_dims_num, + index_list1, index_list2, index_list3, is_broadcast, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_select_p(int8_t* input0, int8_t* input1, bool* condition, int8_t* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) +.. c:function:: void i16_select_p(int16_t* input0, int16_t* input1, bool* condition, int16_t* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) +.. c:function:: void i32_select_p(int32_t* input0, int32_t* input1, bool* condition, int32_t* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) +.. c:function:: void hp_select_p(half* input0, half* input1, bool* condition, half* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) +.. c:function:: void fp_select_p(float* input0, float* input1, bool* condition, float* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) +.. c:function:: void dp_select_p(double* input0, double* input1, bool* condition, double* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) +.. c:function:: void c64_select_p(float* input0, float* input1, bool* condition, float* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) +.. c:function:: void c128_select_p(double* input0, double* input1, bool* condition, double* output, unsigned long long* output_dims, unsigned long long output_dims_num, unsigned long long* index_list1, unsigned long long* index_list2, unsigned long long* index_list3, long long is_broadcast) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 30-31 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + float *input0 = (float *)0x10000000; + float *input1 = (float *)0x10001000; + bool *condition = (bool *)0x10002000; + float *output = (float *)0x10003000; + + // 输出形状 [2, 3, 4] + unsigned long long output_dims[] = {2, 3, 4}; + unsigned long long output_dims_num = 3; + + unsigned long long total_elements = 2 * 3 * 4; + unsigned long long *index_list1 = (unsigned long long *)0x10004000; + unsigned long long *index_list2 = (unsigned long long *)0x10005000; + unsigned long long *index_list3 = (unsigned long long *)0x10006000; + + // 初始化索引映射(无广播) + for (unsigned long long i = 0; i < total_elements; i++) { + index_list1[i] = i; + index_list2[i] = i; + index_list3[i] = i; + } + + long long is_broadcast = 0; + + fp_select_p(input0, input1, condition, output, output_dims, output_dims_num, + index_list1, index_list2, index_list3, is_broadcast); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/sgd.rst.txt b/master/html/_sources/functionlib/dsplib/sgd.rst.txt index b4bb87c..3448a96 100644 --- a/master/html/_sources/functionlib/dsplib/sgd.rst.txt +++ b/master/html/_sources/functionlib/dsplib/sgd.rst.txt @@ -2,6 +2,19 @@ SGD ================= 对权重张量执行带动量与权重衰减的随机梯度下降更新。 + .. math:: + + \begin{aligned} + g'_t &= g_t + weight\_decay \cdot w_{t-1} \\ + m_t &= moment \cdot m_{t-1} + (1 - dampening) \cdot g'_t \\ + u_t &= + \begin{cases} + m_t \cdot moment + g'_t, & \text{if nesterov = True} \\ + m_t, & \text{otherwise} + \end{cases} \\ + w_t &= w_{t-1} - learning\_rate \cdot u_t + \end{aligned} + 输入: - **weight** - 待更新权重张量首地址。 - **accumulate** - 动量累积张量首地址。 @@ -34,25 +47,11 @@ SGD .. c:function:: void fp_sgd_s(float *weight, float *accumulate, const float *gradient, float learning_rate, float dampening, float moment, bool nesterov, float weight_decay, int start, int end, int core_mask) - - .. math:: - - \begin{aligned} - g'_t &= g_t + weight\_decay \cdot w_{t-1} \\ - m_t &= moment \cdot m_{t-1} + (1 - dampening) \cdot g'_t \\ - u_t &= - \begin{cases} - m_t \cdot moment + g'_t, & \text{if nesterov = True} \\ - m_t, & \text{otherwise} - \end{cases} \\ - w_t &= w_{t-1} - learning\_rate \cdot u_t - \end{aligned} - **C调用示例:** .. code-block:: c :linenos: - :emphasize-lines: 17 + :emphasize-lines: 17-19 // FT78NE 多核示例 #include @@ -86,7 +85,7 @@ SGD .. code-block:: c :linenos: - :emphasize-lines: 15 + :emphasize-lines: 15-17 // MT7004 单核示例 #include diff --git a/master/html/_sources/functionlib/dsplib/shape.rst.txt b/master/html/_sources/functionlib/dsplib/shape.rst.txt new file mode 100644 index 0000000..5229ad5 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/shape.rst.txt @@ -0,0 +1,81 @@ +Shape +================= + +形状推断函数(Shape Inference Function)。该函数根据输入张量的形状和算子参数,推断输出张量的形状。该函数不区分数据类型,只处理张量的形状信息。 + +如果所有输出张量都是常量(ConstTensor 或 ConstScalar),则直接返回,不进行形状推断。否则,根据算子类型调用相应的形状推断函数。 + +支持的算子类型: + - **Arithmetic_InferShape** - 算术运算的形状推断 + - **Common_InferShape** - 通用算子的形状推断 + - **Softmax_InferShape** - Softmax 算子的形状推断 + - **MaxMinGrad_InferShape** - MaxMin 梯度算子的形状推断 + - **Dropout_InferShape** - Dropout 算子的形状推断 + - **DynamicQuant_InferShape** - 动态量化算子的形状推断 + - **Fft_InferShape** - FFT 算子的形状推断 + - **Flatten_InferShape** - Flatten 算子的形状推断 + - **LayerNorm_InferShape** - LayerNorm 算子的形状推断 + - **LogSoftmax_InferShape** - LogSoftmax 算子的形状推断 + +输入: + - **inputs** - 输入张量数组(TensorC** 类型)。 + - **inputs_size** - 输入张量的数量。 + - **outputs** - 输出张量数组(TensorC** 类型)。 + - **outputs_size** - 输出张量的数量。 + - **param** - 算子参数(OpParameter* 类型),包含算子类型和其他参数信息。 + +输出: + - **outputs** - 输出张量数组,其中的形状信息会被更新。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该函数不区分数据类型,适用于所有数据类型 + - 函数会自动检查输出是否为常量,如果是常量则跳过形状推断 + +**共享存储/私有存储版本:** + +.. c:function:: void shape(TensorC** inputs, int inputs_size, TensorC** outputs, int outputs_size, OpParameter* param) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 32 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + TensorC** input_tensors_ptrs = (TensorC**)0x10010000; + TensorC** output_tensors_ptrs = (TensorC**)0x10011000; + + TensorC input0; + TensorC input1; + TensorC output; + + int input0_shape[4] = {1,2,3,4}; + int input1_shape[4] = {1,3,4}; + int output_shape[4]; //不用初始化 + memcpy(input0.shape_, input0_shape, 4 * sizeof(int)); + input0.shape_size_ = 4; + memcpy(input1.shape_, input1_shape, 4 * sizeof(int)); + input1.shape_size_ = 3; + input0.data_type_ = kNumberTypeFloat32; + input1.data_type_ = kNumberTypeFloat32; + input0.format_ = Format_NCHW; + input1.format_ = Format_NCHW; + + input_tensors_ptrs[0] = &input0; + input_tensors_ptrs[1] = &input1; + output_tensors_ptrs[0] = &output; + + ArithmeticParameter param; + param.op_parameter_.type_ = Arithmetic_InferShape; + + shape(input_tensors_ptrs, 2, output_tensors_ptrs, 1, (OpParameter*)¶m); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.rst.txt b/master/html/_sources/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.rst.txt new file mode 100644 index 0000000..38234ba --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.rst.txt @@ -0,0 +1,103 @@ +SigmoidCrossEntropyWithLogitsGrad +================================= + +计算 **Sigmoid Cross Entropy With Logits** 的梯度。 +该算子以 logits 和 labels 为输入,输出对 logits 的梯度值。 + +数学表达式为: + +.. math:: + + \sigma(x) = \frac{1}{1 + e^{-x}} + + \quad + + dst_i = \sigma(x_i) - y_i + +其中: + - :math:`x_i` 表示第 *i* 个 logit(Input0) + - :math:`y_i` 表示第 *i* 个标签(Input1) + +为提高数值稳定性,计算中对正负 logits 采用不同形式: + +.. math:: + + \sigma(x) = + \begin{cases} + \dfrac{1}{1 + e^{-x}}, & x > 0 \\ + \dfrac{e^{x}}{1 + e^{x}}, & x \le 0 + \end{cases} + +输入: + - **Input0** - logits 输入数据地址。 + - **Input1** - 标签(labels)数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 梯度计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 ``fp32`` 类型 + - MT7004 支持 ``fp16``、``fp32`` 类型 + - 输入 logits 与 labels 必须具有相同长度 + - 输出为对 logits 的梯度,不包含对 labels 的梯度 + +**共享存储版本:** + +.. c:function:: void fp_sigmoidcrossentropywithlogitsgrad_s(float* Input0, float* Input1, float* output, int length, int core_mask) +.. c:function:: void hp_sigmoidcrossentropywithlogitsgrad_s(half* Input0, half* Input1, half* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *logits = (float *)0xA0000000; // DDR 空间 + float *labels = (float *)0xA0100000; + float *output = (float *)0xC0000000; + + int length = 1024; + int core_mask = 0xff; + + fp_sigmoidcrossentropywithlogitsgrad_s(logits, labels, output, length, core_mask); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_sigmoidcrossentropywithlogitsgrad_p(float* Input0, float* Input1, float* output, int length) +.. c:function:: void hp_sigmoidcrossentropywithlogitsgrad_p(half* Input0, half* Input1, half* output, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + // FT78NE 示例(私有存储) + #include + #include + + int main(int argc, char* argv[]) { + float *logits = (float *)0x10810000; // L2 空间 + float *labels = (float *)0x10820000; + float *output = (float *)0x10830000; + + int length = 1024; + fp_sigmoidcrossentropywithlogitsgrad_p( logits, labels, output, length); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/sigmoidcrossentropywithlogits.rst.txt b/master/html/_sources/functionlib/dsplib/sigmoidcrossentropywithlogits.rst.txt new file mode 100644 index 0000000..e36ed69 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sigmoidcrossentropywithlogits.rst.txt @@ -0,0 +1,88 @@ +SigmoidCrossEntropyWithLogits +================================== + + +计算预测值与真实值之间的sigmoid交叉熵。 + +测量离散分类任务中的分布误差,每个类相互独立,且计算出各个类的交叉熵损失。 + +将输入 logits 设置为 :math:`X`,输入 label 为 :math:`Y`,输出为 :math:`loss`。然后, + +.. math:: + + \begin{aligned} + p &= \text{sigmoid}(X) = \frac{1}{1 + e^{-X}} \\ + loss &= -[Y \cdot \ln(p) + (1 - Y) \cdot \ln(1 - p)] + \end{aligned} + +输入: + - **input0** - 输入 logits 张量地址。 + - **input1** - 输入标签张量地址,与 logits 形状相同。 + - **length** - 张量元素总数。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 输出损失张量地址,与输入张量形状相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型:int8, fp32 + - MT7004 支持的数据类型:fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_sigmoidcrossentropywithlogits_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void fp_sigmoidcrossentropywithlogits_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void hp_sigmoidcrossentropywithlogits_s(half* input0, half* input1, half* output, int length, int core_mask) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // logits在DDR空间 + float *input1 = (float *)0xB0000000; // label在DDR空间 + float *output = (float *)0xC0000000; // 输出损失在DDR空间 + int length = 1000; + int core_mask = 0xff; + + // 计算 sigmoid 交叉熵损失 + fp_sigmoidcrossentropywithlogits_s(input0, input1, output, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_sigmoidcrossentropywithlogits_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void fp_sigmoidcrossentropywithlogits_p(float* input0, float* input1, float* output, int length) +.. c:function:: void hp_sigmoidcrossentropywithlogits_p(half* input0, half* input1, half* output, int length) + +**C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input0 = (half *)0x10000000; // logits在L2空间 + half *input1 = (half *)0x10004000; // label在L2空间 + half *output = (half *)0x10008000; // 输出损失在L2空间 + int length = 1000; + + // 计算 sigmoid 交叉熵损失 + hp_sigmoidcrossentropywithlogits_p(input0, input1, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/sin.rst.txt b/master/html/_sources/functionlib/dsplib/sin.rst.txt new file mode 100644 index 0000000..596abb3 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sin.rst.txt @@ -0,0 +1,85 @@ +Sin +================= + + + +传入一个数组,对每个元素逐元素计算其正弦值并输出。 + +.. math:: + + dst_i = \sin(src_i) + +输入角度单位为弧度。 + +输入: + - **src_data** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst_data** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp, dp, int8, int16, int32 + - MT7004 支持 hp, fp, int16, int32 + - 整数类型在计算时会先转换为浮点数,再按对应类型输出 + +**共享存储版本:** + +.. c:function:: void i8_sin_s(int8_t* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void i16_sin_s(int16_t* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void i32_sin_s(int* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void hp_sin_s(half* src_data, half* dst_data, int length, int core_mask) +.. c:function:: void fp_sin_s(float* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void dp_sin_s(double* src_data, double* dst_data, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input在DDR空间 + float *output = (float *)0xC0000000; + int length = 1024; + int core_mask = 0xff; + fp_sin_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_sin_p(int8_t* src_data, float* dst_data, int length) +.. c:function:: void i16_sin_p(int16_t* src_data, float* dst_data, int length) +.. c:function:: void i32_sin_p(int* src_data, float* dst_data, int length) +.. c:function:: void hp_sin_p(half* src_data, half* dst_data, int length) +.. c:function:: void fp_sin_p(float* src_data, float* dst_data, int length) +.. c:function:: void dp_sin_p(double* src_data, double* dst_data, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; // input在L2空间 + float *output = (float *)0x10820000; + int length = 1024; + fp_sin_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/size.rst.txt b/master/html/_sources/functionlib/dsplib/size.rst.txt new file mode 100644 index 0000000..5d9f9bc --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/size.rst.txt @@ -0,0 +1,74 @@ +Size +================= + + 计算输入张量或向量的元素总个数(即形状各维度的乘积),并将结果写入输出地址。 + + 输入: + - **output** - 输出数据的地址,用于存储计算结果。 + - **shape** - 输入张量的形状(维度)数组地址。 + - **n** - 输入张量的维度数(Rank)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **output** - 存储元素总个数的地址。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - 由于该算子对于不同数据类型的具体实现一致,因此统一使用 ``size_s`` 和 ``size_p`` 命名,不再区分数据类型前缀(如 ``fp_``, ``i8_`` 等)。 + - 支持的数据类型包括:int8, int16, int32, fp32, fp64, cplx64, cplx128。 + +**共享存储版本:** + +.. c:function:: void size_s(int* output, int* shape, int n, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + #include + #include + + int main(int argc, char* argv[]) { + int n = 4; + // 假设 shape 为 {2, 3, 4, 5},元素总数为 120 + int *output = (int *)0xA0000000; + int *shape = (int *)0xA0000100; + + // 实际使用中需确保 shape 地址处已存入维度数据 + int core_mask = 0xff; + + size_s(output, shape, n, core_mask); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void size_p(int* output, int* shape, int n) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main(int argc, char* argv[]) { + int n = 3; + // 假设 shape 为 {10, 10, 4},元素总数为 400 + int *output = (int *)0x10000000; + int *shape = (int *)0x10000040; + + // 实际使用中需确保 shape 地址处已存入维度数据 + size_p(output, shape, n); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/skipgram.rst.txt b/master/html/_sources/functionlib/dsplib/skipgram.rst.txt new file mode 100644 index 0000000..dea874d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/skipgram.rst.txt @@ -0,0 +1,125 @@ +Skipgram +======== + +从输入语句中生成skip-gram。skip-gram是从句子中提取的词语序列,其中相邻词语之间的距离(跳过的词语数量)不超过指定的最大值。此算子主要用于自然语言处理(NLP)任务。 + +输入: + - **sentence** - `StringPack*` 类型,指向输入语句的指针。 + - **words** - `StringPack*` 类型,用于存储从语句中解析出的词语的工作空间。 + - **ngram_size** - `int` 类型,指定每个n-gram中的词语数量。 + - **max_skip_size** - `int` 类型,指定在构成n-gram时可以跳过的最大词语数量。 + - **include_all_ngrams** - `int` 类型,布尔标志。如果为非零,则生成长度从1到 `ngram_size` 的所有n-gram;否则,仅生成长度为 `ngram_size` 的n-gram。 + - **grams** - `StringPack**` 类型,用于存储生成的gram的工作空间。 + - **grams_word_count** - `int*` 类型,用于存储每个gram中词语数量的工作空间。 + - **stack** - `int*` 类型,长度为 `ngram_size` 的临时工作空间。 + - **blank** - `char*` 类型,指向空格字符的指针,用于连接gram中的词语。 + - **output_tensor** - `char*` 类型,最终输出张量的数据地址。 + - **offset** - `int*` 类型,输出参数,用于存储 `output_tensor` 中每个gram的偏移量。 + - **len** - `int*` 类型,输出参数,用于存储 `output_tensor` 中每个gram的长度。 + - **shape** - `int*` 类型,输出参数,用于存储输出张量的维度信息。 + - **core_mask** - (仅限共享版本) 核掩码。 + +输出: + - **output_tensor**、**offset**、**len** 和 **shape** - 这些指针指向的内存区域将被填充,以表示包含生成结果的最终张量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分数据类型,直接操作字符数据。 + + +**共享存储版本:** + +.. c:function:: int skipgram_s(StringPack *sentence, StringPack *words, int ngram_size, int max_skip_size, int include_all_ngrams, StringPack **grams, int *grams_word_count, int *stack, char *blank, char *output_tensor, int *offset, int *len, int *shape, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 31-33 + + #include + #include "skipgram.h" // 假设头文件名 + + typedef struct StringPack { + long long len; + char *data; + } StringPack; + + int main(int argc, char* argv[]) { + // 假设输入和工作空间已在DDR中分配 + StringPack* sentence = (StringPack*)0xA0000000; + StringPack* words = (StringPack*)0xA0010000; + StringPack** grams = (StringPack**)0xA0020000; + int* grams_word_count = (int*)0xA0030000; + int* stack = (int*)0xA0040000; + char* output_tensor = (char*)0xB0000000; + int* offset = (int*)0xB0010000; + int* len = (int*)0xB0020000; + int* shape = (int*)0xB0030000; + char blank_char = ' '; + + // 填充sentence内容 + sentence->data = "mindspore signal processing library"; + sentence->len = 33; + + int ngram_size = 2; + int max_skip_size = 1; + int include_all_ngrams = 1; + int core_mask = 0xff; + + skipgram_s(sentence, words, ngram_size, max_skip_size, include_all_ngrams, + grams, grams_word_count, stack, &blank_char, + output_tensor, offset, len, shape, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: int skipgram_p(StringPack *sentence, StringPack *words, int ngram_size, int max_skip_size, int include_all_ngrams, StringPack **grams, int *grams_word_count, int *stack, char *blank, char *output_tensor, int *offset, int *len, int *shape) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 30-32 + + #include + #include "skipgram.h" // 假设头文件名 + + typedef struct StringPack { + long long len; + char *data; + } StringPack; + + int main(int argc, char* argv[]) { + // 假设输入和工作空间已在私有内存中分配 + StringPack* sentence = (StringPack*)0x10001000; + StringPack* words = (StringPack*)0x10002000; + StringPack** grams = (StringPack**)0x10003000; + int* grams_word_count = (int*)0x10004000; + int* stack = (int*)0x10005000; + char* output_tensor = (char*)0x10006000; + int* offset = (int*)0x10007000; + int* len = (int*)0x10008000; + int* shape = (int*)0x10009000; + char blank_char = ' '; + + // 填充sentence内容 + sentence->data = "mindspore signal processing library"; + sentence->len = 33; + + int ngram_size = 2; + int max_skip_size = 1; + int include_all_ngrams = 1; + + skipgram_p(sentence, words, ngram_size, max_skip_size, include_all_ngrams, + grams, grams_word_count, stack, &blank_char, + output_tensor, offset, len, shape); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/slice.rst.txt b/master/html/_sources/functionlib/dsplib/slice.rst.txt new file mode 100644 index 0000000..be9a03d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/slice.rst.txt @@ -0,0 +1,97 @@ +Slice +================= + +从输入张量中提取一个子集(切片)。根据给定的起始索引(begin)和切片大小(size),在每个维度上截取数据。 + +.. math:: + + output[i_0, i_1, \dots, i_{n-1}] = input[begin_0 + i_0, begin_1 + i_1, \dots, begin_{n-1} + i_{n-1}] + +其中 $n$ 为维度数(ndim),且满足 $0 \le i_k < size_k$。 + +输入: + - **input** - 输入张量数据地址。 + - **input_shape** - 输入张量的形状数组地址。 + - **ndim** - 输入张量的维度数。 + - **begin** - 切片起始索引数组地址(长度为 ndim)。 + - **size** - 切片大小数组地址(长度为 ndim)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果存储地址(输出张量形状即为 size)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 切片操作不改变数据数值,仅改变数据的空间排布,常用于特征提取或张量分解。 + +**共享存储版本:** + +.. c:function:: void i8_slice_s(int8_t* input, int8_t* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) +.. c:function:: void i16_slice_s(int16_t* input, int16_t* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) +.. c:function:: void i32_slice_s(int32_t* input, int32_t* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) +.. c:function:: void hp_slice_s(half* input, half* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) +.. c:function:: void fp_slice_s(float* input, float* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) +.. c:function:: void dp_slice_s(double* input, double* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) +.. c:function:: void c64_slice_s(float* input, float* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) +.. c:function:: void c128_slice_s(double* input, double* output, int* input_shape, int ndim, int* begin, int* size, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + // FT78NE 示例(多核并行切片) + #include + #include "78NE/utils.h" + + int main() { + float *input = (float *)0xA0000000; + float *output = (float *)0xB0000000; + int input_shape[] = {16, 32, 64, 128}; + int begin[] = {2, 4, 8, 16}; + int size[] = {8, 12, 20, 32}; + int ndim = 4; + int core_mask = 0xFF; // 使用8核并行 + + fp_slice_s(input, output, input_shape, ndim, begin, size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_slice_p(int8_t* input, int8_t* output, int* input_shape, int ndim, int* begin, int* size) +.. c:function:: void i16_slice_p(int16_t* input, int16_t* output, int* input_shape, int ndim, int* begin, int* size) +.. c:function:: void i32_slice_p(int32_t* input, int32_t* output, int* input_shape, int ndim, int* begin, int* size) +.. c:function:: void hp_slice_p(half* input, half* output, int* input_shape, int ndim, int* begin, int* size) +.. c:function:: void fp_slice_p(float* input, float* output, int* input_shape, int ndim, int* begin, int* size) +.. c:function:: void dp_slice_p(double* input, double* output, int* input_shape, int ndim, int* begin, int* size) +.. c:function:: void c64_slice_p(float* input, float* output, int* input_shape, int ndim, int* begin, int* size) +.. c:function:: void c128_slice_p(double* input, double* output, int* input_shape, int ndim, int* begin, int* size) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + // MT7004 示例(单核私有存储切片) + #include + + int main() { + float *input = (float *)0x10000000; + float *output = (float *)0x10010000; + int input_shape[] = {4, 8, 16, 32}; + int begin[] = {1, 2, 3, 4}; + int size[] = {2, 3, 6, 8}; + int ndim = 4; + + fp_slice_p(input, output, input_shape, ndim, begin, size); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/smooth1loss.rst.txt b/master/html/_sources/functionlib/dsplib/smooth1loss.rst.txt new file mode 100644 index 0000000..331dd61 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/smooth1loss.rst.txt @@ -0,0 +1,85 @@ +SmoothL1Loss +================= + +计算平滑L1损失函数 + +.. math:: + + \text{loss}(x_1, x_2) = \begin{cases} + \frac{(x_1 - x_2)^2}{2\beta}, & \text{if } |x_1 - x_2| < \beta \\ + |x_1 - x_2| - \frac{\beta}{2}, & \text{otherwise} + \end{cases} + +其中 :math:`x_1` 为预测值,:math:`x_2` 为目标值,:math:`\beta` 为平滑参数。 + +输入: + - **predict** - 预测值数据地址。 + - **target** - 目标值数据地址。 + - **length** - 计算长度。 + - **beta** - 平滑参数(FT78NE平台为float值,MT7004平台为half*指针)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **out** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void i8_smoothl1loss_s(int8_t* out, int8_t* predict, int8_t* target, int length, float beta, int core_mask) +.. c:function:: void fp_smoothl1loss_s(float* out, float* predict, float* target, int length, float beta, int core_mask) +.. c:function:: void hp_smoothl1loss_s(half* out, half* predict, half* target, int length, half* beta, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *predict = (float *)0xA0000000; //predict在DDR空间 + float *target = (float *)0xB0000000; //target在DDR空间 + float *out = (float *)0xC0000000; //out在DDR空间 + int length = 1000; + float beta = 1.0f; + int core_mask = 0xff; + fp_smoothl1loss_s(out, predict, target, length, beta, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_smoothl1loss_p(int8_t* out, int8_t* predict, int8_t* target, int length, float beta) +.. c:function:: void fp_smoothl1loss_p(float* out, float* predict, float* target, int length, float beta) +.. c:function:: void hp_smoothl1loss_p(half* out, half* predict, half* target, int length, half* beta) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *predict = (float *)0x10810000; //predict在L2空间 + float *target = (float *)0x10850000; //target在L2空间 + float *out = (float *)0x108A0000; //out在L2空间 + int length = 1000; + float beta = 1.0f; + fp_smoothl1loss_p(out, predict, target, length, beta); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/smoothl1lossgrad.rst.txt b/master/html/_sources/functionlib/dsplib/smoothl1lossgrad.rst.txt new file mode 100644 index 0000000..86b39f4 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/smoothl1lossgrad.rst.txt @@ -0,0 +1,100 @@ +Smoothl1lossgrad +================= + +计算 Smooth L1 Loss 操作的梯度。该算子是 Smooth L1 Loss 算子的反向传播(backward pass)部分。 + +Smooth L1 Loss 是 L1 Loss 和 L2 Loss 的平滑组合,在损失值较小时使用 L2 Loss,在损失值较大时使用 L1 Loss,以减少异常值的影响。 + +.. math:: + + \text{diff}_i = \text{x1}_i - \text{x2}_i + +.. math:: + + \text{dx1}_i = \begin{cases} + \text{dy}_i, & \text{if } \text{diff}_i > \beta \\ + -\text{dy}_i, & \text{if } \text{diff}_i < -\beta \\ + \frac{\text{diff}_i}{\beta} \times \text{dy}_i, & \text{if } -\beta \leq \text{diff}_i \leq \beta + \end{cases} + +其中 `x1` 是预测值(predict),`x2` 是目标值(target),`dy` 是来自后一层的上游梯度,`dx1` 是对预测值 `x1` 的梯度。`beta` 是平滑参数,控制从 L2 Loss 到 L1 Loss 的过渡点。 + + +输入: + - **dy** - 来自后一层的上游梯度数据地址。 + - **x1** - 前向传播时的预测值数据地址。 + - **x2** - 前向传播时的目标值数据地址。 + - **length** - 计算长度。 + - **beta** - 平滑参数,控制从 L2 Loss 到 L1 Loss 的过渡点。通常取值范围为 0.1 到 1.0。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **dx1** - 计算出的对预测值 `x1` 的梯度数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp16, fp32 + +**共享存储版本:** + +.. c:function:: void fp_smoothl1lossgrad_s(float* dy, float* dx1, float* x1, float* x2, int length, float beta, int core_mask) +.. c:function:: void hp_smoothl1lossgrad_s(half* dy, half* dx1, half* x1, half* x2, int length, half beta, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + float *dy = (float *)0xA0000000; // 上游梯度 + float *x1 = (float *)0xA1000000; // 预测值 + float *x2 = (float *)0xA2000000; // 目标值 + float *dx1 = (float *)0xB0000000; // 输出梯度(对 x1 的梯度) + + int length = 1000; + float beta = 1.0f; // 平滑参数 + int core_mask = 0xff; + + fp_smoothl1lossgrad_s(dy, dx1, x1, x2, length, beta, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_smoothl1lossgrad_p(float* dy, float* dx1, float* x1, float* x2, int length, float beta) +.. c:function:: void hp_smoothl1lossgrad_p(half* dy, half* dx1, half* x1, half* x2, int length, half beta) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + float *dy = (float *)0x10000000; // 上游梯度 + float *x1 = (float *)0x10001000; // 预测值 + float *x2 = (float *)0x10002000; // 目标值 + float *dx1 = (float *)0x10003000; // 输出梯度(对 x1 的梯度) + + int length = 1000; + float beta = 1.0f; // 平滑参数 + + fp_smoothl1lossgrad_p(dy, dx1, x1, x2, length, beta); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/softmax.rst.txt b/master/html/_sources/functionlib/dsplib/softmax.rst.txt new file mode 100644 index 0000000..3a1a51c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/softmax.rst.txt @@ -0,0 +1,94 @@ +Softmax +================= + + + +对输入数组沿指定维度进行 Softmax 计算,输出每个元素的概率值。 + +.. math:: + + \text{output}_{i} = \frac{\exp(\text{input}_{i})}{\sum_j \exp(\text{input}_j)} + \quad \text{for elements along the given axis} + +输入: + - **input_ptr** - 输入数据地址。 + - **axis** - 归一化的轴。 + - **n_dim** - 输入张量维度。 + - **inner_size** - 内部尺寸(轴之后的元素个数)。 + - **outter_size** - 外部尺寸(轴之前的元素个数)。 + - **axis_size** - 归一化轴的元素数量。 + - **sum_data** - 中间累加存储地址(用于存放指数和)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output_ptr** - Softmax 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, int8 + - MT7004 支持hp, fp + +**共享存储版本:** + +.. c:function:: void fp_softmax_s(float* input_ptr, float* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask) +.. c:function:: void hp_softmax_s(half* input_ptr, half* output_ptr, half* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask) +.. c:function:: void i8_softmax_s(int8_t* input_ptr, int8_t* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 16 + + //FT78NE示例 + #include + #include + + int main() { + float *input = (float *)0xA0000000; // input在DDR空间 + float *output = (float *)0xC0000000; + float *sum_data = (float *)0xD0000000; + int axis = 1; + int n_dim = 3; + int inner_size = 4; + int outter_size = 2; + int axis_size = 3; + int core_mask = 0xff; + + fp_softmax_s(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_softmax_p(float* input_ptr, float* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size) +.. c:function:: void hp_softmax_p(half* input_ptr, half* output_ptr, half* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size) +.. c:function:: void i8_softmax_p(int8_t* input_ptr, int8_t* output_ptr, float* sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //FT78NE示例 + #include + #include + + int main() { + float *input = (float *)0x10810000; // input在L2空间 + float *output = (float *)0x10820000; + float *sum_data = (float *)0x10830000; + int axis = 1; + int n_dim = 3; + int inner_size = 4; + int outter_size = 2; + int axis_size = 3; + + fp_softmax_p(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/softmax_cross_entropy_with_logits.rst.txt b/master/html/_sources/functionlib/dsplib/softmax_cross_entropy_with_logits.rst.txt new file mode 100644 index 0000000..eb2cd4c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/softmax_cross_entropy_with_logits.rst.txt @@ -0,0 +1,125 @@ +SoftmaxCrossEntropyWithLogits +================================= + + 计算 Softmax 交叉熵损失及梯度。 + + 该算子首先对输入 ``logits`` 进行 Softmax 归一化得到概率,然后计算其与 ``labels`` 的交叉熵损失。如果启用了 ``need_grads``,则会计算损失相对于 ``logits`` 的梯度。 + + 算法逻辑如下: + + 1. **Softmax**: + + .. math:: + + p_{i,j} = \frac{e^{x_{i,j}}}{\sum_{k} e^{x_{i,k}}} + + 2. **Cross Entropy Loss**: + + .. math:: + + loss_i = - \sum_{j} y_{i,j} \log(p_{i,j}) + + 3. **Gradients** (当 need_grads=1 时): + + .. math:: + + \frac{\partial loss}{\partial x_{i,j}} = p_{i,j} - y_{i,j} + + 输入: + - **logits** - 输入数据地址(未归一化的对数概率)。形状为 :math:`[batch\_size, num\_of\_classes]`。 + - **labels** - 标签数据地址。形状为 :math:`[batch\_size, num\_of\_classes]`。 + - **probs** - 输出概率地址(Softmax 结果)。 + - **grads** - 输出梯度地址。如果 ``need_grads`` 为 0,可忽略。 + - **output** - 输出损失值地址(通常为标量或每个样本的损失)。 + - **sum_data** - 中间计算缓冲区(Workspace),用于存储行和等临时数据。 + - **batch_size** - 批大小。 + - **num_of_classes** - 类别数量。 + - **need_grads** - 是否需要计算梯度 (0: 不计算, 1: 计算)。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **probs** - 更新后的概率分布。 + - **grads** - 计算得到的梯度(如启用)。 + - **output** - 计算得到的交叉熵损失。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - **FT78NE** 支持 Fp32 和 Int8 数据类型。 + - **MT7004** 支持 Fp32 和 FP16 数据类型。 + - 输入数据必须是二维矩阵,行优先存储。 + - Int8 版本通常采用混合精度计算:输入/标签为 ``int8_t``,但输出(概率、梯度、损失)为 ``float`` 以保证精度。 + +**共享存储版本:** + +.. c:function:: void fp_softmax_cross_entropy_with_logits_s(float* logits, float* labels, float* probs, float* grads, float* output, float* sum_data, int batch_size, int num_of_classes, int need_grads, int core_mask) +.. c:function:: void hp_softmax_cross_entropy_with_logits_s(float16* logits, float16* labels, float16* probs, float16* grads, float16* output, float16* sum_data, int batch_size, int num_of_classes, int need_grads, int core_mask) +.. c:function:: void i8_softmax_cross_entropy_with_logits_s(int8_t* logits, int8_t* labels, float* probs, float* grads, float* output, float* sum_data, int batch_size, int num_of_classes, int need_grads, int core_mask) + + **C调用示例(FT78NE - Int8):** + + .. code-block:: c + :linenos: + :emphasize-lines: 19-21 + + #include + #include + + int main(int argc, char* argv[]) { + // 假设所有数据均位于DDR空间 + int8_t* logits = (int8_t*)0xC0000000; + int8_t* labels = (int8_t*)0xC1000000; + float* probs = (float*)0xC2000000; + float* grads = (float*)0xC3000000; + float* output = (float*)0xC4000000; + float* sum_data= (float*)0xC5000000; // Workspace + + int batch_size = 128; + int num_of_classes = 1000; + int need_grads = 1; + int core_mask = 0xff; // 使用所有核心 + + // i8版本:输入为int8,输出及中间计算为float + i8_softmax_cross_entropy_with_logits_s(logits, labels, probs, grads, output, + sum_data, batch_size, num_of_classes, + need_grads, core_mask); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_softmax_cross_entropy_with_logits_p(float* logits, float* labels, float* probs, float* grads, float* output, float* sum_data, int batch_size, int num_of_classes, int need_grads) +.. c:function:: void hp_softmax_cross_entropy_with_logits_p(float16* logits, float16* labels, float16* probs, float16* grads, float16* output, float16* sum_data, int batch_size, int num_of_classes, int need_grads) +.. c:function:: void i8_softmax_cross_entropy_with_logits_p(int8_t* logits, int8_t* labels, float* probs, float* grads, float* output, float* sum_data, int batch_size, int num_of_classes, int need_grads) + + **C调用示例(MT7004 - FP16):** + + .. code-block:: c + :linenos: + :emphasize-lines: 16-18 + + #include + + int main(int argc, char* argv[]) { + // 假设所有数据均位于AM空间 + float16* logits = (float16*)0x10010000; + float16* labels = (float16*)0x10020000; + float16* probs = (float16*)0x10030000; + float16* grads = (float16*)0x10040000; + float16* output = (float16*)0x10050000; + float16* sum_data=(float16*)0x10060000; + + int batch_size = 32; + int num_of_classes = 512; + int need_grads = 1; + + hp_softmax_cross_entropy_with_logits_p(logits, labels, probs, grads, output, + sum_data, batch_size, num_of_classes, + need_grads); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/sparse_softmax_cross_entropy_with_logits.rst.txt b/master/html/_sources/functionlib/dsplib/sparse_softmax_cross_entropy_with_logits.rst.txt new file mode 100644 index 0000000..54a7644 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sparse_softmax_cross_entropy_with_logits.rst.txt @@ -0,0 +1,140 @@ +SparseSoftmaxCrossEntropyWithLogits +======================================= + + 计算稀疏标签下的 Softmax 交叉熵损失及梯度。 + + 与 SoftmaxCrossEntropyWithLogits 不同,该算子的标签(Labels)是类别的索引(整数),而不是 One-hot 向量。这通常能节省内存并加速计算。 + + 算法逻辑如下: + + 1. **Softmax**: + + .. math:: + + p_{i,j} = \frac{e^{x_{i,j}}}{\sum_{k} e^{x_{i,k}}} + + 2. **Cross Entropy Loss**: + 假设 :math:`y_i` 为第 :math:`i` 个样本的标签索引: + + .. math:: + + loss_i = - \log(p_{i, y_i}) + + 3. **Gradients** (当 is_grad=1 时): + + .. math:: + + \frac{\partial loss}{\partial x_{i,j}} = p_{i,j} - \mathbb{1}(j == y_i) + + 输入: + - **input** - 输入 Logits 地址。形状为 :math:`[N, C]`。 + - **losses** - 输出损失地址。 + - **sum_data** - 中间计算缓冲区(Workspace)。 + - **inner_size** - 内部维度大小。对于二维输入 :math:`[N, C]`,通常为 1。 + - **outter_size** - 外部维度大小。对于二维输入 :math:`[N, C]`,通常为 :math:`N` (Batch Size)。 + - **axis_size** - 轴维度大小。对于二维输入 :math:`[N, C]`,通常为 :math:`C` (Num Classes)。 + - **labels** - 标签索引地址。类型为 int,形状为 :math:`[N]`。 + - **is_grad** - 是否计算梯度 (0: 否, 1: 是)。 + - **output** - 输出梯度地址。如果 ``is_grad`` 为 1,则存储计算出的梯度,形状同输入。 + - **batch_size** - 批大小 (N)。 + - **number_of_classes** - 类别数量 (C)。 + - **partial_losses** - 中间计算缓冲区(Workspace),用于归约计算。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + + 输出: + - **losses** - 计算得到的交叉熵损失。 + - **output** - 计算得到的梯度(若启用)。 + + 支持平台: + ``FT78NE`` + ``MT7004`` + + .. note:: + - **FT78NE** 仅支持 Fp32 数据类型。 + - **MT7004** 支持 Fp32 和 FP16 数据类型。 + - 输入数据必须是二维矩阵。 + - FP16 模式下,labels 依然保持为 int 类型,其余浮点指针变为 float16*。 + +**共享存储版本:** + +.. c:function:: void fp_sparse_softmax_cross_entropy_with_logits_s(float* input, float* losses, float* sum_data, uint64_t inner_size, uint64_t outter_size, uint64_t axis_size, int* labels, uint64_t is_grad, float* output, uint64_t batch_size, uint64_t number_of_classes, float* partial_losses, int core_mask) +.. c:function:: void hp_sparse_softmax_cross_entropy_with_logits_s(float16* input, float16* losses, float16* sum_data, uint64_t inner_size, uint64_t outter_size, uint64_t axis_size, int* labels, uint64_t is_grad, float16* output, uint64_t batch_size, uint64_t number_of_classes, float16* partial_losses, int core_mask) + + **C调用示例(FT78NE - Fp32):** + + .. code-block:: c + :linenos: + :emphasize-lines: 24-29 + + #include + #include + + int main(int argc, char* argv[]) { + // 假设所有数据位于 DDR 空间 + float* input = (float*)0xC0000000; + float* losses = (float*)0xC1000000; + float* sum_data = (float*)0xC2000000; + int* labels = (int*)0xC3000000; + float* output = (float*)0xC4000000; + float* partial_losses = (float*)0xC5000000; + + uint64_t batch_size = 64; + uint64_t number_of_classes = 1000; + + // 针对二维输入 [N, C] 的常用配置 + uint64_t outter_size = batch_size; + uint64_t axis_size = number_of_classes; + uint64_t inner_size = 1; + + uint64_t is_grad = 1; // 计算梯度 + int core_mask = 0xff; + + fp_sparse_softmax_cross_entropy_with_logits_s(input, losses, sum_data, + inner_size, outter_size, axis_size, + labels, is_grad, output, + batch_size, number_of_classes, + partial_losses, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_sparse_softmax_cross_entropy_with_logits_p(float* input, float* losses, float* sum_data, uint64_t inner_size, uint64_t outter_size, uint64_t axis_size, int* labels, uint64_t is_grad, float* output, uint64_t batch_size, uint64_t number_of_classes, float* partial_losses) +.. c:function:: void hp_sparse_softmax_cross_entropy_with_logits_p(float16* input, float16* losses, float16* sum_data, uint64_t inner_size, uint64_t outter_size, uint64_t axis_size, int* labels, uint64_t is_grad, float16* output, uint64_t batch_size, uint64_t number_of_classes, float16* partial_losses) + + **C调用示例(MT7004 - FP16):** + + .. code-block:: c + :linenos: + :emphasize-lines: 22-26 + + #include + #include + + int main(int argc, char* argv[]) { + // 假设数据位于 AM 空间 + float16* input = (float16*)0x10010000; + float16* losses = (float16*)0x10020000; + float16* sum_data = (float16*)0x10030000; + int* labels = (int*)0x10040000; // 标签仍为 int + float16* output = (float16*)0x10050000; + float16* partial_losses = (float16*)0x10060000; + + uint64_t batch_size = 32; + uint64_t number_of_classes = 100; + + uint64_t outter_size = batch_size; + uint64_t axis_size = number_of_classes; + uint64_t inner_size = 1; + + uint64_t is_grad = 0; // 仅计算 Loss,不计算梯度 + + hp_sparse_softmax_cross_entropy_with_logits_p(input, losses, sum_data, + inner_size, outter_size, axis_size, + labels, is_grad, output, + batch_size, number_of_classes, + partial_losses); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/sparsefillemptyrows.rst.txt b/master/html/_sources/functionlib/dsplib/sparsefillemptyrows.rst.txt new file mode 100644 index 0000000..028f4df --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sparsefillemptyrows.rst.txt @@ -0,0 +1,122 @@ +SparseFillEmptyRows +==================== + + + +对稀疏张量按行进行补全操作。当某一行在输入稀疏表示中不存在非零元素时, +使用给定的 **default_value** 为该行补充一个元素,并生成新的稀疏表示结果。 +同时可选地输出反向索引映射关系。 + +该算子常用于保证稀疏张量在行维度上的完备性。 + +.. math:: + + \text{if row } r \text{ is empty:} \quad + (r, 0, \dots) \rightarrow default\_value + +输入: + - **N** - 输入稀疏元素个数。 + - **rank** - 稀疏张量的秩(索引维度)。 + - **dense_rows** - 稠密行数。 + - **indices_ptr** - 输入稀疏索引数组地址,大小为 ``N × rank``。 + - **values_ptr** - 输入稀疏值数组地址。 + - **default_value** - 用于填充空行的默认值。 + - **scratch_ptr** - 中间缓冲区,用于存放前缀和信息。 + - **output_reverse_index_map_ptr** - 反向索引映射输出地址(可为 NULL)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output_y_indices_ptr** - 输出稀疏索引数组地址。 + - **output_y_values_ptr** - 输出稀疏值数组地址。 + - **filled_count** - 每一行填充计数结果。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, dp, int8, int16, int32, cplx64, cplx128 + - MT7004 支持hp, fp, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, int8_t *values_ptr, int8_t default_value, int *scratch_ptr, int *output_y_indices_ptr, int8_t *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) +.. c:function:: void i16_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, int16_t *values_ptr, int16_t default_value, int *scratch_ptr, int *output_y_indices_ptr, int16_t *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) +.. c:function:: void i32_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, int32_t *values_ptr, int32_t default_value, int *scratch_ptr, int *output_y_indices_ptr, int32_t *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) +.. c:function:: void hp_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, half *values_ptr, half default_value, int *scratch_ptr, int *output_y_indices_ptr, half *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) +.. c:function:: void fp_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, float *values_ptr, float default_value, int *scratch_ptr, int *output_y_indices_ptr, float *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) +.. c:function:: void dp_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, double *values_ptr, double default_value, int *scratch_ptr, int *output_y_indices_ptr, double *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) +.. c:function:: void c64_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, float *values_ptr, float default_value, int *scratch_ptr, int *output_y_indices_ptr, float *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) +.. c:function:: void c128_sparsefillemptyrows_s(int N, int rank, int dense_rows, int *indices_ptr, double *values_ptr, double default_value, int *scratch_ptr, int *output_y_indices_ptr, double *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17-21 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + int N = 4, rank = 2, dense_rows = 6; + int *indices = (int *)0xA0000000; + float *values = (float *)0xA0010000; + float default_value = 0.0f; + int *scratch = (int *)0xA0020000; + int *out_indices = (int *)0xC0000000; + float *out_values = (float *)0xC0010000; + int *reverse_map = (int *)0xC0020000; + int *filled_count = (int *)0xC0030000; + int core_mask = 0xff; + + fp_sparsefillemptyrows_s( + N, rank, dense_rows, + indices, values, default_value, + scratch, out_indices, out_values, + reverse_map, filled_count, core_mask); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, int8_t *values_ptr, int8_t default_value, int *scratch_ptr, int *output_y_indices_ptr, int8_t *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) +.. c:function:: void i16_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, int16_t *values_ptr, int16_t default_value, int *scratch_ptr, int *output_y_indices_ptr, int16_t *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) +.. c:function:: void i32_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, int32_t *values_ptr, int32_t default_value, int *scratch_ptr, int *output_y_indices_ptr, int32_t *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) +.. c:function:: void hp_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, half *values_ptr, half default_value, int *scratch_ptr, int *output_y_indices_ptr, half *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) +.. c:function:: void fp_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, float *values_ptr, float default_value, int *scratch_ptr, int *output_y_indices_ptr, float *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) +.. c:function:: void dp_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, double *values_ptr, double default_value, int *scratch_ptr, int *output_y_indices_ptr, double *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) +.. c:function:: void c64_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, float *values_ptr, float default_value, int *scratch_ptr, int *output_y_indices_ptr, float *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) +.. c:function:: void c128_sparsefillemptyrows_p(int N, int rank, int dense_rows, int *indices_ptr, double *values_ptr, double default_value, int *scratch_ptr, int *output_y_indices_ptr, double *output_y_values_ptr, int *output_reverse_index_map_ptr, int *filled_count) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15-19 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + int N = 4, rank = 2, dense_rows = 6; + int *indices = (int *)0x10810000; + float *values = (float *)0x10820000; + float default_value = 0.0f; + int *scratch = (int *)0x10830000; + int *out_indices = (int *)0x10840000; + float *out_values = (float *)0x10850000; + int *filled_count = (int *)0x10860000; + + fp_sparsefillemptyrows_p( + N, rank, dense_rows, + indices, values, default_value, + scratch, out_indices, out_values, + NULL, filled_count); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/sparsereshape.rst.txt b/master/html/_sources/functionlib/dsplib/sparsereshape.rst.txt new file mode 100644 index 0000000..d685e56 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sparsereshape.rst.txt @@ -0,0 +1,124 @@ +Sparsereshape +================= + +对稀疏张量进行形状重塑(Reshape)。该算子将稀疏张量的索引从输入形状转换到输出形状,不改变稀疏值本身,只更新索引。该算子不区分数据类型,只处理索引信息。 + + +对于每个稀疏元素: +- 计算其在输入形状中的线性索引: + + :math:`\text{ori\_index} = \sum_{j=0}^{\text{input\_rank}-1} \text{in\_indices}[j] \times \text{in\_stride}[j]` + +- 将线性索引转换为输出形状的多维索引: + + :math:`\text{out\_indices}[j] = \text{ori\_index} / \text{out\_stride}[j]`,然后 :math:`\text{ori\_index} = \text{ori\_index} \% \text{out\_stride}[j]` + +输入: + - **in_indices_ptr** - 输入稀疏张量的索引数组,大小为 `N * input\_rank`,每 `input\_rank` 个元素表示一个非零元素的索引。 + - **in_inshape_ptr** - 输入稀疏张量的形状数组,大小为 `input\_rank`,例如 `[2, 3]` 表示 2×3 的矩阵。 + - **in_outshape_ptr** - 目标输出形状数组,大小为 `output_rank`,例如 `[3, 2]` 表示要将形状转换为 3×2。 + - **out_outshape_ptr** - 输出形状数组,应该与 `in_outshape_ptr` 一致,大小为 `output_rank`。 + - **input_rank** - 输入稀疏张量的维度数。 + - **output_rank** - 输出稀疏张量的维度数。 + - **N** - 稀疏张量中非零元素的数量。 + - **in_stride** - 输入形状的步长数组(临时空间),大小为 `input_rank`,用于中间计算。 + - **out_stride** - 输出形状的步长数组(临时空间),大小为 `output_rank`,用于中间计算。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **out_indices_ptr** - 转换后的索引数组,大小为 `N * output_rank`,每 `output_rank` 个元素表示一个非零元素在输出形状中的索引。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32 + - MT7004 支持fp32 + +**共享存储版本:** + +.. c:function:: void sparsereshape_s(int* in_indices_ptr, int* in_inshape_ptr, int* in_outshape_ptr, int* out_indices_ptr, int* out_outshape_ptr, int input_rank, int output_rank, int N, int* in_stride, int* out_stride, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 31-33 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + int N = 3; // 有3个非零元素 + int input_rank = 2; // 输入是2维 + int output_rank = 2; // 输出也是2维 + + // 输入形状 [2, 3],输出形状 [3, 2] + int in_inshape[] = {2, 3}; + int in_outshape[] = {3, 2}; + int out_outshape[] = {3, 2}; // 应该和 in_outshape 一致 + + // 输入索引:[[0,0], [0,1], [1,2]] + int *in_indices_ptr = (int *)0xA0000000; + in_indices_ptr[0] = 0; in_indices_ptr[1] = 0; // 第一个元素在位置(0,0) + in_indices_ptr[2] = 0; in_indices_ptr[3] = 1; // 第二个元素在位置(0,1) + in_indices_ptr[4] = 1; in_indices_ptr[5] = 2; // 第三个元素在位置(1,2) + + // 输出索引(待填充) + int *out_indices_ptr = (int *)0xB0000000; + + // 临时空间:步长数组 + int *in_stride = (int *)0xC0000000; // 大小为 input_rank + int *out_stride = (int *)0xC0100000; // 大小为 output_rank + + int core_mask = 0xff; + + sparsereshape_s(in_indices_ptr, in_inshape, in_outshape, out_indices_ptr, + out_outshape, input_rank, output_rank, N, in_stride, + out_stride, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void sparsereshape_p(int* in_indices_ptr, int* in_inshape_ptr, int* in_outshape_ptr, int* out_indices_ptr, int* out_outshape_ptr, int input_rank, int output_rank, int N, int* in_stride, int* out_stride) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 24-26 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + int N = 3; + int input_rank = 2; + int output_rank = 2; + + int in_inshape[] = {2, 3}; + int in_outshape[] = {3, 2}; + int out_outshape[] = {3, 2}; + + int *in_indices_ptr = (int *)0x10000000; + in_indices_ptr[0] = 0; in_indices_ptr[1] = 0; + in_indices_ptr[2] = 0; in_indices_ptr[3] = 1; + in_indices_ptr[4] = 1; in_indices_ptr[5] = 2; + + int *out_indices_ptr = (int *)0x10001000; + int *in_stride = (int *)0x10002000; + int *out_stride = (int *)0x10003000; + + sparsereshape_p(in_indices_ptr, in_inshape, in_outshape, out_indices_ptr, + out_outshape, input_rank, output_rank, N, in_stride, + out_stride); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/sparsesegmentsum.rst.txt b/master/html/_sources/functionlib/dsplib/sparsesegmentsum.rst.txt new file mode 100644 index 0000000..e9f4d37 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sparsesegmentsum.rst.txt @@ -0,0 +1,157 @@ +Sparsesegmentsum +================= + +对稀疏分段数据进行求和。该算子根据分段ID(segment_ids)将输入数据的分段进行求和,输出除第0维外其他维度与输入相同的张量。 + +算法流程: +1. 计算输入数据的总元素数和每个切片的大小(按第0维切分) +2. 计算输出数据的形状:第0维为 `max(segment_ids) + 1`,其他维度与输入相同 +3. 初始化输出数据为0 +4. 对每个索引,将其对应的数据切片累加到对应的分段中 + +对于每个索引 `i`: +- 获取对应的分段ID:`segment_id = in_segment_ids[i]` +- 获取对应的输入索引:`input_index = in_indices[i]` +- 将输入数据的切片 `in_data[input_index * n : (input_index + 1) * n]` 累加到输出数据的切片 `out_data[segment_id * n : (segment_id + 1) * n]` + +.. math:: + + \text{out\_data\_shape}[0] = \max(\text{in\_segment\_ids}) + 1 + +.. math:: + + \text{out\_data\_shape}[i] = \text{in\_data\_shape}[i], \quad \text{for } i = 1, 2, \ldots, \text{in\_data\_shape\_size} - 1 + +.. math:: + + n = \frac{\prod_{i=0}^{\text{in\_data\_shape\_size}-1} \text{in\_data\_shape}[i]}{\text{in\_data\_shape}[0]} + +.. math:: + + \text{out\_data}[j + \text{segment\_id} \times n] \mathrel{+}= \text{in\_data}[j + \text{input\_index} \times n], \quad \text{for } j = 0, 1, \ldots, n-1 + +其中 `n` 是每个切片的大小(按第0维切分后的元素数)。 + +输入: + - **in_data** - 输入数据数组,形状由 `in_data_shape` 和 `in_data_shape_size` 确定。 + - **in_indices** - 输入索引数组,大小为 `in_indices_size`,每个元素表示输入数据第0维的索引。 + - **in_segment_ids** - 分段ID数组,大小为 `in_indices_size`,与 `in_indices` 一一对应,表示每个索引所属的分段。 + - **in_data_shape** - 输入数据的形状数组,大小为 `in_data_shape_size`。 + - **in_data_shape_size** - 输入数据的维度数。 + - **in_indices_size** - `in_indices` 和 `in_segment_ids` 数组的大小。 + +输出: + - **out_data** - 输出数据数组,第0维大小为 `max(in_segment_ids) + 1`,其他维度与输入相同。 + - **out_data_shape** - 输出数据的形状数组,大小为 `in_data_shape_size`,由算子内部计算。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void fp_sparsesegmentsum_s(float* in_data, float* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape, int core_mask) +.. c:function:: void hp_sparsesegmentsum_s(half* in_data, half* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape, int core_mask) +.. c:function:: void i16_sparsesegmentsum_s(int16_t* in_data, int16_t* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape, int core_mask) +.. c:function:: void i32_sparsesegmentsum_s(int32_t* in_data, int32_t* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape, int core_mask) +.. c:function:: void dp_sparsesegmentsum_s(double* in_data, double* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape, int core_mask) +.. c:function:: void c64_sparsesegmentsum_s(float* in_data, float* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape, int core_mask) +.. c:function:: void c128_sparsesegmentsum_s(double* in_data, double* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 31-33 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + // 输入数据形状 [4, 3, 2],表示4个样本,每个样本3×2的矩阵 + int in_data_shape[] = {4, 3, 2}; + int in_data_shape_size = 3; + + // 输入数据:4个样本的数据 + float *in_data = (float *)0xA0000000; + // in_data包含 4 * 3 * 2 = 24 个元素 + + // 索引数组:选择第0、2、3个样本 + int in_indices[] = {0, 2, 3}; + int in_indices_size = 3; + + // 分段ID:第0个样本属于分段0,第2个样本属于分段1,第3个样本属于分段1 + int in_segment_ids[] = {0, 1, 1}; + + // 输出数据形状(待计算) + int out_data_shape[3]; + + // 输出数据:分段0有1个样本,分段1有2个样本 + // 输出形状应该是 [2, 3, 2](max(segment_ids)+1=2) + float *out_data = (float *)0xB0000000; + + int core_mask = 0xff; + + fp_sparsesegmentsum_s(in_data, out_data, in_indices, in_segment_ids, + in_data_shape, in_data_shape_size, in_indices_size, + out_data_shape, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_sparsesegmentsum_p(float* in_data, float* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape) +.. c:function:: void hp_sparsesegmentsum_p(half* in_data, half* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape) +.. c:function:: void i16_sparsesegmentsum_p(int16_t* in_data, int16_t* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape) +.. c:function:: void i32_sparsesegmentsum_p(int32_t* in_data, int32_t* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape) +.. c:function:: void dp_sparsesegmentsum_p(double* in_data, double* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape) +.. c:function:: void c64_sparsesegmentsum_p(float* in_data, float* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape) +.. c:function:: void c128_sparsesegmentsum_p(double* in_data, double* out_data, int* in_indices, int* in_segment_ids, int* in_data_shape, int in_data_shape_size, int in_indices_size, int* out_data_shape) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 29-31 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + // 输入数据形状 [4, 3, 2],表示4个样本,每个样本3×2的矩阵 + int in_data_shape[] = {4, 3, 2}; + int in_data_shape_size = 3; + + // 输入数据:4个样本的数据 + float *in_data = (float *)0x10000000; + // in_data包含 4 * 3 * 2 = 24 个元素 + + // 索引数组:选择第0、2、3个样本 + int in_indices[] = {0, 2, 3}; + int in_indices_size = 3; + + // 分段ID:第0个样本属于分段0,第2个样本属于分段1,第3个样本属于分段1 + int in_segment_ids[] = {0, 1, 1}; + + // 输出数据形状(待计算) + int out_data_shape[3]; + + // 输出数据:分段0有1个样本,分段1有2个样本 + // 输出形状应该是 [2, 3, 2](max(segment_ids)+1=2) + float *out_data = (float *)0x10010000; + + fp_sparsesegmentsum_p(in_data, out_data, in_indices, in_segment_ids, + in_data_shape, in_data_shape_size, in_indices_size, + out_data_shape); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/sparsetodense.rst.txt b/master/html/_sources/functionlib/dsplib/sparsetodense.rst.txt new file mode 100644 index 0000000..99b5da8 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sparsetodense.rst.txt @@ -0,0 +1,191 @@ +Sparsetodense +================= + +将稀疏张量转换为密集张量。该算子根据稀疏索引和值,将稀疏元素填充到密集张量的对应位置。支持标量值模式(所有位置使用相同的值)和向量值模式(每个索引对应不同的值)。 + +对于每个稀疏元素,根据其索引计算在密集张量中的线性位置,根据 `is_scalar` 标志,将对应的稀疏值写入密集张量的对应位置 + +.. math:: + + \text{output}[\text{index}] = \begin{cases} + \text{sparse\_values}[0], & \text{if } \text{is\_scalar} = 1 \\ + \text{sparse\_values}[i], & \text{if } \text{is\_scalar} = 0 + \end{cases} + +输入: + - **indices_vec** - 稀疏张量的索引数组,大小为 `sparse_length * 4`,每4个连续的int值表示一个稀疏元素在4维密集张量中的位置坐标 `[dim0, dim1, dim2, dim3]`。 + - **sparse_values** - 稀疏张量的值数组。如果 `is_scalar = 1`,只需要 `sparse_values[0]` 有效;如果 `is_scalar = 0`,需要 `sparse_length` 个值。 + - **output_strides** - 输出密集张量的步长数组,大小为4,用于将多维索引转换为线性索引。`output_strides[0]`, `output_strides[1]`, `output_strides[2]` 对应前3维的步长,第4维的步长为1。 + - **sparse_length** - 稀疏张量中非零元素的数量。 + - **is_scalar** - 是否为标量值标志。1 表示所有稀疏元素使用相同的值(`sparse_values[0]`),0 表示每个索引对应不同的值。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 输出密集张量的数据数组。输出数组的大小由 `output_strides` 和最大索引值确定。未在 `indices_vec` 中指定的位置保持原值(调用前需要初始化)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_sparsetodense_s(int* indices_vec, int8_t* sparse_values, int8_t* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) +.. c:function:: void i16_sparsetodense_s(int* indices_vec, int16_t* sparse_values, int16_t* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) +.. c:function:: void i32_sparsetodense_s(int* indices_vec, int32_t* sparse_values, int32_t* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) +.. c:function:: void hp_sparsetodense_s(int* indices_vec, half* sparse_values, half* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) +.. c:function:: void fp_sparsetodense_s(int* indices_vec, float* sparse_values, float* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) +.. c:function:: void dp_sparsetodense_s(int* indices_vec, double* sparse_values, double* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) +.. c:function:: void c64_sparsetodense_s(int* indices_vec, float* sparse_values, float* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) +.. c:function:: void c128_sparsetodense_s(int* indices_vec, double* sparse_values, double* output, int* output_strides, int sparse_length, int is_scalar, int core_mask) + +**C调用示例(向量值模式):** + +.. code-block:: c + :linenos: + :emphasize-lines: 40-41 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + // 输出密集张量形状 [2, 3, 4, 5] + int output_shape[] = {2, 3, 4, 5}; + + // 计算步长:stride0 = 3*4*5 = 60, stride1 = 4*5 = 20, stride2 = 5, stride3 = 1 + int output_strides[4]; + output_strides[0] = 3 * 4 * 5; // 60 + output_strides[1] = 4 * 5; // 20 + output_strides[2] = 5; // 5 + output_strides[3] = 1; // 第4维步长为1(未使用) + + // 稀疏元素:3个非零元素 + int sparse_length = 3; + + // 索引:[[0,0,0,0], [0,1,2,3], [1,2,3,4]] + int *indices_vec = (int *)0xA0000000; + indices_vec[0] = 0; indices_vec[1] = 0; indices_vec[2] = 0; indices_vec[3] = 0; // 位置(0,0,0,0) + indices_vec[4] = 0; indices_vec[5] = 1; indices_vec[6] = 2; indices_vec[7] = 3; // 位置(0,1,2,3) + indices_vec[8] = 1; indices_vec[9] = 2; indices_vec[10] = 3; indices_vec[11] = 4; // 位置(1,2,3,4) + + // 稀疏值:每个索引对应不同的值 + float *sparse_values = (float *)0xA1000000; + sparse_values[0] = 1.0f; + sparse_values[1] = 2.0f; + sparse_values[2] = 3.0f; + + // 输出密集张量(需要预先初始化为0或默认值) + float *output = (float *)0xB0000000; + int output_size = 2 * 3 * 4 * 5; // 120 + memset(output, 0, output_size * sizeof(float)); + + int is_scalar = 0; // 向量值模式 + int core_mask = 0xff; + + fp_sparsetodense_s(indices_vec, sparse_values, output, output_strides, + sparse_length, is_scalar, core_mask); + + return 0; + } + +**C调用示例(标量值模式):** + +.. code-block:: c + :linenos: + :emphasize-lines: 32-33 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + int output_strides[4]; + output_strides[0] = 3 * 4 * 5; // 60 + output_strides[1] = 4 * 5; // 20 + output_strides[2] = 5; // 5 + output_strides[3] = 1; + + int sparse_length = 3; + + // 索引:[[0,0,0,0], [0,1,2,3], [1,2,3,4]] + int *indices_vec = (int *)0xA0000000; + indices_vec[0] = 0; indices_vec[1] = 0; indices_vec[2] = 0; indices_vec[3] = 0; + indices_vec[4] = 0; indices_vec[5] = 1; indices_vec[6] = 2; indices_vec[7] = 3; + indices_vec[8] = 1; indices_vec[9] = 2; indices_vec[10] = 3; indices_vec[11] = 4; + + // 所有位置使用相同的值 + float *sparse_values = (float *)0xA1000000; + sparse_values[0] = 5.0f; // 只有这个值会被使用 + + float *output = (float *)0xB0000000; + int output_size = 2 * 3 * 4 * 5; + memset(output, 0, output_size * sizeof(float)); + + int is_scalar = 1; // 标量值模式 + int core_mask = 0xff; + + fp_sparsetodense_s(indices_vec, sparse_values, output, output_strides, + sparse_length, is_scalar, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_sparsetodense_p(int* indices_vec, int8_t* sparse_values, int8_t* output, int* output_strides, int sparse_length, int is_scalar) +.. c:function:: void i16_sparsetodense_p(int* indices_vec, int16_t* sparse_values, int16_t* output, int* output_strides, int sparse_length, int is_scalar) +.. c:function:: void i32_sparsetodense_p(int* indices_vec, int32_t* sparse_values, int32_t* output, int* output_strides, int sparse_length, int is_scalar) +.. c:function:: void hp_sparsetodense_p(int* indices_vec, half* sparse_values, half* output, int* output_strides, int sparse_length, int is_scalar) +.. c:function:: void fp_sparsetodense_p(int* indices_vec, float* sparse_values, float* output, int* output_strides, int sparse_length, int is_scalar) +.. c:function:: void dp_sparsetodense_p(int* indices_vec, double* sparse_values, double* output, int* output_strides, int sparse_length, int is_scalar) +.. c:function:: void c64_sparsetodense_p(int* indices_vec, float* sparse_values, float* output, int* output_strides, int sparse_length, int is_scalar) +.. c:function:: void c128_sparsetodense_p(int* indices_vec, double* sparse_values, double* output, int* output_strides, int sparse_length, int is_scalar) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 31-32 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + int output_strides[4]; + output_strides[0] = 3 * 4 * 5; // 60 + output_strides[1] = 4 * 5; // 20 + output_strides[2] = 5; // 5 + output_strides[3] = 1; + + int sparse_length = 3; + + int *indices_vec = (int *)0x10000000; + indices_vec[0] = 0; indices_vec[1] = 0; indices_vec[2] = 0; indices_vec[3] = 0; + indices_vec[4] = 0; indices_vec[5] = 1; indices_vec[6] = 2; indices_vec[7] = 3; + indices_vec[8] = 1; indices_vec[9] = 2; indices_vec[10] = 3; indices_vec[11] = 4; + + float *sparse_values = (float *)0x10001000; + sparse_values[0] = 1.0f; + sparse_values[1] = 2.0f; + sparse_values[2] = 3.0f; + + float *output = (float *)0x10002000; + int output_size = 2 * 3 * 4 * 5; + memset(output, 0, output_size * sizeof(float)); + + int is_scalar = 0; + + fp_sparsetodense_p(indices_vec, sparse_values, output, output_strides, + sparse_length, is_scalar); + + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/splice.rst.txt b/master/html/_sources/functionlib/dsplib/splice.rst.txt new file mode 100644 index 0000000..378ef42 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/splice.rst.txt @@ -0,0 +1,94 @@ +Splice +================= + +对输入数据按给定上下文索引进行拼接操作,将多行特征数据按指定索引映射拼接到输出中。通常用于声学模型或时间序列模型的上下文扩展。 + +输入: + - **src_data** - 输入数据地址。 + - **src_row** - 输入行数。 + - **src_col** - 输入列数。 + - **dst_row** - 输出行数。 + - **dst_col** - 输出列数。 + - **context_dim** - 上下文维度,即每行需要拼接的上下文数量。 + - **forward_indexes** - 上下文索引数组。 + - **forward_indexes_dims** - 上下文索引数组长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst_data** - 拼接后的输出数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, dp, int8, int16, int32, clx64, cplx128 + - MT7004 支持hp, fp, i16, i32, cplx64 + +**共享存储版本:** + +.. c:function:: void fp_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, float* src_data, float* dst_data, int core_mask) +.. c:function:: void hp_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, half* src_data, half* dst_data, int core_mask) +.. c:function:: void dp_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, double* src_data, double* dst_data, int core_mask) +.. c:function:: void i8_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, int8_t* src_data, int8_t* dst_data, int core_mask) +.. c:function:: void i16_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, int16_t* src_data, int16_t* dst_data, int core_mask) +.. c:function:: void i32_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, int* src_data, int* dst_data, int core_mask) +.. c:function:: void c64_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, float* src_data, float* dst_data, int core_mask) +.. c:function:: void c128_splice_s(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, double* src_data, double* dst_data, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + #include + #include + + int main() { + float *input = (float *)0xA0000000; // 输入在DDR空间 + float *output = (float *)0xC0000000; + int src_row = 100, src_col = 64; + int dst_row = 100, dst_col = 64*3; + int context_dim = 3; + int forward_indexes[300]; // 举例 + int forward_indexes_dims = 300; + int core_mask = 0xff; + + fp_splice_s(src_row, src_col, dst_row, dst_col, context_dim, forward_indexes_dims, forward_indexes, input, output, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, float* src_data, float* dst_data) +.. c:function:: void hp_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, half* src_data, half* dst_data) +.. c:function:: void dp_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, double* src_data, double* dst_data) +.. c:function:: void i8_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, int8_t* src_data, int8_t* dst_data) +.. c:function:: void i16_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, int16_t* src_data, int16_t* dst_data) +.. c:function:: void i32_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, int* src_data, int* dst_data) +.. c:function:: void c64_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, float* src_data, float* dst_data) +.. c:function:: void c128_splice_p(int src_row, int src_col, int dst_row, int dst_col, int context_dim, int forward_indexes_dims, int* forward_indexes, double* src_data, double* dst_data) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + #include + #include + + int main() { + float *input = (float *)0x10810000; // 输入在L2空间 + float *output = (float *)0x10820000; + int src_row = 100, src_col = 64; + int dst_row = 100, dst_col = 64*3; + int context_dim = 3; + int forward_indexes[300]; // 举例 + int forward_indexes_dims = 300; + + fp_splice_p(src_row, src_col, dst_row, dst_col, context_dim, forward_indexes_dims, forward_indexes, input, output); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/split.rst.txt b/master/html/_sources/functionlib/dsplib/split.rst.txt new file mode 100644 index 0000000..d468f51 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/split.rst.txt @@ -0,0 +1,103 @@ +Split +================= + +将一个张量(Tensor)沿着指定的轴(axis)拆分为多个子张量。子张量在指定轴上的大小由 ``split_sizes`` 数组决定。 + +.. math:: + + \text{input.shape} = [d_0, d_1, \dots, d_{axis}, \dots, d_{n-1}] + +.. math:: + + \text{对于第 } j \text{ 个输出: } output_j\text{.shape} = [d_0, d_1, \dots, split\_sizes[j], \dots, d_{n-1}] + +输入: + - **input** - 输入数据起始地址。 + - **axis** - 指定拆分的维度轴。 + - **input_shape** - 输入张量的形状数组地址。 + - **input_ndim** - 输入张量的维度数。 + - **num_split** - 拆分出的子张量个数。 + - **split_sizes** - 一个数组,包含每个子张量在拆分轴上的长度。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **outputs** - 指针数组地址,其中每个元素指向一个子张量的存储地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - ``split_sizes`` 的元素之和必须等于输入张量在 ``axis`` 维度的长度。 + - 对于复数类型(cplx64 / cplx128),拆分逻辑与实数一致,但需注意地址偏移按复数对计算。 + +**共享存储版本:** + +.. c:function:: void i8_split_s(int8_t* input, int8_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) +.. c:function:: void i16_split_s(int16_t* input, int16_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) +.. c:function:: void i32_split_s(int32_t* input, int32_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) +.. c:function:: void hp_split_s(half* input, half* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) +.. c:function:: void fp_split_s(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) +.. c:function:: void dp_split_s(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) +.. c:function:: void c64_split_s(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) +.. c:function:: void c128_split_s(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 17 + + //FT78NE示例(共享存储) + #include + #include "78NE/utils.h" + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + float *out0 = (float *)0xB0000000; + float *out1 = (float *)0xB1000000; + float *outputs[] = { out0, out1 }; + int input_shape[] = { 2, 10, 4 }; + int split_sizes[] = { 6, 4 }; + int axis = 1; + int input_ndim = 3; + int num_split = 2; + int core_mask = 0b1011; + + fp_split_s(input, outputs, axis, input_shape, input_ndim, num_split, split_sizes, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_split_p(int8_t* input, int8_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) +.. c:function:: void i16_split_p(int16_t* input, int16_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) +.. c:function:: void i32_split_p(int32_t* input, int32_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) +.. c:function:: void hp_split_p(half* input, half* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) +.. c:function:: void fp_split_p(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) +.. c:function:: void dp_split_p(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) +.. c:function:: void c64_split_p(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) +.. c:function:: void c128_split_p(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* split_sizes) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //MT7004 示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // 私有存储地址 + float *out0 = (float *)0x10010000; + float *out1 = (float *)0x10020000; + float *outputs[] = { out0, out1 }; + int input_shape[] = { 20, 10 }; + int split_sizes[] = { 10, 10 }; + int axis = 0; + fp_split_p(input, outputs, axis, input_shape, 2, 2, split_sizes); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/split_with_overlap.rst.txt b/master/html/_sources/functionlib/dsplib/split_with_overlap.rst.txt new file mode 100644 index 0000000..8455328 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/split_with_overlap.rst.txt @@ -0,0 +1,103 @@ +SplitWithOverlap +================= + +沿指定轴(axis)将输入张量切分为多个输出张量。与标准 Split 不同,该算子允许通过 `start_indices` 和 `end_indices` 自定义每个输出块的起始和结束位置,从而支持输出块之间的重叠(Overlap)。 + +.. math:: + + \text{对于第 } j \text{ 个输出张量,其在 axis 轴上的第 } k \text{ 个元素对应:} + +.. math:: + + Output[j]_{(\dots, k, \dots)} = Input_{(\dots, start\_indices[j] + k, \dots)} \quad \text{其中 } 0 \le k < (end\_indices[j] - start\_indices[j]) + +输入: + - **input** - 输入张量数据地址。 + - **outputs** - 输出张量地址数组(指针数组)。 + - **axis** - 进行切分的轴索引。 + - **input_shape** - 输入张量的形状数组。 + - **input_ndim** - 输入张量的维度。 + - **num_split** - 输出张量的数量。 + - **start_indices** - 每个输出张量在切分轴上的起始索引数组。 + - **end_indices** - 每个输出张量在切分轴上的结束索引数组。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **outputs** - 各个输出张量中填充了切分后的数据。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 算子支持不连续切分或有重叠的切分。 + - 每个输出张量在非切分轴上的维度与输入张量保持一致。 + +**共享存储版本:** + +.. c:function:: void i8_split_with_overlap_s(int8_t* input, int8_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) +.. c:function:: void i16_split_with_overlap_s(int16_t* input, int16_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) +.. c:function:: void i32_split_with_overlap_s(int32_t* input, int32_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) +.. c:function:: void hp_split_with_overlap_s(half* input, half* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) +.. c:function:: void fp_split_with_overlap_s(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) +.. c:function:: void dp_split_with_overlap_s(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) +.. c:function:: void c64_split_with_overlap_s(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) +.. c:function:: void c128_split_with_overlap_s(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例(共享存储) + #include "78NE/utils.h" + + int main() { + float *input = (float *)0xA0000000; + float *out0 = (float *)0xB0000000; + float *out1 = (float *)0xB1000000; + float *outputs[] = {out0, out1}; + int input_shape[] = {8, 200, 10}; + int start_indices[] = {0, 150}; + int end_indices[] = {80, 200}; + int core_mask = 0xFF; + + fp_split_with_overlap_s(input, outputs, 1, input_shape, 3, 2, start_indices, end_indices, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_split_with_overlap_p(int8_t* input, int8_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) +.. c:function:: void i16_split_with_overlap_p(int16_t* input, int16_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) +.. c:function:: void i32_split_with_overlap_p(int32_t* input, int32_t* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) +.. c:function:: void hp_split_with_overlap_p(half* input, half* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) +.. c:function:: void fp_split_with_overlap_p(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) +.. c:function:: void dp_split_with_overlap_p(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) +.. c:function:: void c64_split_with_overlap_p(float* input, float* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) +.. c:function:: void c128_split_with_overlap_p(double* input, double* outputs[], int axis, int* input_shape, int input_ndim, int num_split, int* start_indices, int* end_indices) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //MT7004 示例(私有存储) + #include + + int main() { + float *input = (float *)0x10810000; + float *out0 = (float *)0x10820000; + float *outputs[] = {out0}; + int input_shape[] = {4, 10, 5}; + int start_idx[] = {0}; + int end_idx[] = {5}; + + fp_split_with_overlap_p(input, outputs, 1, input_shape, 3, 1, start_idx, end_idx); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/sqrt.rst.txt b/master/html/_sources/functionlib/dsplib/sqrt.rst.txt new file mode 100644 index 0000000..0c50bca --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sqrt.rst.txt @@ -0,0 +1,87 @@ +Sqrt +================= + +传入一个数组,对每个元素逐元素计算平方根并输出。 + +.. math:: + + dst_i = \sqrt{src_i} + +当输入值为负数时,结果为 NaN(符合 IEEE 浮点标准)。 + +输入: + - **src_data** - 输入数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst_data** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp, dp, int8, int16, int32, cplx64, cplx128 + - MT7004 支持 hp, fp, int16, int32, cplx64 + - 复数类型计算规则为:对复数模长取平方根并保持相位不变 + +**共享存储版本:** + +.. c:function:: void i8_sqrt_s(int8_t* src_data, int8_t* dst_data, int length, int core_mask) +.. c:function:: void i16_sqrt_s(int16_t* src_data, int16_t* dst_data, int length, int core_mask) +.. c:function:: void i32_sqrt_s(int* src_data, int* dst_data, int length, int core_mask) +.. c:function:: void hp_sqrt_s(half* src_data, half* dst_data, int length, int core_mask) +.. c:function:: void fp_sqrt_s(float* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void dp_sqrt_s(double* src_data, double* dst_data, int length, int core_mask) +.. c:function:: void c64_sqrt_s(float* src_data, float* dst_data, int length, int core_mask) +.. c:function:: void c128_sqrt_s(double* src_data, double* dst_data, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + float *output = (float *)0xC0000000; + int length = 1024; + int core_mask = 0xff; + fp_sqrt_s(input, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_sqrt_p(int8_t* src_data, int8_t* dst_data, int length) +.. c:function:: void i16_sqrt_p(int16_t* src_data, int16_t* dst_data, int length) +.. c:function:: void i32_sqrt_p(int* src_data, int* dst_data, int length) +.. c:function:: void hp_sqrt_p(half* src_data, half* dst_data, int length) +.. c:function:: void fp_sqrt_p(float* src_data, float* dst_data, int length) +.. c:function:: void dp_sqrt_p(double* src_data, double* dst_data, int length) +.. c:function:: void c64_sqrt_p(float* src_data, float* dst_data, int length) +.. c:function:: void c128_sqrt_p(double* src_data, double* dst_data, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; + float *output = (float *)0x10820000; + int length = 1024; + fp_sqrt_p(input, output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/sqrtgrad.rst.txt b/master/html/_sources/functionlib/dsplib/sqrtgrad.rst.txt new file mode 100644 index 0000000..4d01eb7 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/sqrtgrad.rst.txt @@ -0,0 +1,95 @@ +Sqrtgrad +================= + +计算 Sqrt(平方根)操作的梯度。该算子是 Sqrt 算子的反向传播(backward pass)部分。 + +.. math:: + + \text{output}_i = \frac{1}{2} \times \frac{\text{input2}_i}{\text{input1}_i} + +其中 `input1` 是前向传播时 Sqrt 的输出(即 :math:`y = \sqrt{x}`),`input2` 是来自后一层的上游梯度 :math:`dy`,`output` 是对原始输入 :math:`x` 的梯度 :math:`dx`。 + +输入: + - **input1** - 前向传播时 Sqrt 的输出数据地址(即 :math:`y = \sqrt{x}`)。 + - **input2** - 来自后一层的上游梯度数据地址(即 :math:`dy`)。 + - **size** - 计算长度。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 计算出的对原始输入的梯度数据地址(即 :math:`dx`)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_sqrtgrad_s(int8_t* input1, int8_t* input2, float* output, int size, int core_mask) +.. c:function:: void i16_sqrtgrad_s(int16_t* input1, int16_t* input2, float* output, int size, int core_mask) +.. c:function:: void i32_sqrtgrad_s(int32_t* input1, int32_t* input2, float* output, int size, int core_mask) +.. c:function:: void hp_sqrtgrad_s(half* input1, half* input2, half* output, int size, int core_mask) +.. c:function:: void fp_sqrtgrad_s(float* input1, float* input2, float* output, int size, int core_mask) +.. c:function:: void dp_sqrtgrad_s(double* input1, double* input2, double* output, int size, int core_mask) +.. c:function:: void c64_sqrtgrad_s(float* input1, float* input2, float* output, int size, int core_mask) +.. c:function:: void c128_sqrtgrad_s(double* input1, double* input2, double* output, int size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + float *input1 = (float *)0xA0000000; // Sqrt的输出 y = sqrt(x) + float *input2 = (float *)0xA1000000; // 上游梯度 dy + float *output = (float *)0xB0000000; // 输出梯度 dx + + int size = 1000; + int core_mask = 0xff; + + fp_sqrtgrad_s(input1, input2, output, size, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_sqrtgrad_p(int8_t* input1, int8_t* input2, float* output, int size) +.. c:function:: void i16_sqrtgrad_p(int16_t* input1, int16_t* input2, float* output, int size) +.. c:function:: void i32_sqrtgrad_p(int32_t* input1, int32_t* input2, float* output, int size) +.. c:function:: void hp_sqrtgrad_p(half* input1, half* input2, half* output, int size) +.. c:function:: void fp_sqrtgrad_p(float* input1, float* input2, float* output, int size) +.. c:function:: void dp_sqrtgrad_p(double* input1, double* input2, double* output, int size) +.. c:function:: void c64_sqrtgrad_p(float* input1, float* input2, float* output, int size) +.. c:function:: void c128_sqrtgrad_p(double* input1, double* input2, double* output, int size) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + float *input1 = (float *)0x10000000; // Sqrt的输出 y = sqrt(x) + float *input2 = (float *)0x10001000; // 上游梯度 dy + float *output = (float *)0x10002000; // 输出梯度 dx + + int size = 1000; + + fp_sqrtgrad_p(input1, input2, output, size); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/square.rst.txt b/master/html/_sources/functionlib/dsplib/square.rst.txt new file mode 100644 index 0000000..eb290f8 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/square.rst.txt @@ -0,0 +1,86 @@ +Square +================= + +逐元素计算输入数据的平方。 + +.. math:: + + \text{dst}_i = \text{src}_i \times \text{src}_i + +对于输入 `src` 中的每个元素,计算其平方值。 + +输入: + - **src** - 输入数据地址。 + - **length** - 计算长度(对于复数类型,指复数的个数)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **dst** - 计算结果地址,其大小与 `src` 相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_square_s(int8_t* src, int8_t* dst, int length, int core_mask) +.. c:function:: void i16_square_s(int16_t* src, int16_t* dst, int length, int core_mask) +.. c:function:: void i32_square_s(int32_t* src, int32_t* dst, int length, int core_mask) +.. c:function:: void hp_square_s(half* src, half* dst, int length, int core_mask) +.. c:function:: void fp_square_s(float* src, float* dst, int length, int core_mask) +.. c:function:: void dp_square_s(double* src, double* dst, int length, int core_mask) +.. c:function:: void c64_square_s(float* src, float* dst, int length, int core_mask) +.. c:function:: void c128_square_s(double* src, double* dst, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *src = (float *)0xA0000000; // input在DDR空间 + float *dst = (float *)0xB0000000; // output + int length = 1000; + int core_mask = 0xff; + fp_square_s(src, dst, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_square_p(int8_t* src, int8_t* dst, int length) +.. c:function:: void i16_square_p(int16_t* src, int16_t* dst, int length) +.. c:function:: void i32_square_p(int32_t* src, int32_t* dst, int length) +.. c:function:: void hp_square_p(half* src, half* dst, int length) +.. c:function:: void fp_square_p(float* src, float* dst, int length) +.. c:function:: void dp_square_p(double* src, double* dst, int length) +.. c:function:: void c64_square_p(float* src, float* dst, int length) +.. c:function:: void c128_square_p(double* src, double* dst, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *src = (float *)0x10000000; // input在L2空间 + float *dst = (float *)0x10001000; // output + int length = 1000; + fp_square_p(src, dst, length); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/squaredifference.rst.txt b/master/html/_sources/functionlib/dsplib/squaredifference.rst.txt new file mode 100644 index 0000000..f313ce3 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/squaredifference.rst.txt @@ -0,0 +1,89 @@ +Squaredifference +================= + +逐元素计算两个输入数组对应元素的差的平方。 + +.. math:: + + \text{output}_i = (\text{input0}_i - \text{input1}_i)^2 + +对于输入 `input0` 和 `input1` 中对应位置的每个元素,计算它们差的平方值。 + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **length** - 计算长度(对于复数类型,指复数的个数)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 计算结果地址,其大小与输入相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_squaredifference_s(int8_t* input0, int8_t* input1, int8_t* output, int length, int core_mask) +.. c:function:: void i16_squaredifference_s(int16_t* input0, int16_t* input1, int16_t* output, int length, int core_mask) +.. c:function:: void i32_squaredifference_s(int32_t* input0, int32_t* input1, int32_t* output, int length, int core_mask) +.. c:function:: void hp_squaredifference_s(half* input0, half* input1, half* output, int length, int core_mask) +.. c:function:: void fp_squaredifference_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void dp_squaredifference_s(double* input0, double* input1, double* output, int length, int core_mask) +.. c:function:: void c64_squaredifference_s(float* input0, float* input1, float* output, int length, int core_mask) +.. c:function:: void c128_squaredifference_s(double* input0, double* input1, double* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // 第一个输入在DDR空间 + float *input1 = (float *)0xA1000000; // 第二个输入在DDR空间 + float *output = (float *)0xB0000000; // output + int length = 1000; + int core_mask = 0xff; + fp_squaredifference_s(input0, input1, output, length, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_squaredifference_p(int8_t* input0, int8_t* input1, int8_t* output, int length) +.. c:function:: void i16_squaredifference_p(int16_t* input0, int16_t* input1, int16_t* output, int length) +.. c:function:: void i32_squaredifference_p(int32_t* input0, int32_t* input1, int32_t* output, int length) +.. c:function:: void hp_squaredifference_p(half* input0, half* input1, half* output, int length) +.. c:function:: void fp_squaredifference_p(float* input0, float* input1, float* output, int length) +.. c:function:: void dp_squaredifference_p(double* input0, double* input1, double* output, int length) +.. c:function:: void c64_squaredifference_p(float* input0, float* input1, float* output, int length) +.. c:function:: void c128_squaredifference_p(double* input0, double* input1, double* output, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; // 第一个输入在L2空间 + float *input1 = (float *)0x10001000; // 第二个输入在L2空间 + float *output = (float *)0x10002000; // output + int length = 1000; + fp_squaredifference_p(input0, input1, output, length); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/stack.rst.txt b/master/html/_sources/functionlib/dsplib/stack.rst.txt new file mode 100644 index 0000000..9a01a70 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/stack.rst.txt @@ -0,0 +1,109 @@ +Stack +================= + +沿一个新轴对输入张量序列进行堆叠。所有输入张量必须具有相同的形状。 + +.. math:: + + \text{若输入 } num\_inputs \text{ 个形状为 } (d_0, d_1, \dots, d_{n-1}) \text{ 的张量,并在 } axis \text{ 处堆叠:} + +.. math:: + + output\_shape = (d_0, \dots, d_{axis-1}, num\_inputs, d_{axis}, \dots, d_{n-1}) + +输入: + - **inputs** - 指针数组,包含所有输入张量起始地址的数组。 + - **input_shape** - 所有输入张量共用的形状(shape)数组。 + - **num_inputs** - 输入张量的数量。 + - **axis** - 堆叠操作插入新维度的位置。 + - **input_ndim** - 输入张量的维度数(秩)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 计算结果存储地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 所有输入张量必须具有完全一致的 `input_shape`。 + - 输出张量的维度将比输入张量多 1。 + +**共享存储版本:** + +.. c:function:: void i8_stack_s(int8_t* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, int8_t* output, int core_mask) +.. c:function:: void i16_stack_s(int16_t* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, int16_t* output, int core_mask) +.. c:function:: void i32_stack_s(int32_t* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, int32_t* output, int core_mask) +.. c:function:: void hp_stack_s(half* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, half* output, int core_mask) +.. c:function:: void fp_stack_s(float* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, float* output, int core_mask) +.. c:function:: void dp_stack_s(double* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, double* output, int core_mask) +.. c:function:: void c64_stack_s(float* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, float* output, int core_mask) +.. c:function:: void c128_stack_s(double* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, double* output, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 20 + + // FT78NE 示例(共享存储) + #include + #include "78NE/utils.h" + + int main() { + // 输入形状为 {4, 10, 8, 12} + int input_shape[] = { 4, 10, 8, 12 }; + int input_ndim = 4; + int num_inputs = 3; + int axis = 1; + int core_mask = 0b1011; + + float *in0 = (float *)0xA0000000; + float *in1 = (float *)0xA2000000; + float *in2 = (float *)0xA4000000; + float *inputs[] = { in0, in1, in2 }; + float *output = (float *)0xB0000000; + // 输出形状将变为 {4, 3, 10, 8, 12} + + fp_stack_s(inputs, input_shape, num_inputs, axis, input_ndim, output, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_stack_p(int8_t* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, int8_t* output) +.. c:function:: void i16_stack_p(int16_t* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, int16_t* output) +.. c:function:: void i32_stack_p(int32_t* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, int32_t* output) +.. c:function:: void hp_stack_p(half* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, half* output) +.. c:function:: void fp_stack_p(float* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, float* output) +.. c:function:: void dp_stack_p(double* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, double* output) +.. c:function:: void c64_stack_p(float* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, float* output) +.. c:function:: void c128_stack_p(double* inputs[], int* input_shape, int num_inputs, int axis, int input_ndim, double* output) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // MT7004 示例(私有存储) + #include + + int main() { + int input_shape[] = { 2, 3, 4, 5 }; + int input_ndim = 4; + int num_inputs = 2; + int axis = 0; + + float *in0 = (float *)0x10810000; + float *in1 = (float *)0x10812000; + float *inputs[] = { in0, in1 }; + float *output = (float *)0x10820000; + + fp_stack_p(inputs, input_shape, num_inputs, axis, input_ndim, output); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/stridedslice.rst.txt b/master/html/_sources/functionlib/dsplib/stridedslice.rst.txt new file mode 100644 index 0000000..bf3b308 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/stridedslice.rst.txt @@ -0,0 +1,171 @@ +Stridedslice +================= + +对输入张量进行步长切片(Strided Slice)操作。该算子根据起始位置(begins)、结束位置(ends)和步长(strides)从输入张量中提取子张量。该算子不区分数据类型,适用于所有数据类型。 + +算法支持三种运行模式: +1. **全拷贝模式**(soft_copy_mode = 1):当输出是输入的连续子区域时,使用内存拷贝。 +2. **快速运行模式**(fast_run = 1):优化的快速路径,适用于特定场景。 +3. **普通模式**:逐块拷贝模式,适用于一般情况。 + +对于每个维度 `i`,输出张量的索引范围由 `begins[i]`、`ends[i]` 和 `strides[i]` 决定: + +.. math:: + + \text{output\_shape}[i] = \left\lceil \frac{\text{ends}[i] - \text{begins}[i]}{\text{strides}[i]} \right\rceil + +输入: + - **input** - 输入数据指针(void*)。 + - **in_shape** - 输入张量的形状数组(int*),大小为 `in_shape_size`。 + - **out_shape** - 输出张量的形状数组(int*),大小为 `out_shape_size`。 + - **in_shape_size** - 输入张量的维度数(int)。 + - **out_shape_size** - 输出张量的维度数(int)。 + - **begins** - 起始位置数组(int*),每个维度切片开始的位置。 + - **ends** - 结束位置数组(int*),每个维度切片结束的位置。 + - **strides** - 步长数组(int*),每个维度的步长。 + - **data_type_bytes** - 数据类型所占字节数(int)。 + - **soft_copy_mode** - 全拷贝模式标志(int),1 表示启用,0 表示不启用。 + - **fast_run** - 快速运行模式标志(int),1 表示启用,0 表示不启用。 + - **split_axis** - 策略2:分割轴(int),当输入和输出张量只有1个维度不同时使用。 + - **inner_size** - 策略2:内部大小(int)。 + - **outer** - 策略2:外部大小(int)。 + - **parallel_on_outer** - 策略2:是否在外层并行(int)。 + - **parallel_on_split_axis** - 策略2:是否在分割轴并行(int)。 + - **cal_num_per_thread** - 策略2:每个线程的计算数量(int)。 + - **caled_num** - 策略2:已计算数量(int)。 + - **temp_space** - 临时空间指针(void*)。 + - **inner** - 策略2:内部大小(int),与 `inner_size` 类似。 + +输出: + - **output** - 输出数据指针(void*)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + - 该算子不区分数据类型,适用于所有数据类型 + - 算子会根据 `soft_copy_mode` 和 `fast_run` 标志自动选择最优的执行路径 + - 策略2相关参数用于优化特定场景(输入和输出张量只有1个维度不同) + +**共享存储版本:** + +.. c:function:: void stridedslice(void* input, void* output, int* in_shape, int* out_shape, int in_shape_size, int out_shape_size, int* begins, int* ends, int* strides, int data_type_bytes, int soft_copy_mode, int fast_run, int split_axis, int inner_size, int outer, int parallel_on_outer, int parallel_on_split_axis, int cal_num_per_thread, int caled_num, void* temp_space, int inner, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 47-50 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + // 输入张量形状 [2, 3, 4, 5] + int in_shape[] = {2, 3, 4, 5}; + int in_shape_size = 4; + + // 输出张量形状 [1, 2, 2, 3] + // 从 [0, 1, 1, 2] 开始,到 [2, 3, 4, 5] 结束,步长为 [1, 1, 2, 1] + int out_shape[] = {1, 2, 2, 3}; + int out_shape_size = 4; + + // 切片参数 + int begins[] = {0, 1, 1, 2}; // 每个维度的起始位置 + int ends[] = {2, 3, 4, 5}; // 每个维度的结束位置 + int strides[] = {1, 1, 2, 1}; // 每个维度的步长 + + // 数据类型:float32,占4字节 + int data_type_bytes = 4; + + // 输入输出数据指针 + float *input = (float *)0xA0000000; + float *output = (float *)0xB0000000; + + // 运行模式 + int soft_copy_mode = 0; // 不启用全拷贝模式 + int fast_run = 0; // 不启用快速运行模式 + + // 策略2参数(不使用时可设为0) + int split_axis = 0; + int inner_size = 0; + int outer = 0; + int parallel_on_outer = 0; + int parallel_on_split_axis = 0; + int cal_num_per_thread = 0; + int caled_num = 0; + int inner = 0; + + // 临时空间 + void *temp_space = (void *)0xC0000000; + + // 核掩码 + + stridedslice(input, output, in_shape, out_shape, in_shape_size, out_shape_size, + begins, ends, strides, data_type_bytes, soft_copy_mode, fast_run, + split_axis, inner_size, outer, parallel_on_outer, parallel_on_split_axis, + cal_num_per_thread, caled_num, temp_space, inner); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void stridedslice(void* input, void* output, int* in_shape, int* out_shape, int in_shape_size, int out_shape_size, int* begins, int* ends, int* strides, int data_type_bytes, int soft_copy_mode, int fast_run, int split_axis, int inner_size, int outer, int parallel_on_outer, int parallel_on_split_axis, int cal_num_per_thread, int caled_num, void* temp_space, int inner) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 37-40 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + int in_shape[] = {2, 3, 4, 5}; + int in_shape_size = 4; + + int out_shape[] = {1, 2, 2, 3}; + int out_shape_size = 4; + + int begins[] = {0, 1, 1, 2}; + int ends[] = {2, 3, 4, 5}; + int strides[] = {1, 1, 2, 1}; + + int data_type_bytes = 4; + + float *input = (float *)0x10000000; + float *output = (float *)0x10010000; + + int soft_copy_mode = 0; + int fast_run = 0; + + // 策略2参数(不使用) + int split_axis = 0; + int inner_size = 0; + int outer = 0; + int parallel_on_outer = 0; + int parallel_on_split_axis = 0; + int cal_num_per_thread = 0; + int caled_num = 0; + int inner = 0; + + void *temp_space = (void *)0x10020000; + + stridedslice(input, output, in_shape, out_shape, in_shape_size, out_shape_size, + begins, ends, strides, data_type_bytes, soft_copy_mode, fast_run, + split_axis, inner_size, outer, parallel_on_outer, parallel_on_split_axis, + cal_num_per_thread, caled_num, temp_space, inner); + + return 0; + } + + diff --git a/master/html/_sources/functionlib/dsplib/stridedslicegrad.rst.txt b/master/html/_sources/functionlib/dsplib/stridedslicegrad.rst.txt new file mode 100644 index 0000000..e837493 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/stridedslicegrad.rst.txt @@ -0,0 +1,118 @@ +Stridedslicegrad +================= + +计算 StridedSlice(步长切片)操作的梯度。该算子是 StridedSlice 算子的反向传播(backward pass)部分。 + +该算子将上游传来的梯度(对应 StridedSlice 输出的形状)映射回原始输入的位置(对应 StridedSlice 输入的形状)。对于输入梯度中的每个元素,根据原始 StridedSlice 操作的 `begins` 和 `strides` 参数,计算其在原始输入中的位置,并将梯度值写入该位置。 + +.. math:: + + \text{output}[\text{idx}] = \text{inputs}[\text{pos}] + +其中 `idx` 是根据 `pos` 在 `in_shape` 中的多维索引、`strides` 和 `begins` 计算出的在 `dx_shape` 中的线性索引。 + + +输入: + - **inputs** - 上游传来的梯度张量数据地址(即 :math:`dy`),形状为 :math:`in\_shape`(原始 StridedSlice 操作的输出形状)。 + - **dx_shape** - 输出梯度张量的形状数组(int*),大小为8,对应原始 StridedSlice 操作的输入形状。对于维度小于8的张量,高位维度形状为1。 + - **strides** - 原始 StridedSlice 操作的步长数组(int*),大小为8。对于维度小于8的张量,高位维度步长为1。 + - **begins** - 原始 StridedSlice 操作的起始索引数组(int*),大小为8。对于维度小于8的张量,高位维度起始索引为0。 + - **in_shape** - 原始 StridedSlice 操作的输出形状数组(int*),大小为8,即输入梯度 `inputs` 的形状。对于维度小于8的张量,高位维度形状为1。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出: + - **output** - 输出梯度张量数据地址(即 :math:`dx`),形状为 :math:`dx\_shape`(原始 StridedSlice 操作的输入形状)。该张量在调用前通常被初始化为全零。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - MT7004 支持fp16, fp32 + - FT78NE 支持fp32 + - 输出张量 `output` 在调用前需要预先初始化为全零 + - 形状数组固定为8维,对于维度小于8的张量,高位维度形状为1 + +**共享存储版本:** + +.. c:function:: void hp_stridedslicegrad_s(half* inputs, half* output, int* dx_shape, int* strides, int* begins, int* in_shape, int core_mask) +.. c:function:: void fp_stridedslicegrad_s(float* inputs, float* output, int* dx_shape, int* strides, int* begins, int* in_shape, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 35 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + // 原始 StridedSlice 操作: + // 输入形状 [2, 3, 4, 5] + // 输出形状 [1, 2, 2, 3] + // begins = [0, 1, 1, 2], strides = [1, 1, 2, 1] + + // 输出梯度形状(原始输入形状) + int dx_shape[8] = {2, 3, 4, 5, 1, 1, 1, 1}; + + // 输入梯度形状(原始输出形状) + int in_shape[8] = {1, 2, 2, 3, 1, 1, 1, 1}; + + // 原始 StridedSlice 参数 + int begins[8] = {0, 1, 1, 2, 0, 0, 0, 0}; + int strides[8] = {1, 1, 2, 1, 1, 1, 1, 1}; + + // 输入梯度(上游传来的梯度) + float *inputs = (float *)0xA0000000; // 形状为 in_shape + // inputs 包含 1 * 2 * 2 * 3 = 12 个元素 + + // 输出梯度(待计算) + float *output = (float *)0xB0000000; // 形状为 dx_shape + // output 包含 2 * 3 * 4 * 5 = 120 个元素 + + // 初始化输出为全零 + memset(output, 0, 120 * sizeof(float)); + + int core_mask = 0xff; + + fp_stridedslicegrad_s(inputs, output, dx_shape, strides, begins, in_shape, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void hp_stridedslicegrad_p(half* inputs, half* output, int* dx_shape, int* strides, int* begins, int* in_shape) +.. c:function:: void fp_stridedslicegrad_p(float* inputs, float* output, int* dx_shape, int* strides, int* begins, int* in_shape) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 19 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + int dx_shape[8] = {2, 3, 4, 5, 1, 1, 1, 1}; + int in_shape[8] = {1, 2, 2, 3, 1, 1, 1, 1}; + + int begins[8] = {0, 1, 1, 2, 0, 0, 0, 0}; + int strides[8] = {1, 1, 2, 1, 1, 1, 1, 1}; + + float *inputs = (float *)0x10000000; + float *output = (float *)0x10001000; + + // 初始化输出为全零 + memset(output, 0, 120 * sizeof(float)); + + fp_stridedslicegrad_p(inputs, output, dx_shape, strides, begins, in_shape); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/subfusion.rst.txt b/master/html/_sources/functionlib/dsplib/subfusion.rst.txt new file mode 100644 index 0000000..24b3e26 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/subfusion.rst.txt @@ -0,0 +1,185 @@ +Subfusion +================= + +逐元素计算两个输入数组的减法运算,支持三种模式:带缩放因子的减法、减法后应用 ReLU 激活、减法后应用 ReLU6 激活。 + +**模式1 - 带缩放因子的减法(subext):** + +.. math:: + + \text{output}_i = \text{input0}_i - \text{input1}_i \times \alpha + +**模式2 - 减法后应用 ReLU(subrelu):** + +.. math:: + + \text{output}_i = \max(0, \text{input0}_i - \text{input1}_i) + +**模式3 - 减法后应用 ReLU6(subrelu6):** + +.. math:: + + \text{output}_i = \min(\max(0, \text{input0}_i - \text{input1}_i), 6) + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **alpha** - 缩放因子(仅 subext 模式需要),用于缩放 `input1`。 + - **size** - 计算长度(对于复数类型,指复数的个数)。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 计算结果地址,其大小与输入相同。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +**subext(带缩放因子):** + +.. c:function:: void hp_subext_s(half* input0, half* input1, half alpha, half* output, int size, int core_mask) +.. c:function:: void fp_subext_s(float* input0, float* input1, float alpha, float* output, int size, int core_mask) +.. c:function:: void dp_subext_s(double* input0, double* input1, double alpha, double* output, int size, int core_mask) +.. c:function:: void c64_subext_s(float* input0, float* input1, float alpha, float* output, int size, int core_mask) +.. c:function:: void c128_subext_s(double* input0, double* input1, double alpha, double* output, int size, int core_mask) + +**subrelu(减法+ReLU):** + +.. c:function:: void i8_subrelu_s(int8_t* input0, int8_t* input1, int8_t* output, int size, int core_mask) +.. c:function:: void i16_subrelu_s(int16_t* input0, int16_t* input1, int16_t* output, int size, int core_mask) +.. c:function:: void i32_subrelu_s(int32_t* input0, int32_t* input1, int32_t* output, int size, int core_mask) +.. c:function:: void hp_subrelu_s(half* input0, half* input1, half* output, int size, int core_mask) +.. c:function:: void fp_subrelu_s(float* input0, float* input1, float* output, int size, int core_mask) +.. c:function:: void dp_subrelu_s(double* input0, double* input1, double* output, int size, int core_mask) +.. c:function:: void c64_subrelu_s(float* input0, float* input1, float* output, int size, int core_mask) +.. c:function:: void c128_subrelu_s(double* input0, double* input1, double* output, int size, int core_mask) + +**subrelu6(减法+ReLU6):** + +.. c:function:: void i8_subrelu6_s(int8_t* input0, int8_t* input1, int8_t* output, int size, int core_mask) +.. c:function:: void i16_subrelu6_s(int16_t* input0, int16_t* input1, int16_t* output, int size, int core_mask) +.. c:function:: void i32_subrelu6_s(int32_t* input0, int32_t* input1, int32_t* output, int size, int core_mask) +.. c:function:: void hp_subrelu6_s(half* input0, half* input1, half* output, int size, int core_mask) +.. c:function:: void fp_subrelu6_s(float* input0, float* input1, float* output, int size, int core_mask) +.. c:function:: void dp_subrelu6_s(double* input0, double* input1, double* output, int size, int core_mask) +.. c:function:: void c64_subrelu6_s(float* input0, float* input1, float* output, int size, int core_mask) +.. c:function:: void c128_subrelu6_s(double* input0, double* input1, double* output, int size, int core_mask) + +**C调用示例(subext):** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; // 第一个输入在DDR空间 + float *input1 = (float *)0xA1000000; // 第二个输入在DDR空间 + float *output = (float *)0xB0000000; // output + float alpha = 0.5f; // 缩放因子 + int size = 1000; + int core_mask = 0xff; + fp_subext_s(input0, input1, alpha, output, size, core_mask); + return 0; + } + +**C调用示例(subrelu):** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; + float *input1 = (float *)0xA1000000; + float *output = (float *)0xB0000000; + int size = 1000; + int core_mask = 0xff; + fp_subrelu_s(input0, input1, output, size, core_mask); + return 0; + } + +**C调用示例(subrelu6):** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; + float *input1 = (float *)0xA1000000; + float *output = (float *)0xB0000000; + int size = 1000; + int core_mask = 0xff; + fp_subrelu6_s(input0, input1, output, size, core_mask); + return 0; + } + +**私有存储版本:** + +**subext(带缩放因子):** + +.. c:function:: void hp_subext_p(half* input0, half* input1, half alpha, half* output, int size) +.. c:function:: void fp_subext_p(float* input0, float* input1, float alpha, float* output, int size) +.. c:function:: void dp_subext_p(double* input0, double* input1, double alpha, double* output, int size) +.. c:function:: void c64_subext_p(float* input0, float* input1, float alpha, float* output, int size) +.. c:function:: void c128_subext_p(double* input0, double* input1, double alpha, double* output, int size) + +**subrelu(减法+ReLU):** + +.. c:function:: void i8_subrelu_p(int8_t* input0, int8_t* input1, int8_t* output, int size) +.. c:function:: void i16_subrelu_p(int16_t* input0, int16_t* input1, int16_t* output, int size) +.. c:function:: void i32_subrelu_p(int32_t* input0, int32_t* input1, int32_t* output, int size) +.. c:function:: void hp_subrelu_p(half* input0, half* input1, half* output, int size) +.. c:function:: void fp_subrelu_p(float* input0, float* input1, float* output, int size) +.. c:function:: void dp_subrelu_p(double* input0, double* input1, double* output, int size) +.. c:function:: void c64_subrelu_p(float* input0, float* input1, float* output, int size) +.. c:function:: void c128_subrelu_p(double* input0, double* input1, double* output, int size) + +**subrelu6(减法+ReLU6):** + +.. c:function:: void i8_subrelu6_p(int8_t* input0, int8_t* input1, int8_t* output, int size) +.. c:function:: void i16_subrelu6_p(int16_t* input0, int16_t* input1, int16_t* output, int size) +.. c:function:: void i32_subrelu6_p(int32_t* input0, int32_t* input1, int32_t* output, int size) +.. c:function:: void hp_subrelu6_p(half* input0, half* input1, half* output, int size) +.. c:function:: void fp_subrelu6_p(float* input0, float* input1, float* output, int size) +.. c:function:: void dp_subrelu6_p(double* input0, double* input1, double* output, int size) +.. c:function:: void c64_subrelu6_p(float* input0, float* input1, float* output, int size) +.. c:function:: void c128_subrelu6_p(double* input0, double* input1, double* output, int size) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; // 第一个输入在L2空间 + float *input1 = (float *)0x10001000; // 第二个输入在L2空间 + float *output = (float *)0x10002000; // output + int size = 1000; + fp_subrelu_p(input0, input1, output, size); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/subgrad.rst.txt b/master/html/_sources/functionlib/dsplib/subgrad.rst.txt new file mode 100644 index 0000000..8a9ce9f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/subgrad.rst.txt @@ -0,0 +1,100 @@ +Subgrad +================= + +计算逐元素减法(Sub)操作的梯度。该算子是 Sub 算子的反向传播(backward pass)部分,支持广播。 + +.. math:: + + \text{dx1} = \frac{\partial L}{\partial X1} = \frac{\partial L}{\partial Y} \times 1 = \frac{\partial L}{\partial Y} + +.. math:: + + \text{dx2} = \frac{\partial L}{\partial X2} = \frac{\partial L}{\partial Y} \times (-1) = -\frac{\partial L}{\partial Y} + +其中对于前向操作 :math:`Y = X1 - X2`,`dy` 是来自后一层的上游梯度,`dx1` 和 `dx2` 分别是对 `X1` 和 `X2` 的梯度。 + + +输入: + - **dy** - 上游梯度数据地址(即 :math:`\frac{\partial L}{\partial Y}`)。 + - **x1_dims** - 前向传播时第一个输入 `x1` 的维度信息数组(int*)。 + - **x2_dims** - 前向传播时第二个输入 `x2` 的维度信息数组(int*)。 + - **dy_dims** - 上游梯度 `dy` 的维度信息数组(int*)。 + - **num_dims** - 维度数(int)。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出: + - **dx1** - 对 `x1` 的梯度数据地址。 + - **dx2** - 对 `x2` 的梯度数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - MT7004 支持fp16, fp32 + - FT78NE 支持fp32 + - 当输入张量被广播时,算子会自动处理广播维度的梯度累加 + +**共享存储版本:** + +.. c:function:: void hp_subgrad_s(half* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, half* dx1, half* dx2, int core_mask) +.. c:function:: void fp_subgrad_s(float* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, float* dx1, float* dx2, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 18 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + float *dy = (float *)0xA0000000; // 上游梯度在DDR空间 + float *dx1 = (float *)0xB0000000; // dx1输出 + float *dx2 = (float *)0xC0000000; // dx2输出 + + // 示例:x1形状 [3, 1],x2形状 [1, 4],dy形状 [3, 4] + int x1_dims[] = {3, 1}; + int x2_dims[] = {1, 4}; + int dy_dims[] = {3, 4}; + int num_dims = 2; + + int core_mask = 0xff; + + fp_subgrad_s(dy, x1_dims, x2_dims, dy_dims, num_dims, dx1, dx2, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void hp_subgrad_p(half* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, half* dx1, half* dx2) +.. c:function:: void fp_subgrad_p(float* dy, int* x1_dims, int* x2_dims, int* dy_dims, int num_dims, float* dx1, float* dx2) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + float *dy = (float *)0x10000000; // 上游梯度在L2空间 + float *dx1 = (float *)0x10001000; // dx1输出 + float *dx2 = (float *)0x10002000; // dx2输出 + + int x1_dims[] = {3, 1}; + int x2_dims[] = {1, 4}; + int dy_dims[] = {3, 4}; + int num_dims = 2; + + fp_subgrad_p(dy, x1_dims, x2_dims, dy_dims, num_dims, dx1, dx2); + + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/switch.rst.txt b/master/html/_sources/functionlib/dsplib/switch.rst.txt new file mode 100644 index 0000000..d9a27ec --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/switch.rst.txt @@ -0,0 +1,124 @@ +Switch +================= + +根据条件选择输入张量。该算子根据布尔条件 `condition` 的值,选择 `input_x` 或 `input_y` 作为输出。该算子不区分数据类型,适用于所有数据类型。 + +.. math:: + + \text{output} = \begin{cases} + \text{input\_x}, & \text{if } \text{condition} = \text{True} \\ + \text{input\_y}, & \text{if } \text{condition} = \text{False} + \end{cases} + +该算子不复制数据,只是将输出指针指向选中的输入张量。因此,输出张量共享输入张量的数据指针和元数据。 + +输入: + - **input_x** - 第一个输入张量(TensorC* 类型)。当 `condition` 为 True 时被选中。 + - **input_y** - 第二个输入张量(TensorC* 类型)。当 `condition` 为 False 时被选中。 + - **condition** - 条件值(bool 类型),决定选择哪个输入张量。 + +输出: + - **output** - 输出张量指针的指针(TensorC** 类型),指向选中的输入张量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分数据类型,适用于所有数据类型 + - 算子不复制数据,输出张量共享输入张量的数据指针 + - 输出张量的所有元数据(形状、数据类型、格式等)与选中的输入张量相同 + +**共享存储版本:** + +.. c:function:: void switch_s(TensorC* input_x, TensorC* input_y, TensorC** output, bool condition) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 32 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + TensorC input_x; + TensorC input_y; + TensorC* output; + + // 初始化 input_x + int x_shape[3] = {2, 3, 4}; + memcpy(input_x.shape_, x_shape, 3 * sizeof(int)); + input_x.shape_size_ = 3; + input_x.data_type_ = kNumberTypeFloat32; + input_x.format_ = Format_NCHW; + input_x.data_ = (void *)0xA0000000; + input_x.category_ = 0; // 非常量 + input_x.shape_changed_ = false; + + // 初始化 input_y + int y_shape[3] = {2, 3, 4}; + memcpy(input_y.shape_, y_shape, 3 * sizeof(int)); + input_y.shape_size_ = 3; + input_y.data_type_ = kNumberTypeFloat32; + input_y.format_ = Format_NCHW; + input_y.data_ = (void *)0xB0000000; + input_y.category_ = 0; + input_y.shape_changed_ = false; + + bool condition = true; // 选择 input_x + + switch_s(&input_x, &input_y, &output, condition); + + return 0; + } + + +**私有存储版本:** + +.. c:function:: void switch_p(TensorC* input_x, TensorC* input_y, TensorC** output, bool condition) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 32 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + TensorC input_x; + TensorC input_y; + TensorC* output; + + // 初始化 input_x + int x_shape[3] = {2, 3, 4}; + memcpy(input_x.shape_, x_shape, 3 * sizeof(int)); + input_x.shape_size_ = 3; + input_x.data_type_ = kNumberTypeFloat32; + input_x.format_ = Format_NCHW; + input_x.data_ = (void *)0x10000000; + input_x.category_ = 0; // 非常量 + input_x.shape_changed_ = false; + + // 初始化 input_y + int y_shape[3] = {2, 3, 4}; + memcpy(input_y.shape_, y_shape, 3 * sizeof(int)); + input_y.shape_size_ = 3; + input_y.data_type_ = kNumberTypeFloat32; + input_y.format_ = Format_NCHW; + input_y.data_ = (void *)0x10001000; + input_y.category_ = 0; + input_y.shape_changed_ = false; + + bool condition = true; // 选择 input_x + + switch_p(&input_x, &input_y, &output, condition); + + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/switchlayer.rst.txt b/master/html/_sources/functionlib/dsplib/switchlayer.rst.txt new file mode 100644 index 0000000..f9eb921 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/switchlayer.rst.txt @@ -0,0 +1,129 @@ +Switchlayer +================= + +根据索引从输入张量数组中选择一个张量,并将其数据复制到输出。该算子不区分数据类型,适用于所有数据类型。 + +.. math:: + + \text{output} = \text{input\_tensors}[\text{index}] + +该算子会将选中的输入张量的数据复制到输出张量中。复制的大小为 `size * type_size` 字节。 + +输入: + - **input_tensors** - 输入张量数组(TensorC** 类型),包含多个待选择的张量。 + - **index** - 选择的索引(int 类型),指定从 `input_tensors` 数组中选择哪个张量。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出: + - **output** - 输出张量(TensorC* 类型),包含复制后的数据。输出张量的 `size` 和 `type_size` 应与选中的输入张量相同。 + +TensorC 结构体定义: + - **type_size** - 数据类型大小(long long),单位字节,例如 float32 为 4,float16 为 2。 + - **size** - 元素个数(long long)。 + - **data** - 数据指针(void*)。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分数据类型,适用于所有数据类型 + - 算子会复制数据,输出张量与输入张量数据独立 + - 调用前需要确保 `output->data` 指向的内存空间足够大(至少 `size * type_size` 字节) + - 选中的输入张量的 `size` 和 `type_size` 应与输出张量匹配 + +**共享存储版本:** + +.. c:function:: void switchlayer_s(TensorC** input_tensors, TensorC* output, int index, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 36 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + TensorC input0, input1, input2; + TensorC output; + + // 初始化 input0 + input0.type_size = 4; // float32 + input0.size = 1000; + input0.data = (void *)0xA0000000; + + // 初始化 input1 + input1.type_size = 4; // float32 + input1.size = 1000; + input1.data = (void *)0xA1000000; + + // 初始化 input2 + input2.type_size = 4; // float32 + input2.size = 1000; + input2.data = (void *)0xA2000000; + + // 初始化 output + output.type_size = 4; // float32 + output.size = 1000; + output.data = (void *)0xB0000000; // 需要预先分配足够的内存 + + // 创建输入张量数组 + TensorC* input_tensors[3] = {&input0, &input1, &input2}; + + int index = 1; // 选择 input1 + int core_mask = 0xff; + + switchlayer_s(input_tensors, &output, index, core_mask); + + // 此时 output.data 包含 input1.data 的副本 + + return 0; + } + +**私有存储版本:** + +.. c:function:: void switchlayer_p(TensorC** input_tensors, TensorC* output, int index) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 30 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + TensorC input0, input1, input2; + TensorC output; + + input0.type_size = 4; // float32 + input0.size = 1000; + input0.data = (void *)0x10000000; + + input1.type_size = 4; + input1.size = 1000; + input1.data = (void *)0x10001000; + + input2.type_size = 4; + input2.size = 1000; + input2.data = (void *)0x10002000; + + output.type_size = 4; + output.size = 1000; + output.data = (void *)0x10003000; // 需要预先分配足够的内存 + + TensorC* input_tensors[3] = {&input0, &input1, &input2}; + + int index = 0; // 选择 input0 + + switchlayer_p(input_tensors, &output, index); + + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/tensor_scatter_add.rst.txt b/master/html/_sources/functionlib/dsplib/tensor_scatter_add.rst.txt new file mode 100644 index 0000000..f318e1d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensor_scatter_add.rst.txt @@ -0,0 +1,101 @@ +TensorScatterAdd +=================== + +在输入张量的指定位置执行加法操作。 +根据给定的 `indices` 和 `updates`,在输入张量中相应的位置将更新值加到原值上,生成新的输出张量。 + +.. math:: + + output[indices] = input + updates + +输入: + - **input** - 输入张量数据地址。 + - **input_shape** - 输入张量形状数组。 + - **input_rank** - 输入张量维度数。 + - **indices** - 指定更新位置的索引数组。 + - **updates** - 更新数据地址。 + - **num_unit** - 每个更新单元的长度。 + - **index_depth** - 索引深度(`indices` 的最后一维长度)。 + - **output_unit_offsets(int*, 可选)** - 单元偏移数组(仅私有版本使用)。 + - **strides(int*, 可选)** - 步长数组(仅私有版本使用)。 + - **core_mask(int, 可选)** - 核掩码(仅共享存储版本使用)。 + +输出: + - **output** - 输出张量数据地址,存放更新后的结果。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void fp_tensor_scatter_add_s(float *input, int *input_shape, int input_rank, int *indices, float *updates, float *output, int num_unit, int index_depth, int core_mask) +.. c:function:: void i32_tensor_scatter_add_s(int32_t *input, int *input_shape, int input_rank, int *indices, int32_t *updates, int32_t *output, int num_unit, int index_depth, int core_mask) +.. c:function:: void i16_tensor_scatter_add_s(int16_t *input, int *input_shape, int input_rank, int *indices, int16_t *updates, int16_t *output, int num_unit, int index_depth, int core_mask) +.. c:function:: void i8_tensor_scatter_add_s(int8_t *input, int *input_shape, int input_rank, int *indices, int8_t *updates, int8_t *output, int num_unit, int index_depth, int core_mask) +.. c:function:: void hp_tensor_scatter_add_s(half *input, int *input_shape, int input_rank, int *indices, half *updates, half *output, int num_unit, int index_depth, int core_mask) +.. c:function:: void dp_tensor_scatter_add_s(double *input, int *input_shape, int input_rank, int *indices, double *updates, double *output, int num_unit, int index_depth, int core_mask) +.. c:function:: void c64_tensor_scatter_add_s(float *input, int *input_shape, int input_rank, int *indices, float *updates, float *output, int num_unit, int index_depth, int core_mask) +.. c:function:: void c128_tensor_scatter_add_s(double *input, int *input_shape, int input_rank, int *indices, double *updates, double *output, int num_unit, int index_depth, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + // FT78NE 示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + float *output = (float *)0xA1000000; + int input_shape[2] = {4, 4}; + int input_rank = 2; + int indices[2] = {1, 2}; + float updates[1] = {3.14}; + int num_unit = 1; + int index_depth = 2; + int core_mask = 0xff; + fp_tensor_scatter_add_s(input, input_shape, input_rank, indices, updates, output, num_unit, index_depth, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_tensor_scatter_add_p(float *input, int *input_shape, int input_rank, int *indices, float *updates, float *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) +.. c:function:: void i32_tensor_scatter_add_p(int32_t *input, int *input_shape, int input_rank, int *indices, int32_t *updates, int32_t *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) +.. c:function:: void i16_tensor_scatter_add_p(int16_t *input, int *input_shape, int input_rank, int *indices, int16_t *updates, int16_t *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) +.. c:function:: void i8_tensor_scatter_add_p(int8_t *input, int *input_shape, int input_rank, int *indices, int8_t *updates, int8_t *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) +.. c:function:: void hp_tensor_scatter_add_p(half *input, int *input_shape, int input_rank, int *indices, half *updates, half *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) +.. c:function:: void dp_tensor_scatter_add_p(double *input, int *input_shape, int input_rank, int *indices, double *updates, double *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) +.. c:function:: void c64_tensor_scatter_add_p(float *input, int *input_shape, int input_rank, int *indices, float *updates, float *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) +.. c:function:: void c128_tensor_scatter_add_p(double *input, int *input_shape, int input_rank, int *indices, double *updates, double *output, int num_unit, int index_depth, int *output_unit_offsets, int *strides) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 示例 + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; + float *output = (float *)0x10010000; + int *output_unit_offsets = (int *)0x10020000; + int *strides = (int *)0x10030000; + int input_shape[2] = {4, 4}; + int input_rank = 2; + int indices[2] = {0, 1}; + float updates[1] = {2.71}; + int num_unit = 1; + int index_depth = 2; + fp_tensor_scatter_add_p(input, input_shape, input_rank, indices, updates, output, num_unit, index_depth, output_unit_offsets, strides); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/tensorarray.rst.txt b/master/html/_sources/functionlib/dsplib/tensorarray.rst.txt new file mode 100644 index 0000000..9a9794f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorarray.rst.txt @@ -0,0 +1,203 @@ +Tensorarray +================= + +张量数组操作,支持从张量数组中读取指定索引的张量,或将张量写入到张量数组的指定索引位置。该算子不区分数据类型,适用于所有数据类型。 + +**TensorArrayRead(读取操作):** + +从张量数组中读取指定索引的张量,并将其数据复制到输出。 + +.. math:: + + \text{output} = \text{tensors}[\text{index}] + +**TensorArrayWrite(写入操作):** + +将输入张量的数据复制到张量数组的指定索引位置。 + +.. math:: + + \text{tensors}[\text{index}] = \text{input} + +两个操作都会复制数据,复制大小为 `size * type_size` 字节。 + +输入(TensorArrayRead): + - **tensors** - 张量数组(Tensor** 类型),包含多个张量。 + - **index** - 读取的索引(int 类型),指定从 `tensors` 数组中读取哪个张量。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出(TensorArrayRead): + - **output** - 输出张量(Tensor* 类型),包含复制后的数据。输出张量的 `size` 和 `type_size` 应与选中的输入张量相同。 + +输入(TensorArrayWrite): + - **input** - 输入张量(Tensor* 类型),待写入的数据。 + - **index** - 写入的索引(int 类型),指定写入到 `tensors` 数组的哪个位置。 + - **tensors** - 张量数组(Tensor** 类型),目标数组。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出(TensorArrayWrite): + - **tensors[index]** - 张量数组指定位置的数据会被更新。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分数据类型,适用于所有数据类型 + - 算子会复制数据,输入和输出张量数据独立 + - 调用前需要确保目标内存空间足够大(至少 `size * type_size` 字节) + - 读取和写入时,涉及的张量的 `size` 和 `type_size` 应匹配 + +**共享存储版本:** + +**TensorArrayRead(读取操作):** + +.. c:function:: void tensorarrayread_s(Tensor** tensors, int index, Tensor* output, int core_mask) + +**TensorArrayWrite(写入操作):** + +.. c:function:: void tensorarraywrite_s(Tensor* input, int index, Tensor** tensors, int core_mask) + +**C调用示例(TensorArrayRead):** + +.. code-block:: c + :linenos: + :emphasize-lines: 34 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + Tensor tensor0, tensor1, tensor2; + Tensor output; + + // 初始化张量数组中的张量 + tensor0.type_size = 4; // float32 + tensor0.size = 1000; + tensor0.data = (void *)0xA0000000; + + tensor1.type_size = 4; + tensor1.size = 1000; + tensor1.data = (void *)0xA1000000; + + tensor2.type_size = 4; + tensor2.size = 1000; + tensor2.data = (void *)0xA2000000; + + // 初始化输出张量 + output.type_size = 4; + output.size = 1000; + output.data = (void *)0xB0000000; // 需要预先分配足够的内存 + + // 创建张量数组 + Tensor* tensors[3] = {&tensor0, &tensor1, &tensor2}; + + int index = 1; // 读取 tensor1 + int core_mask = 0xff; + + tensorarrayread_s(tensors, index, &output, core_mask); + + // 此时 output.data 包含 tensor1.data 的副本 + + return 0; + } + +**C调用示例(TensorArrayWrite):** + +.. code-block:: c + :linenos: + :emphasize-lines: 30 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + Tensor input; + Tensor tensor0, tensor1, tensor2; + + // 初始化输入张量 + input.type_size = 4; // float32 + input.size = 1000; + input.data = (void *)0xA0000000; + + // 初始化张量数组中的张量(需要预先分配内存) + tensor0.type_size = 4; + tensor0.size = 1000; + tensor0.data = (void *)0xB0000000; + + tensor1.type_size = 4; + tensor1.size = 1000; + tensor1.data = (void *)0xB0100000; + + tensor2.type_size = 4; + tensor2.size = 1000; + tensor2.data = (void *)0xB0200000; + + // 创建张量数组 + Tensor* tensors[3] = {&tensor0, &tensor1, &tensor2}; + + int index = 1; // 写入到 tensor1 的位置 + int core_mask = 0xff; + + tensorarraywrite_s(&input, index, tensors, core_mask); + + // 此时 tensors[1]->data 包含 input.data 的副本 + + return 0; + } + +**私有存储版本:** + +**TensorArrayRead(读取操作):** + +.. c:function:: void tensorarrayread_p(Tensor** tensors, int index, Tensor* output) + +**TensorArrayWrite(写入操作):** + +.. c:function:: void tensorarraywrite_p(Tensor* input, int index, Tensor** tensors) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + Tensor tensor0, tensor1, tensor2; + Tensor output; + + tensor0.type_size = 4; // float32 + tensor0.size = 1000; + tensor0.data = (void *)0x10000000; + + tensor1.type_size = 4; + tensor1.size = 1000; + tensor1.data = (void *)0x10001000; + + tensor2.type_size = 4; + tensor2.size = 1000; + tensor2.data = (void *)0x10002000; + + output.type_size = 4; + output.size = 1000; + output.data = (void *)0x10003000; // 需要预先分配足够的内存 + + Tensor* tensors[3] = {&tensor0, &tensor1, &tensor2}; + + int index = 0; // 读取 tensor0 + + tensorarrayread_p(tensors, index, &output); + + return 0; + } + + diff --git a/master/html/_sources/functionlib/dsplib/tensorarrayread.rst.txt b/master/html/_sources/functionlib/dsplib/tensorarrayread.rst.txt new file mode 100644 index 0000000..4b7eec6 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorarrayread.rst.txt @@ -0,0 +1,123 @@ +Tensorarrayread +================= + +从张量数组中读取指定索引的张量,并将其数据复制到输出。该算子不区分数据类型,适用于所有数据类型。 + +.. math:: + + \text{output\_data} = \text{handle\_data}[\text{index}] + +该算子会将 `handle_data[index]` 指向的数据复制到 `output_data` 中,复制的大小为 `handle_size[index]` 字节。 + +输入: + - **handle_data** - 张量数组的数据指针数组(void** 类型),每个元素指向一个张量的数据。 + - **handle_size** - 每个张量的大小数组(int* 类型),`handle_size[i]` 表示 `handle_data[i]` 指向的数据大小(字节)。 + - **index** - 读取的索引(int 类型),指定从 `handle_data` 数组中读取哪个张量。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出: + - **output_data** - 输出数据指针(void* 类型),包含复制后的数据。 + - **output_size** - 输出大小指针(int* 类型),指向存储输出大小的变量。调用后,`*output_size` 会被设置为 `handle_size[index]`。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int16, int32, cplx64 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + - 算子会复制数据,输出数据与输入数据独立 + - 调用前需要确保 `output_data` 指向的内存空间足够大(至少 `handle_size[index]` 字节) + - `index` 必须在 `handle_data` 数组的有效范围内 + +**共享存储版本:** + +.. c:function:: void fp_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) +.. c:function:: void hp_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) +.. c:function:: void dp_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) +.. c:function:: void i8_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) +.. c:function:: void i16_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) +.. c:function:: void i32_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) +.. c:function:: void c64_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) +.. c:function:: void c128_tensorarrayread_s(void** handle_data, int* handle_size, int index, void* output_data, int* output_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 25 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + // 张量数组包含3个张量 + float *data0 = (float *)0xA0000000; // 第0个张量的数据 + float *data1 = (float *)0xA1000000; // 第1个张量的数据 + float *data2 = (float *)0xA2000000; // 第2个张量的数据 + + // 每个张量的大小(字节) + int sizes[3] = {1000 * sizeof(float), 1000 * sizeof(float), 1000 * sizeof(float)}; + + // 创建数据指针数组 + void* handle_data[3] = {data0, data1, data2}; + + // 输出数据 + float *output_data = (float *)0xB0000000; // 需要预先分配足够的内存 + int output_size; // 输出大小,调用后会被设置 + + int index = 1; // 读取第1个张量 + int core_mask = 0xff; + + fp_tensorarrayread_s(handle_data, sizes, index, output_data, &output_size, core_mask); + + // 此时 output_data 包含 data1 的副本 + // output_size == sizes[1] == 1000 * sizeof(float) + + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) +.. c:function:: void hp_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) +.. c:function:: void dp_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) +.. c:function:: void i8_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) +.. c:function:: void i16_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) +.. c:function:: void i32_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) +.. c:function:: void c64_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) +.. c:function:: void c128_tensorarrayread_p(void** handle_data, int* handle_size, int index, void* output_data, int* output_size) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 20 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + float *data0 = (float *)0x10000000; + float *data1 = (float *)0x10001000; + float *data2 = (float *)0x10002000; + + int sizes[3] = {1000 * sizeof(float), 1000 * sizeof(float), 1000 * sizeof(float)}; + + void* handle_data[3] = {data0, data1, data2}; + + float *output_data = (float *)0x10003000; // 需要预先分配足够的内存 + int output_size; + + int index = 0; // 读取第0个张量 + + fp_tensorarrayread_p(handle_data, sizes, index, output_data, &output_size); + + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/tensorarraywrite.rst.txt b/master/html/_sources/functionlib/dsplib/tensorarraywrite.rst.txt new file mode 100644 index 0000000..a627f0e --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorarraywrite.rst.txt @@ -0,0 +1,41 @@ +TensorarrayWrite +================= + +将张量写入到张量数组的指定索引位置。该算子不区分数据类型,适用于所有数据类型。 + +.. math:: + + \text{output\_data} = \text{handle\_data}[\text{index}] + +该算子会将 `handle_data[index]` 指向的数据复制到 `output_data` 中,复制的大小为 `handle_size[index]` 字节。 + +输入: + - **handle_data** - 张量数组的数据指针数组(void** 类型),每个元素指向一个张量的数据。 + - **handle_size** - 每个张量的大小数组(int* 类型),`handle_size[i]` 表示 `handle_data[i]` 指向的数据大小(字节)。 + - **index** - 读取的索引(int 类型),指定从 `handle_data` 数组中读取哪个张量。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出: + - **output_data** - 输出数据指针(void* 类型),包含复制后的数据。 + - **output_size** - 输出大小指针(int* 类型),指向存储输出大小的变量。调用后,`*output_size` 会被设置为 `handle_size[index]`。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int16, int32, cplx64 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + - 算子会复制数据,输出数据与输入数据独立 + - 调用前需要确保 `output_data` 指向的内存空间足够大(至少 `handle_size[index]` 字节) + - `index` 必须在 `handle_data` 数组的有效范围内 + +**共享存储版本:** + +见TensorArrayRead。 + + +**私有存储版本:** + +见TensorArrayRead。 + diff --git a/master/html/_sources/functionlib/dsplib/tensorlistfromtensor.rst.txt b/master/html/_sources/functionlib/dsplib/tensorlistfromtensor.rst.txt new file mode 100644 index 0000000..e2e56cd --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorlistfromtensor.rst.txt @@ -0,0 +1,121 @@ +Tensorlistfromtensor +====================== + +将输入张量按照第一个维度拆分成多个张量。假设输入张量的形状为 `[N1, N2, N3, ...]`,该算子将输入张量拆分成 `N1` 个张量,每个输出张量的形状为 `[N2, N3, ...]`。 + +.. math:: + + \text{output\_tensors}[i] = \text{input\_tensor}[\text{slice\_at\_dim0} = i] + +其中 `i = 0, 1, \ldots, N1-1`,每个输出张量包含 `N1` 个切片中的一个。 + + +输入: + - **input_tensor_values** - 输入张量的数据指针,大小为 `input_tensor_total_elements` 个元素。 + - **input_tensor_shape** - 输入张量的形状数组(int* 类型),`input_tensor_shape[0]` 表示第一个维度的大小(即输出张量的数量)。 + - **input_tensor_total_elements** - 输入张量的总元素数(int 类型)。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出: + - **output_tensors** - 输出张量数组(指针数组),大小为 `input_tensor_shape[0]`。每个元素 `output_tensors[i]` 指向第 `i` 个输出张量的数据。每个输出张量的大小为 `input_tensor_total_elements / input_tensor_shape[0]` 个元素。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp32, int8, int16, int32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + - 算子会复制数据,输出张量与输入张量数据独立 + - 调用前需要确保所有 `output_tensors[i]` 指向的内存空间足够大(至少 `input_tensor_total_elements / input_tensor_shape[0]` 个元素) + - 输出张量的数量等于 `input_tensor_shape[0]` + - 每个输出张量的元素数为 `input_tensor_total_elements / input_tensor_shape[0]` + +**共享存储版本:** + +.. c:function:: void i8_tensorlistfromtensor_s(int8_t* input_tensor_values, int* input_tensor_shape, int8_t** output_tensors, int input_tensor_total_elements, int core_mask) +.. c:function:: void i16_tensorlistfromtensor_s(int16_t* input_tensor_values, int* input_tensor_shape, int16_t** output_tensors, int input_tensor_total_elements, int core_mask) +.. c:function:: void i32_tensorlistfromtensor_s(int32_t* input_tensor_values, int* input_tensor_shape, int32_t** output_tensors, int input_tensor_total_elements, int core_mask) +.. c:function:: void hp_tensorlistfromtensor_s(half* input_tensor_values, int* input_tensor_shape, half** output_tensors, int input_tensor_total_elements, int core_mask) +.. c:function:: void fp_tensorlistfromtensor_s(float* input_tensor_values, int* input_tensor_shape, float** output_tensors, int input_tensor_total_elements, int core_mask) +.. c:function:: void dp_tensorlistfromtensor_s(double* input_tensor_values, int* input_tensor_shape, double** output_tensors, int input_tensor_total_elements, int core_mask) +.. c:function:: void c64_tensorlistfromtensor_s(float* input_tensor_values, int* input_tensor_shape, float** output_tensors, int input_tensor_total_elements, int core_mask) +.. c:function:: void c128_tensorlistfromtensor_s(double* input_tensor_values, int* input_tensor_shape, double** output_tensors, int input_tensor_total_elements, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 26 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + // 输入张量形状 [3, 4, 5],总元素数 = 3 * 4 * 5 = 60 + // 输出3个张量,每个形状 [4, 5],元素数 = 60 / 3 = 20 + + int input_tensor_shape[] = {3, 4, 5}; + int input_tensor_total_elements = 3 * 4 * 5; // 60 + + // 输入张量数据 + float *input_tensor_values = (float *)0xA0000000; + // input_tensor_values 包含 60 个 float 元素 + + // 输出张量数组(需要预先分配内存) + float *output0 = (float *)0xB0000000; // 第0个输出张量,20个元素 + float *output1 = (float *)0xB0100000; // 第1个输出张量,20个元素 + float *output2 = (float *)0xB0200000; // 第2个输出张量,20个元素 + + float* output_tensors[3] = {output0, output1, output2}; + + int core_mask = 0xff; + + fp_tensorlistfromtensor_s(input_tensor_values, input_tensor_shape, + output_tensors, input_tensor_total_elements, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_tensorlistfromtensor_p(int8_t* input_tensor_values, int* input_tensor_shape, int8_t** output_tensors, int input_tensor_total_elements) +.. c:function:: void i16_tensorlistfromtensor_p(int16_t* input_tensor_values, int* input_tensor_shape, int16_t** output_tensors, int input_tensor_total_elements) +.. c:function:: void i32_tensorlistfromtensor_p(int32_t* input_tensor_values, int* input_tensor_shape, int32_t** output_tensors, int input_tensor_total_elements) +.. c:function:: void hp_tensorlistfromtensor_p(half* input_tensor_values, int* input_tensor_shape, half** output_tensors, int input_tensor_total_elements) +.. c:function:: void fp_tensorlistfromtensor_p(float* input_tensor_values, int* input_tensor_shape, float** output_tensors, int input_tensor_total_elements) +.. c:function:: void dp_tensorlistfromtensor_p(double* input_tensor_values, int* input_tensor_shape, double** output_tensors, int input_tensor_total_elements) +.. c:function:: void c64_tensorlistfromtensor_p(float* input_tensor_values, int* input_tensor_shape, float** output_tensors, int input_tensor_total_elements) +.. c:function:: void c128_tensorlistfromtensor_p(double* input_tensor_values, int* input_tensor_shape, double** output_tensors, int input_tensor_total_elements) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 18-19 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + int input_tensor_shape[] = {3, 4, 5}; + int input_tensor_total_elements = 3 * 4 * 5; + + float *input_tensor_values = (float *)0x10000000; + + float *output0 = (float *)0x10010000; + float *output1 = (float *)0x10011000; + float *output2 = (float *)0x10012000; + + float* output_tensors[3] = {output0, output1, output2}; + + fp_tensorlistfromtensor_p(input_tensor_values, input_tensor_shape, + output_tensors, input_tensor_total_elements); + + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/tensorlistgetitem.rst.txt b/master/html/_sources/functionlib/dsplib/tensorlistgetitem.rst.txt new file mode 100644 index 0000000..317bdaf --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorlistgetitem.rst.txt @@ -0,0 +1,83 @@ +Tensorlistgetitem +================= + +从张量列表中获取指定索引的张量数据。该算子将输入张量列表中的特定元素复制到输出张量中,支持多种数据类型。 + +.. math:: + + \text{output\_tensor} = \text{input\_tensor\_list}[\text{index}] + +其中 `input_tensor_list` 是输入的张量列表,`index` 是要获取的元素索引,`output_tensor` 是输出张量。 + +输入: + - **src** - 输入张量指针,指向要获取的张量数据。 + - **dst** - 输出张量指针,用于存储获取的张量数据。 + - **data_type** - 数据类型对应的字节数,如果为0,表示kTypeUnknown(未知类型)。 + - **length** - 元素个数,默认dst和src元素个数相等。 + - **core_mask** - 核掩码(int),仅共享存储版本需要。 + +输出: + - **dst** - 输出张量指针,存储从输入张量列表获取的数据。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分具体的数据类型,数据类型信息通过data_type参数传递 + - 当data_type为0(kTypeUnknown)时,算子会将输出张量的内存清零 + - 当data_type不为0时,算子会按字节复制数据 + - 调用前需要确保dst指向的内存空间足够大(至少length个元素) + +**共享存储版本:** + +.. c:function:: void tensorlistgetitem_s(void *src, void *dst, int data_type, int length, int core_mask) + +**C调用示例(共享存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + float *src = (float *)0xA0000000; // 输入张量数据 + float *dst = (float *)0xB0000000; // 输出张量数据 + int data_type = 4; // float类型,4字节 + int length = 10; // 10个元素 + int core_mask = 0xff; // 核掩码 + + // 调用共享存储版本的函数 + tensorlistgetitem_s(src, dst, data_type, length, core_mask); + + return 0; + } + +**共享存储版本:** + +.. c:function:: void tensorlistgetitem_p(void *src, void *dst, int data_type, int length) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main(int argc, char* argv[]) { + void *src = (void *)0x10000000; + void *dst = (void *)0x10010000; + int data_type = 0; // kTypeUnknown + int length = 10; // 10个元素(以字节为单位) + + // 当data_type为0时,dst将被清零 + tensorlistgetitem_p(src, dst, data_type, length); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/tensorlistreserve.rst.txt b/master/html/_sources/functionlib/dsplib/tensorlistreserve.rst.txt new file mode 100644 index 0000000..d2b3de2 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorlistreserve.rst.txt @@ -0,0 +1,64 @@ +Tensorlistreserve +================= + +为张量列表(TensorList)预分配指定数量和形状的张量空间。该算子用于初始化TensorList结构,设置其内部张量的形状信息和TensorList本身的形状。 + +该算子不区分具体的数据类型,主要负责设置TensorList的元数据信息。 + +输入: + - **num_elements** - TensorList中包含的张量数量(int类型)。 + - **element_shape** - 每个张量的形状数组(int*类型),表示TensorList中每个张量的维度信息。 + - **shape_size** - 每个张量的形状大小(int类型),即element_shape数组的长度。 + - **tensor_list_c_element_shape_size** - 输出参数,用于存储TensorList内部每个张量的形状大小(int*类型)。 + - **tensor_list_c_element_shape** - 输出参数,用于存储TensorList内部每个张量的形状(int*类型)。 + - **tensor_c_shape_size** - 输出参数,用于存储TensorList本身的形状大小(int*类型)。 + - **tensor_c_shape** - 输出参数,用于存储TensorList本身的形状(int*类型)。 + +输出: + - **tensor_list_c_element_shape_size** - 被设置为shape_size的值。 + - **tensor_list_c_element_shape** - 被填充为element_shape的值。 + - **tensor_c_shape_size** - 被设置为1,表示TensorList本身是一个1维结构。 + - **tensor_c_shape** - 被设置为num_elements,表示TensorList包含num_elements个张量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分具体的数据类型,主要操作TensorList的元数据 + - TensorList中的每个张量必须具有相同的形状(element_shape) + - 调用前需要确保tensor_list_c_element_shape和tensor_c_shape指向的内存空间足够大 + +**函数定义:** + +.. c:function:: void Tensorlistreserve(int num_elements, int *element_shape, int shape_size, int *tensor_list_c_element_shape_size, int *tensor_list_c_element_shape, int *tensor_c_shape_size, int *tensor_c_shape) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17-20 + + #include + #include + + int main(int argc, char* argv[]) { + // 参数设置 + int num_elements = 3; // TensorList包含3个张量 + int element_shape[] = {4, 5}; // 每个张量的形状为[4, 5] + int shape_size = 2; // element_shape数组长度为2 + + // 输出参数 + int tensor_list_c_element_shape_size; + int tensor_list_c_element_shape[2]; // 空间大小应至少为shape_size + int tensor_c_shape_size; + int tensor_c_shape; + + // 调用算子 + Tensorlistreserve(num_elements, element_shape, shape_size, + &tensor_list_c_element_shape_size, + tensor_list_c_element_shape, + &tensor_c_shape_size, + &tensor_c_shape); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/tensorlistsetitem.rst.txt b/master/html/_sources/functionlib/dsplib/tensorlistsetitem.rst.txt new file mode 100644 index 0000000..7b6765c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorlistsetitem.rst.txt @@ -0,0 +1,114 @@ +Tensorlistsetitem +================= + +将输入的张量(item)替换到张量列表(TensorList)中指定索引位置。该算子创建一个新的张量列表,其中在指定索引位置的张量被替换为输入的新张量,而其他位置的张量保持不变。 + +该算子不区分具体的数据类型,通过copy_size参数指定每个张量的数据量大小。 + +输入: + - **in_data** - 输入张量列表指针(void**类型),指向原始张量列表中的各个张量数据。 + - **in_item** - 输入张量指针(void*类型),表示要替换到张量列表中的新张量数据。 + - **index** - 要替换的张量在列表中的索引位置(int类型)。 + - **tensorlist_size** - 张量列表中包含的张量个数(int类型)。 + - **copy_size** - 每个张量的数据量大小数组(int*类型),以字节为单位,表示每个张量需要复制的字节数。 + - **core_mask** - 核掩码(int类型),仅共享存储版本需要。 + +输出: + - **out_data** - 输出张量列表指针(void**类型),存储替换操作后的张量列表,其中index位置的张量被替换为in_item。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分具体的数据类型,通过copy_size参数控制复制的字节数 + - 调用前需要确保out_data指向的内存空间足够大,能够存储所有张量数据 + - 索引index必须在有效范围内(0 <= index < tensorlist_size) + - 算子会复制所有张量数据,输出张量列表与输入张量列表数据独立 + +**共享存储版本:** + +.. c:function:: void tensorlistsetitem_s(void** in_data, void* in_item, void** out_data, int index, int tensorlist_size, int* copy_size, int core_mask) + +**C调用示例(共享存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 31 + + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + int tensorlist_size = 3; // 张量列表包含3个张量 + int index = 1; // 替换索引为1的张量 + + // 输入张量列表 + float *in_tensor0 = (float *)0xA0000000; + float *in_tensor1 = (float *)0xA0010000; + float *in_tensor2 = (float *)0xA0020000; + void* in_data[3] = {in_tensor0, in_tensor1, in_tensor2}; + + // 输入item + float *in_item = (float *)0xA0030000; + + // 输出张量列表 + float *out_tensor0 = (float *)0xB0000000; + float *out_tensor1 = (float *)0xB0010000; + float *out_tensor2 = (float *)0xB0020000; + void* out_data[3] = {out_tensor0, out_tensor1, out_tensor2}; + + // 每个张量的数据量大小 + int copy_size[3] = {40, 40, 40}; + + // 核掩码 + int core_mask = 0xff; + + // 调用共享存储版本的函数 + tensorlistsetitem_s(in_data, in_item, out_data, index, tensorlist_size, copy_size, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void tensorlistsetitem_p(void** in_data, void* in_item, void** out_data, int index, int tensorlist_size, int* copy_size) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 28 + + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + int tensorlist_size = 3; // 张量列表包含3个张量 + int index = 1; // 替换索引为1的张量 + + // 输入张量列表(假设已初始化) + float *in_tensor0 = (float *)0x10000000; // 第一个张量 + float *in_tensor1 = (float *)0x10001000; // 第二个张量(将被替换) + float *in_tensor2 = (float *)0x10002000; // 第三个张量 + void* in_data[3] = {in_tensor0, in_tensor1, in_tensor2}; + + // 输入item(新张量数据) + float *in_item = (float *)0x10003000; + + // 输出张量列表(需要预先分配内存) + float *out_tensor0 = (float *)0x10004000; + float *out_tensor1 = (float *)0x10005000; + float *out_tensor2 = (float *)0x10006000; + void* out_data[3] = {out_tensor0, out_tensor1, out_tensor2}; + + // 每个张量的数据量大小(假设每个张量有10个float元素,每个float 4字节) + int copy_size[3] = {40, 40, 40}; // 40字节 = 10 * 4 + + // 调用私有存储版本的函数 + tensorlistsetitem_p(in_data, in_item, out_data, index, tensorlist_size, copy_size); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/tensorliststack.rst.txt b/master/html/_sources/functionlib/dsplib/tensorliststack.rst.txt new file mode 100644 index 0000000..50d94d3 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tensorliststack.rst.txt @@ -0,0 +1,116 @@ +Tensorliststack +=============== + +将多个张量(Tensor)堆叠成一个更大的张量。此算子可以处理不同数据类型的张量,将它们按顺序拼接成一个连续的内存块。 + +.. math:: + + \text{output\_data} = [\text{tensor}_1, \text{tensor}_2, \ldots, \text{tensor}_n] + +其中每个张量的数据类型和元素数量可以不同。 + +输入: + - **tensor_num** - 张量数量,tensor_num > 0 + - **tensor_element_nums** - 每个张量的元素数量(int* 类型) + - **tensor_data_type** - 每个张量元素的数据类型,以字节数表示 + - **tensor_data** - 每个张量数据的起始地址(void** 类型) + - **output_data** - 输出结果的数组起始位置(void* 类型) + - **unknown_type_offset** - 未知类型数据在输出结果中的偏移量 + - **core_mask** - 核掩码(int),仅共享存储版本需要 + +输出: + - **output_data** - 堆叠后的张量数据,按输入顺序连续存储 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子不区分具体的数据类型,数据类型信息通过tensor_data_type参数传递 + - 当tensor_data_type[i]为0(kTypeUnknown)时,算子会将输出内存清零 + - 当tensor_data_type[i]不为0时,算子会按字节复制数据 + - 调用前需要确保output_data指向的内存空间足够大以容纳所有张量数据 + - TensorList中不同的Tensor数据类型可能不同,类型信息已经在算子中包含 + +**共享存储版本:** + +.. c:function:: void tensorliststack_s(int tensor_num, int *tensor_element_nums, int *tensor_data_type, void **tensor_data, void *output_data, int unknown_type_offset, int core_mask) + +**C调用示例(共享存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 27-28 + + //FT78NE示例 + #include + #include + #include + + int main(int argc, char* argv[]) { + // 假设在DDR空间 + int tensor_num = 2; + + // 每个张量的元素数量 + int tensor_element_nums[] = {3, 2}; + + // 每个张量的数据类型(以字节数表示) + int tensor_data_type[] = {4, 8}; // 32位int(4字节), 64位double(8字节) + + // 张量数据 + int *tensor1 = (int *)0xA0000000; + double *tensor2 = (double *)0xA0100000; + + void *tensor_data[] = {tensor1, tensor2}; + + void *output_data = (void *)0xB0000000; // 输出数据 + int unknown_type_offset = 0; + int core_mask = 0xff; + + // 调用共享存储版本的函数 + tensorliststack_s(tensor_num, tensor_element_nums, tensor_data_type, + tensor_data, output_data, unknown_type_offset, core_mask); + + return 0; + } + +**私有存储版本:** + +.. c:function:: void tensorliststack_p(int tensor_num, int *tensor_element_nums, int *tensor_data_type, void **tensor_data, void *output_data, int unknown_type_offset) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 26-27 + + //FT78NE示例 + #include + #include + #include + + int main(int argc, char* argv[]) { + // 假设在L2空间 + int tensor_num = 2; + + // 每个张量的元素数量 + int tensor_element_nums[] = {3, 2}; + + // 每个张量的数据类型(以字节数表示) + int tensor_data_type[] = {4, 8}; // 32位int(4字节), 64位double(8字节) + + // 张量数据 + int *tensor1 = (int *)0x10000000; + double *tensor2 = (double *)0x10100000; + + void *tensor_data[] = {tensor1, tensor2}; + + void *output_data = (void *)0x10200000; // 输出数据 + int unknown_type_offset = 0; + + // 调用私有存储版本的函数 + tensorliststack_p(tensor_num, tensor_element_nums, tensor_data_type, + tensor_data, output_data, unknown_type_offset); + + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/tile.rst.txt b/master/html/_sources/functionlib/dsplib/tile.rst.txt new file mode 100644 index 0000000..5f13c1d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tile.rst.txt @@ -0,0 +1,98 @@ +Tile +================= + +沿着指定的维度 ``tile_dim``,将输入张量的基础块重复平铺 ``tile_num`` 次。该算子通过“加倍拷贝法”利用 DMA 传输实现高效的数据复制。 + +.. math:: + + \text{对于输入张量,其在维度 } tile\_dim \text{ 及其之后维度的连续块大小为 } stride \text{ 个元素。} \\ + \text{算子将该块在输出中连续复制 } tile\_num \text{ 次,构造出平铺后的维度效果。} + +输入: + - **input** - 输入张量数据地址。 + - **output** - 输出张量数据地址。 + - **input_shape** - 输入张量的形状数组地址。 + - **tile_dim** - 执行平铺操作的维度索引(0 到 ndim-1)。 + - **tile_num** - 在指定维度上平铺的次数。 + - **stride** - 平铺块的长度。通常定义为从 ``tile_dim`` 维度到最后一个维度所有形状值的乘积。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 填充后的平铺结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + - 对于复数类型(cplx64/cplx128),``stride`` 依然表示复数元素的个数,算子内部会自动处理双倍宽度的内存拷贝。 + - 算子内部使用 DMA 传输,能够有效提升 DDR 到 DDR 的拷贝效率。 + +**共享存储版本:** + +.. c:function:: void i8_tile_s(int8_t* input, int8_t* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) +.. c:function:: void i16_tile_s(int16_t* input, int16_t* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) +.. c:function:: void i32_tile_s(int32_t* input, int32_t* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) +.. c:function:: void hp_tile_s(half* input, half* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) +.. c:function:: void fp_tile_s(float* input, float* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) +.. c:function:: void dp_tile_s(double* input, double* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) +.. c:function:: void c64_tile_s(float* input, float* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) +.. c:function:: void c128_tile_s(double* input, double* output, int *input_shape, int tile_dim, int tile_num, int stride, int core_mask) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 14 + + // FT78NE 示例:多核并行平铺 + #include + #include "78NE/utils.h" + + int main() { + float *input = (float *)0xA0000000; + float *output = (float *)0xB0000000; + int input_shape[] = {16, 20, 16, 24}; + int tile_dim = 2; + int tile_num = 10; + int stride = 16 * 24; // tile_dim[2]*input_shape[3] + int core_mask = 0b1011; + + fp_tile_s(input, output, input_shape, tile_dim, tile_num, stride, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_tile_p(int8_t* input, int8_t* output, int *input_shape, int tile_dim, int tile_num, int stride) +.. c:function:: void i16_tile_p(int16_t* input, int16_t* output, int *input_shape, int tile_dim, int tile_num, int stride) +.. c:function:: void i32_tile_p(int32_t* input, int32_t* output, int *input_shape, int tile_dim, int tile_num, int stride) +.. c:function:: void hp_tile_p(half* input, half* output, int *input_shape, int tile_dim, int tile_num, int stride) +.. c:function:: void fp_tile_p(float* input, float* output, int *input_shape, int tile_dim, int tile_num, int stride) +.. c:function:: void dp_tile_p(double* input, double* output, int *input_shape, int tile_dim, int tile_num, int stride) +.. c:function:: void c64_tile_p(float* input, float* output, int *input_shape, int tile_dim, int tile_num, int stride) +.. c:function:: void c128_tile_p(double* input, double* output, int *input_shape, int tile_dim, int tile_num, int stride) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + // MT7004 示例:单核私有空间平铺 + #include + + int main() { + float *input = (float *)0x10000000; + float *output = (float *)0x10010000; + int input_shape[] = {4, 10, 8, 12}; + int tile_dim = 2; + int tile_num = 5; + int stride = 8 * 12; + + fp_tile_p(input, output, input_shape, tile_dim, tile_num, stride); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/topkfusion.rst.txt b/master/html/_sources/functionlib/dsplib/topkfusion.rst.txt new file mode 100644 index 0000000..13e89bc --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/topkfusion.rst.txt @@ -0,0 +1,139 @@ +TopkFusion +================= + + +在指定维度上选取 Top-K 个最大值,并同时输出对应的索引。 + +该算子在融合实现中同时完成: + - Top-K 值计算 + - 索引提取 + - 可选的结果排序 + +其计算过程为: +在指定维度上,对每个 slice 进行排序或选择,输出前 ``K`` 个最大元素及其原始索引。 + +当 ``sorted = false`` 时,仅保证输出为 Top-K 元素集合, +不保证其在输出中的顺序;当 ``sorted = true`` 时, +输出结果按值从大到小排序。 + + +输入: + - **input** - 输入张量的数据地址。 + 数据类型需与所调用的 TopkFusion 接口类型一致,例如:``fp_*`` 接口对应 ``float*``,``hp_*`` 接口对应 ``half*``,``i16_*`` 接口对应 ``int16_t*``,``i32_*`` 接口对应 ``int32_t*`` + + - **output** - Top-K 值输出张量的数据地址。 + 数据类型与 ``input`` 保持一致。 + + - **output_index** - Top-K 索引输出张量的数据地址,类型固定为 ``int32_t*``。 + + - **parameter** - Top-K 参数结构体指针 ``TopkParameter*``,其定义如下: + + .. code-block:: c + + typedef struct TopkParameter { + // primitive parameter + OpParameter op_parameter_; + int k_; // 需要选取的 Top-K 个数 + int axis_; // 执行 Top-K 的维度 + bool sorted_; // 是否对 Top-K 结果按值排序 + + // other parameter + int dim_size_; // axis 维度长度 + int outer_loop_num_; // axis 之前维度展开后的循环次数 + int inner_loop_num_; // axis 之后维度展开后的循环次数 + void *topk_node_list_; // 临时 TopK 节点缓冲区 + } TopkParameter; + + - **core_mask** - 核掩码(仅共享存储版本使用)。 + + +输出: + - **output** - Top-K 值输出张量的数据地址。 + - **output_index** - Top-K 索引输出张量的数据地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持的数据类型: + fp32, fp64, int8, int16, int32 + - MT7004 支持的数据类型: + fp16, fp32, int16, int32 + - ``output`` 与 ``output_index`` 的布局与输入张量保持一致,仅在 Top-K 维度长度变为 ``K`` + - TopkFusion 算子内部使用 ``void*`` 进行类型复用, + 但调用时 ``input`` 与 ``output`` 的实际数据类型 + 必须与所选接口前缀严格一致。 + + + +**共享存储版本:** + +.. c:function:: void fp_topk_fusion_s(void* input, void* output, int32_t* output_index, TopkParameter* parameter, int core_mask) +.. c:function:: void hp_topk_fusion_s(void* input, void* output, int32_t* output_index, TopkParameter* parameter, int core_mask) +.. c:function:: void i16_topk_fusion_s(void* input, void* output, int32_t* output_index, TopkParameter* parameter, int core_mask) +.. c:function:: void i32_topk_fusion_s(void* input, void* output, int32_t* output_index, TopkParameter* parameter, int core_mask) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 19 + + // FT78NE 多核示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + float *output = (float *)0xB0000000; + int32_t *output_index = (int32_t *)0xB1000000; + + TopkParameter param; + param.k_ = 5; + param.dim_size_ = 100; + param.outer_loop_num_ = 1; + param.inner_loop_num_ = 1; + param.sorted_ = 1; + + int core_mask = 0xff; + + fp_topk_fusion_s(input, output, output_index, ¶m, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_topk_fusion_p(void* input, void* output, int32_t* output_index, TopkParameter* parameter) +.. c:function:: void hp_topk_fusion_p(void* input, void* output, int32_t* output_index, TopkParameter* parameter) +.. c:function:: void i16_topk_fusion_p(void* input, void* output, int32_t* output_index, TopkParameter* parameter) +.. c:function:: void i32_topk_fusion_p(void* input, void* output, int32_t* output_index, TopkParameter* parameter) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + // MT7004 单核示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input = (half *)0x10000000; + half *output = (half *)0x10010000; + int32_t *output_index = (int32_t *)0x10020000; + + TopkParameter param; + param.k_ = 3; + param.dim_size_ = 64; + param.outer_loop_num_ = 1; + param.inner_loop_num_ = 1; + param.sorted_ = 0; + + hp_topk_fusion_p(input, output, output_index, ¶m); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/transpose.rst.txt b/master/html/_sources/functionlib/dsplib/transpose.rst.txt new file mode 100644 index 0000000..02dd2ca --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/transpose.rst.txt @@ -0,0 +1,92 @@ +Transpose +================= + +对输入数组按照指定维度顺序(perm)进行转置操作,并输出结果数组。 + +输入: + - **in_data** - 输入数据地址。 + - **num_axes** - 数据维度数。 + - **output_shape** - 输出形状数组。 + - **perm** - 转置维度顺序数组。 + - **strides** - 输入数据每维步长。 + - **out_strides** - 输出数据每维步长。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **out_data** - 转置结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持fp, dp, int8, int16, int32, clx64, cplx128 + - MT7004 支持hp, fp, i16, i32, cplx64 + +**共享存储版本:** + +.. c:function:: void fp_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const float* in_data, float* out_data, int core_mask) +.. c:function:: void hp_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const half* in_data, half* out_data, int core_mask) +.. c:function:: void dp_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const double* in_data, double* out_data, int core_mask) +.. c:function:: void i8_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const int8_t* in_data, int8_t* out_data, int core_mask) +.. c:function:: void i16_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const int16_t* in_data, int16_t* out_data, int core_mask) +.. c:function:: void i32_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const int* in_data, int* out_data, int core_mask) +.. c:function:: void c64_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const float* in_data, float* out_data, int core_mask) +.. c:function:: void c128_transpose_s(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const double* in_data, double* out_data, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + #include + #include + + int main() { + float *input = (float *)0xA0000000; // 输入在DDR空间 + float *output = (float *)0xC0000000; + int num_axes = 4; + int output_shape[4] = {1, 3, 224, 224}; + int perm[4] = {0, 2, 3, 1}; + int strides[4] = {150528, 50176, 224, 1}; + int out_strides[4] = {150528, 50176, 224, 1}; + int core_mask = 0xff; + + fp_transpose_s(num_axes, output_shape, perm, strides, out_strides, input, output, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const float* in_data, float* out_data) +.. c:function:: void hp_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const half* in_data, half* out_data) +.. c:function:: void dp_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const double* in_data, double* out_data) +.. c:function:: void i8_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const int8_t* in_data, int8_t* out_data) +.. c:function:: void i16_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const int16_t* in_data, int16_t* out_data) +.. c:function:: void i32_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const int* in_data, int* out_data) +.. c:function:: void c64_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const float* in_data, float* out_data) +.. c:function:: void c128_transpose_p(int num_axes, const int* output_shape, int* perm, int* strides, int* out_strides, const double* in_data, double* out_data) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 13 + + #include + #include + + int main() { + float *input = (float *)0x10810000; // 输入在L2空间 + float *output = (float *)0x10820000; + int num_axes = 4; + int output_shape[4] = {1, 3, 224, 224}; + int perm[4] = {0, 2, 3, 1}; + int strides[4] = {150528, 50176, 224, 1}; + int out_strides[4] = {150528, 50176, 224, 1}; + + fp_transpose_p(num_axes, output_shape, perm, strides, out_strides, input, output); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/tril.rst.txt b/master/html/_sources/functionlib/dsplib/tril.rst.txt new file mode 100644 index 0000000..02176aa --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/tril.rst.txt @@ -0,0 +1,136 @@ +Tril +================= + +返回输入张量的下三角部分,其余部分设置为零 + +对于输入形状为 :math:`(*, H, W)` 的张量,返回每个 :math:`H \times W` 矩阵的下三角部分。 + +.. math:: + + \text{output}[i,j] = \begin{cases} + \text{input}[i,j], & \text{if } j \leq i + k \\ + 0, & \text{otherwise} + \end{cases} + +其中 :math:`k` 为对角线偏移量: + - :math:`k = 0`:主对角线(默认) + - :math:`k > 0`:主对角线上方第 k 条对角线 + - :math:`k < 0`:主对角线下方第 k 条对角线 + +输入: + - **src** - 输入数据地址。 + - **k** - 对角线偏移量,默认为 0。 + - **height** - 矩阵高度。 + - **width** - 矩阵宽度。 + - **out_elems** - 输出矩阵的数量(批量大小)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_tril_s(int8_t* dst, int8_t* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i16_tril_s(int16_t* dst, int16_t* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i32_tril_s(int32_t* dst, int32_t* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void hp_tril_s(half* dst, half* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void fp_tril_s(float* dst, float* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void dp_tril_s(double* dst, double* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c64_tril_s(float (*dst)[2], float (*src)[2], int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c128_tril_s(double (*dst)[2], double (*src)[2], int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xB0000000; //output在DDR空间 + int64_t k = 0; // 主对角线 + int64_t height = 4; // 矩阵高度 + int64_t width = 4; // 矩阵宽度 + int64_t out_elems = 1; // 矩阵数量 + int core_mask = 0xff; + fp_tril_s(output, input, core_mask, k, height, width, out_elems); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_tril_p(int8_t* dst, int8_t* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i16_tril_p(int16_t* dst, int16_t* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i32_tril_p(int32_t* dst, int32_t* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void hp_tril_p(half* dst, half* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void fp_tril_p(float* dst, float* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void dp_tril_p(double* dst, double* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c64_tril_p(float (*dst)[2], float (*src)[2], int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c128_tril_p(double (*dst)[2], double (*src)[2], int64_t k, int64_t height, int64_t width, int64_t out_elems) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; //input在L2空间 + float *output = (float *)0x10850000; //output在L2空间 + int64_t k = 0; // 主对角线 + int64_t height = 4; // 矩阵高度 + int64_t width = 4; // 矩阵宽度 + int64_t out_elems = 1; // 矩阵数量 + fp_tril_p(output, input, k, height, width, out_elems); + return 0; + } + +.. note:: + **示例说明:** + + 对于 4x4 矩阵,k=0 时的下三角输出示例: + + .. code-block:: text + + 输入矩阵: 输出矩阵: + 1 2 3 4 1 0 0 0 + 5 6 7 8 => 5 6 0 0 + 9 10 11 12 9 10 11 0 + 13 14 15 16 13 14 15 16 + + k=1 时(上移一条对角线,包含更多元素): + + .. code-block:: text + + 输入矩阵: 输出矩阵: + 1 2 3 4 1 2 0 0 + 5 6 7 8 => 5 6 7 0 + 9 10 11 12 9 10 11 12 + 13 14 15 16 13 14 15 16 + + k=-1 时(下移一条对角线,更严格的下三角): + + .. code-block:: text + + 输入矩阵: 输出矩阵: + 1 2 3 4 0 0 0 0 + 5 6 7 8 => 5 0 0 0 + 9 10 11 12 9 10 0 0 + 13 14 15 16 13 14 15 0 diff --git a/master/html/_sources/functionlib/dsplib/triu.rst.txt b/master/html/_sources/functionlib/dsplib/triu.rst.txt new file mode 100644 index 0000000..b12017a --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/triu.rst.txt @@ -0,0 +1,129 @@ +Triu +================= + +返回输入张量的上三角部分,其余部分设置为零 + +对于输入形状为 :math:`(*, H, W)` 的张量,返回每个 :math:`H \times W` 矩阵的上三角部分。 + +.. math:: + + \text{output}[i,j] = \begin{cases} + \text{input}[i,j], & \text{if } j \geq i + k \\ + 0, & \text{otherwise} + \end{cases} + +其中 :math:`k` 为对角线偏移量: + - :math:`k = 0`:主对角线(默认) + - :math:`k > 0`:主对角线上方第 k 条对角线 + - :math:`k < 0`:主对角线下方第 k 条对角线 + +输入: + - **src** - 输入数据地址。 + - **k** - 对角线偏移量,默认为 0。 + - **height** - 矩阵高度。 + - **width** - 矩阵宽度。 + - **out_elems** - 输出矩阵的数量(批量大小)。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **dst** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_triu_s(int8_t* dst, int8_t* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i16_triu_s(int16_t* dst, int16_t* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i32_triu_s(int32_t* dst, int32_t* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void hp_triu_s(half* dst, half* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void fp_triu_s(float* dst, float* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void dp_triu_s(double* dst, double* src, int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c64_triu_s(float (*dst)[2], float (*src)[2], int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c128_triu_s(double (*dst)[2], double (*src)[2], int core_mask, int64_t k, int64_t height, int64_t width, int64_t out_elems) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 13 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; //input在DDR空间 + float *output = (float *)0xB0000000; //output在DDR空间 + int64_t k = 0; // 主对角线 + int64_t height = 4; // 矩阵高度 + int64_t width = 4; // 矩阵宽度 + int64_t out_elems = 1; // 矩阵数量 + int core_mask = 0xff; + fp_triu_s(output, input, core_mask, k, height, width, out_elems); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_triu_p(int8_t* dst, int8_t* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i16_triu_p(int16_t* dst, int16_t* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void i32_triu_p(int32_t* dst, int32_t* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void hp_triu_p(half* dst, half* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void fp_triu_p(float* dst, float* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void dp_triu_p(double* dst, double* src, int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c64_triu_p(float (*dst)[2], float (*src)[2], int64_t k, int64_t height, int64_t width, int64_t out_elems) +.. c:function:: void c128_triu_p(double (*dst)[2], double (*src)[2], int64_t k, int64_t height, int64_t width, int64_t out_elems) + + **C调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 12 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10810000; //input在L2空间 + float *output = (float *)0x10850000; //output在L2空间 + int64_t k = 0; // 主对角线 + int64_t height = 4; // 矩阵高度 + int64_t width = 4; // 矩阵宽度 + int64_t out_elems = 1; // 矩阵数量 + fp_triu_p(output, input, k, height, width, out_elems); + return 0; + } + +**示例说明:** + +对于 4x4 矩阵,k=0 时的上三角输出示例:: + + 输入矩阵: 输出矩阵: + 1 2 3 4 1 2 3 4 + 5 6 7 8 => 0 6 7 8 + 9 10 11 12 0 0 11 12 + 13 14 15 16 0 0 0 16 + +k=1 时(上移一条对角线):: + + 输入矩阵: 输出矩阵: + 1 2 3 4 0 2 3 4 + 5 6 7 8 => 0 0 7 8 + 9 10 11 12 0 0 0 12 + 13 14 15 16 0 0 0 0 + +k=-1 时(下移一条对角线):: + + 输入矩阵: 输出矩阵: + 1 2 3 4 1 2 3 4 + 5 6 7 8 => 5 6 7 8 + 9 10 11 12 0 10 11 12 + 13 14 15 16 0 0 15 16 diff --git a/master/html/_sources/functionlib/dsplib/uniform_real.rst.txt b/master/html/_sources/functionlib/dsplib/uniform_real.rst.txt new file mode 100644 index 0000000..b495a7c --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/uniform_real.rst.txt @@ -0,0 +1,73 @@ +UniformReal +================= +生成服从均匀分布的随机数序列,输出范围为 :math:`[0, 1)`。 + +.. math:: + + output_i \sim \mathcal{U}(0, 1) + +输入: + - **length** - 输出数据长度。 + - **seed** - 随机数种子 1。 + - **seed2** - 随机数种子 2。 + - **core_mask(int, 可选)** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 生成的随机数结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32 + - MT7004 支持 fp32, fp16 + - 随机数生成基于输入种子 `seed` 与 `seed2`,相同的种子对应确定性输出。 + +**共享存储版本:** + +.. c:function:: void fp_uniform_real_s(float* output, int length, int seed, int seed2, int core_mask) +.. c:function:: void hp_uniform_real_s(half* output, int length, int seed, int seed2, int core_mask) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 10 + + // FT78NE 示例 + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0xA0000000; // output 在 DDR 空间 + int length = 1024; + int seed = 1234; + int seed2 = 5678; + int core_mask = 0xff; + fp_uniform_real_s(output, length, seed, seed2, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void fp_uniform_real_p(float* output, int length, int seed, int seed2) +.. c:function:: void hp_uniform_real_p(half* output, int length, int seed, int seed2) + + **C 调用示例:** + + .. code-block:: c + :linenos: + :emphasize-lines: 9 + + // MT7004 示例 + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0x10010000; + int length = 1024; + int seed = 1234; + int seed2 = 5678; + fp_uniform_real_p(output, length, seed, seed2); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/unique.rst.txt b/master/html/_sources/functionlib/dsplib/unique.rst.txt new file mode 100644 index 0000000..4d26e1a --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/unique.rst.txt @@ -0,0 +1,95 @@ +Unique +================= + + +对输入张量进行去重操作,返回去重后的元素组成的张量。 + +.. math:: + + \text{output} = \text{unique}(\text{input}) + +该算子遍历输入张量的所有元素,对于每个元素,如果它尚未出现在输出张量中,则将其添加到输出张量中。最终输出张量包含了输入张量中的所有唯一元素,且保持它们首次出现的相对顺序。 + +输入: + - **input** - 输入张量的数据地址。 + - **input_len** - 输入张量的元素数量。 + +输出: + - **output0** - 去重后的元素组成的张量。 + - **output0_len** - 实际去重后的元素数量。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_Unique_s(int8_t *input, int input_len, int8_t *output0, int32_t *output0_len, int core_mask) +.. c:function:: void i16_Unique_s(int16_t *input, int input_len, int16_t *output0, int32_t *output0_len, int core_mask) +.. c:function:: void i32_Unique_s(int32_t *input, int input_len, int32_t *output0, int32_t *output0_len, int core_mask) +.. c:function:: void hp_Unique_s(half *input, int input_len, half *output0, int32_t *output0_len, int core_mask) +.. c:function:: void fp_Unique_s(float *input, int input_len, float *output0, int32_t *output0_len, int core_mask) +.. c:function:: void dp_Unique_s(double *input, int input_len, double *output0, int32_t *output0_len, int core_mask) +.. c:function:: void c64_Unique_s(float *input, int input_len, float *output0, int32_t *output0_len, int core_mask) +.. c:function:: void c128_Unique_s(double *input, int input_len, double *output0, int32_t *output0_len, int core_mask) + + +**C调用示例(共享存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 12 + + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // 输入张量在DDR空间 + int input_len = 10; // 输入张量长度 + float *output0 = (float *)0xB0000000; // 输出张量在DDR空间 + int32_t output0_len; // 输出张量长度 + int core_mask = 0xff; // 核掩码 + + // 调用Unique算子 + fp_Unique_s(input, input_len, output0, &output0_len, core_mask); + + printf("Unique elements count: %d\n", output0_len); + return 0; + } + +**私有存储版本:** + +.. c:function:: void i8_Unique_p(int8_t *input, int input_len, int8_t *output0, int32_t *output0_len) +.. c:function:: void i16_Unique_p(int16_t *input, int input_len, int16_t *output0, int32_t *output0_len) +.. c:function:: void i32_Unique_p(int32_t *input, int input_len, int32_t *output0, int32_t *output0_len) +.. c:function:: void hp_Unique_p(half *input, int input_len, half *output0, int32_t *output0_len) +.. c:function:: void fp_Unique_p(float *input, int input_len, float *output0, int32_t *output0_len) +.. c:function:: void dp_Unique_p(double *input, int input_len, double *output0, int32_t *output0_len) +.. c:function:: void c64_Unique_p(float *input, int input_len, float *output0, int32_t *output0_len) +.. c:function:: void c128_Unique_p(double *input, int input_len, double *output0, int32_t *output0_len) + +**C调用示例(私有存储版本):** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // 输入张量在L2空间 + int input_len = 10; // 输入张量长度 + float *output0 = (float *)0x11000000; // 输出张量在L2空间 + int32_t output0_len; // 输出张量长度 + + // 调用Unique算子 + fp_Unique_p(input, input_len, output0, &output0_len); + + printf("Unique elements count: %d\n", output0_len); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/unsortedsegmentsum.rst.txt b/master/html/_sources/functionlib/dsplib/unsortedsegmentsum.rst.txt new file mode 100644 index 0000000..14a0f1d --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/unsortedsegmentsum.rst.txt @@ -0,0 +1,113 @@ +UnsortedSegmentSum +==================== + +对输入张量按给定的 segment 索引进行无序分段求和操作。 + +对于每个输入元素,根据其对应的 ``index`` 值,将其累加到输出张量中 +对应的 segment 位置。不同 segment 之间的顺序不做任何排序保证。 + +.. math:: + + \text{output}_{s, j} = \sum_{i \,|\, index_i = s} \text{input}_{i, j} + +其中: + +- :math:`i \in [0, \text{dim0})` +- :math:`j \in [0, \text{dim1})` +- :math:`s \in [0, \text{id\_max})` + +输入: + - **input** - 输入张量的数据地址。 + 数据类型需与所调用的 UnsortedSegmentSum 接口类型一致。 + + - **index** - segment 索引数组地址,类型为 ``int*``, + 长度为 ``dim0``,用于指定每一行输入数据所属的 segment。 + + - **dim0** - 输入张量的第 0 维大小(segment 数量维度)。 + + - **dim1** - 输入张量的第 1 维大小(每个 segment 内的元素数量)。 + + - **id_max** - segment 的最大数量,决定输出张量的第 0 维大小。 + + - **core_mask** - 核掩码(仅共享存储版本使用)。 + +输出: + - **output** - 输出张量的数据地址, + 形状为 ``[id_max, dim1]``, + 数据类型与 ``input`` 保持一致。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - 该算子为无序分段求和(Unsorted),不保证 segment 内或 segment 间的顺序。 + - 输出张量在计算前会被初始化为 0。 + - ``index`` 中的取值范围应满足 ``0 <= index[i] < id_max``。 + +**共享存储版本:** + +.. c:function:: void fp_unsorted_segment_sum_s(float* input, float* output, int* index, int dim0, int dim1, int id_max, int core_mask) +.. c:function:: void dp_unsorted_segment_sum_s(double* input, double* output, int* index, int dim0, int dim1, int id_max, int core_mask) +.. c:function:: void i8_unsorted_segment_sum_s(int8_t* input, int8_t* output, int* index, int dim0, int dim1, int id_max, int core_mask) +.. c:function:: void i16_unsorted_segment_sum_s(int16_t* input, int16_t* output, int* index, int dim0, int dim1, int id_max, int core_mask) +.. c:function:: void i32_unsorted_segment_sum_s(int32_t* input, int32_t* output, int* index, int dim0, int dim1, int id_max, int core_mask) +.. c:function:: void c64_unsorted_segment_sum_s(float* input, float* output, int* index, int dim0, int dim1, int id_max, int core_mask) +.. c:function:: void c128_unsorted_segment_sum_s(double* input, double* output, int* index, int dim0, int dim1, int id_max, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + // FT78NE 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; // input 在 DDR 空间 + float *output = (float *)0xB0000000; + int *index = (int *)0xA1000000; + + int dim0 = 128; + int dim1 = 64; + int id_max = 16; + int core_mask = 0xff; + + fp_unsorted_segment_sum_s(input, output, index, dim0, dim1, id_max, core_mask); + return 0; + } + +**私有存储版本:** + +.. c:function:: void fp_unsorted_segment_sum_p(float* input, float* output, int* index, int dim0, int dim1, int id_max) +.. c:function:: void dp_unsorted_segment_sum_p(double* input, double* output, int* index, int dim0, int dim1, int id_max) +.. c:function:: void i8_unsorted_segment_sum_p(int8_t* input, int8_t* output, int* index, int dim0, int dim1, int id_max) +.. c:function:: void i16_unsorted_segment_sum_p(int16_t* input, int16_t* output, int* index, int dim0, int dim1, int id_max) +.. c:function:: void i32_unsorted_segment_sum_p(int32_t* input, int32_t* output, int* index, int dim0, int dim1, int id_max) +.. c:function:: void c64_unsorted_segment_sum_p(float* input, float* output, int* index, int dim0, int dim1, int id_max) +.. c:function:: void c128_unsorted_segment_sum_p(double* input, double* output, int* index, int dim0, int dim1, int id_max) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 14 + + // MT7004 示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0x10000000; // input 在 L2 空间 + float *output = (float *)0x10010000; + int *index = (int *)0x10020000; + + int dim0 = 64; + int dim1 = 32; + int id_max = 8; + + fp_unsorted_segment_sum_p(input, output, index, dim0, dim1, id_max); + return 0; + } diff --git a/master/html/_sources/functionlib/dsplib/unstack.rst.txt b/master/html/_sources/functionlib/dsplib/unstack.rst.txt new file mode 100644 index 0000000..271b46f --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/unstack.rst.txt @@ -0,0 +1,89 @@ +Unstack +================= + + + +沿指定轴将输入张量拆分为多个子张量,输出子张量按内存顺序排列。 + +.. math:: + + output_i = input[\text{index along axis}] + \quad \text{for each slice along } axis + +输入: + - **input** - 输入张量地址。 + - **shape** - 输入张量各维度大小数组。 + - **ndim** - 输入张量维度。 + - **axis** - 拆分的维度索引。 + - **data_size** - 单个元素字节大小。 + +输出: + - **output** - 输出张量地址数组,长度等于 ``shape[axis]``。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp32, fp64, int8, int16, int32, cplx64, cplx128 + - MT7004 支持 fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void unstack_s(void* input, void* output, int* shape, int ndim, int axis, Uint32 data_size, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 17 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *input = (float *)0xA0000000; + + float *out0 = (float *)0xC0000000; + float *out1 = (float *)0xC0100000; + float *outputs[2] = {out0, out1}; + + int shape[2] = {2, 128}; + int ndim = 2; + int axis = 0; + int core_mask = 0xff; + + unstack_s(input, outputs, shape, ndim, axis, sizeof(float), core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void unstack_p(void* input, void* output, int* shape, int ndim, int axis, Uint32 data_size) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 15 + + //MT7004示例 + #include + #include + + int main(int argc, char* argv[]) { + half *input = (half *)0x10800000; + half *out0 = (half *)0x10810000; + half *out1 = (half *)0x10820000; + half *outputs[2] = {out0, out1}; + + int shape[3] = {1, 2, 64}; + int ndim = 3; + int axis = 1; + + unstack_p(input, outputs, shape, ndim, axis, sizeof(half)); + return 0; + } + diff --git a/master/html/_sources/functionlib/dsplib/where.rst.txt b/master/html/_sources/functionlib/dsplib/where.rst.txt new file mode 100644 index 0000000..62d94aa --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/where.rst.txt @@ -0,0 +1,94 @@ +Where +================= + + + +逐元素根据条件选择两个输入中的元素 + +.. math:: + + output_i = \begin{cases} + input0_i, & \text{if } condition_i \text{ is True} \\ + input1_i, & \text{if } condition_i \text{ is False} + \end{cases} + +输入: + - **input0** - 第一个输入数据地址。 + - **input1** - 第二个输入数据地址。 + - **condition** - 条件数据地址。 + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅共享存储版本需要)。 + +输出: + - **output** - 计算结果地址。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持int8, int16, int32, fp16, fp32, fp64, cplx64, cplx128 + - MT7004 支持fp16, fp32, int16, int32, cplx64 + +**共享存储版本:** + +.. c:function:: void i8_where_s(int8_t* input0, int8_t* input1, bool* condition, int8_t* output, int length, int core_mask) +.. c:function:: void i16_where_s(int16_t* input0, int16_t* input1, bool* condition, int16_t* output, int length, int core_mask) +.. c:function:: void i32_where_s(int32_t* input0, int32_t* input1, bool* condition, int32_t* output, int length, int core_mask) +.. c:function:: void hp_where_s(half* input0, half* input1, bool* condition, half* output, int length, int core_mask) +.. c:function:: void fp_where_s(float* input0, float* input1, bool* condition, float* output, int length, int core_mask) +.. c:function:: void dp_where_s(double* input0, double* input1, bool* condition, double* output, int length, int core_mask) +.. c:function:: void c64_where_s(float* input0, float* input1, bool* condition, float* output, int length, int core_mask) +.. c:function:: void c128_where_s(double* input0, double* input1, bool* condition, double* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 11 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0xA0000000; //input0在DDR空间 + float *input1 = (float *)0xB0000000; //input1在DDR空间 + bool *condition = (bool *)0xC0000000; //condition在DDR空间 + float *output = (float *)0xD0000000; //output在DDR空间 + int length = 1000; + int core_mask = 0xff; + fp_where_s(input0, input1, condition, output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_where_p(int8_t* input0, int8_t* input1, bool* condition, int8_t* output, int length) +.. c:function:: void i16_where_p(int16_t* input0, int16_t* input1, bool* condition, int16_t* output, int length) +.. c:function:: void i32_where_p(int32_t* input0, int32_t* input1, bool* condition, int32_t* output, int length) +.. c:function:: void hp_where_p(half* input0, half* input1, bool* condition, half* output, int length) +.. c:function:: void fp_where_p(float* input0, float* input1, bool* condition, float* output, int length) +.. c:function:: void dp_where_p(double* input0, double* input1, bool* condition, double* output, int length) +.. c:function:: void c64_where_p(float* input0, float* input1, bool* condition, float* output, int length) +.. c:function:: void c128_where_p(double* input0, double* input1, bool* condition, double* output, int length) + + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + int main(int argc, char* argv[]) { + float *input0 = (float *)0x10000000; //input0在L2空间 + float *input1 = (float *)0x10001000; //input1在L2空间 + bool *condition = (bool *)0x10002000; //condition在L2空间 + float *output = (float *)0x10003000; //output在L2空间 + int length = 1000; + fp_where_p(input0, input1, condition, output, length); + return 0; + } \ No newline at end of file diff --git a/master/html/_sources/functionlib/dsplib/zeroslike.rst.txt b/master/html/_sources/functionlib/dsplib/zeroslike.rst.txt new file mode 100644 index 0000000..c54abe8 --- /dev/null +++ b/master/html/_sources/functionlib/dsplib/zeroslike.rst.txt @@ -0,0 +1,87 @@ +ZerosLike +================= + + + +将输出数组逐元素置为 0。 +该算子不依赖输入数据,仅根据给定长度与数据类型,对输出缓冲区进行清零操作,常用于初始化张量或中间结果。 + +.. math:: + + dst_i = 0 + +输入: + - **length** - 计算长度。 + - **core_mask** - 核掩码(仅适用于共享存储版本)。 + +输出: + - **output** - 输出数据地址,逐元素被置为 0。 + +支持平台: + ``FT78NE`` + ``MT7004`` + +.. note:: + - FT78NE 支持 fp, dp, int8, int16, int32, cplx64, cplx128 + - MT7004 支持 hp, fp, int16, int32, cplx64 + - 复数类型中实部与虚部均被置为 0 + +**共享存储版本:** + +.. c:function:: void i8_zerolike_s(int8_t* output, int length, int core_mask) +.. c:function:: void i16_zerolike_s(int16_t* output, int length, int core_mask) +.. c:function:: void i32_zerolike_s(int32_t* output, int length, int core_mask) +.. c:function:: void hp_zerolike_s(half* output, int length, int core_mask) +.. c:function:: void fp_zerolike_s(float* output, int length, int core_mask) +.. c:function:: void dp_zerolike_s(double* output, int length, int core_mask) +.. c:function:: void c64_zerolike_s(float* output, int length, int core_mask) +.. c:function:: void c128_zerolike_s(double* output, int length, int core_mask) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 10 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0xA0000000; // output在DDR空间 + int length = 1024; + int core_mask = 0xff; + + fp_zerolike_s(output, length, core_mask); + return 0; + } + + +**私有存储版本:** + +.. c:function:: void i8_zerolike_p(int8_t* output, int length) +.. c:function:: void i16_zerolike_p(int16_t* output, int length) +.. c:function:: void i32_zerolike_p(int32_t* output, int length) +.. c:function:: void hp_zerolike_p(half* output, int length) +.. c:function:: void fp_zerolike_p(float* output, int length) +.. c:function:: void dp_zerolike_p(double* output, int length) +.. c:function:: void c64_zerolike_p(float* output, int length) +.. c:function:: void c128_zerolike_p(double* output, int length) + +**C调用示例:** + +.. code-block:: c + :linenos: + :emphasize-lines: 9 + + //FT78NE示例 + #include + #include + + int main(int argc, char* argv[]) { + float *output = (float *)0x10810000; // output在L2空间 + int length = 1024; + + fp_zerolike_p(output, length); + return 0; + } diff --git a/master/html/_sources/functionlib/rstfiles.rst.txt b/master/html/_sources/functionlib/rstfiles.rst.txt new file mode 100644 index 0000000..3282db7 --- /dev/null +++ b/master/html/_sources/functionlib/rstfiles.rst.txt @@ -0,0 +1,7 @@ +算子库支持 +========== + +.. toctree:: + :maxdepth: 2 + + dsplib/dsplib_index \ No newline at end of file diff --git a/master/html/_sources/rstfiles.rst.txt b/master/html/_sources/rstfiles.rst.txt new file mode 100644 index 0000000..84ad063 --- /dev/null +++ b/master/html/_sources/rstfiles.rst.txt @@ -0,0 +1,13 @@ +.. MindSpore Signal+ 使用手册 documentation master file, created by + sphinx-quickstart on Tue Aug 12 10:28:49 2025. + You can adapt this file completely to your liking, but it should at least + contain the root `toctree` directive. + +MindSpore Signal+ 使用手册 +========================== + + +.. toctree:: + :maxdepth: 2 + + functionlib/rstfiles diff --git a/master/html/appdevelop/ai_dsp/gray_cnn.html b/master/html/appdevelop/ai_dsp/gray_cnn.html index 2390613..17ca0e0 100644 --- a/master/html/appdevelop/ai_dsp/gray_cnn.html +++ b/master/html/appdevelop/ai_dsp/gray_cnn.html @@ -22,9 +22,7 @@ - - - + @@ -35,7 +33,7 @@ - + MindSpore Signal+ 使用手册
@@ -46,24 +44,8 @@
@@ -72,16 +54,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/appdevelop/ai_dsp/index.html b/master/html/appdevelop/ai_dsp/index.html index 92c9a20..50b5e66 100644 --- a/master/html/appdevelop/ai_dsp/index.html +++ b/master/html/appdevelop/ai_dsp/index.html @@ -22,9 +22,7 @@ - - - + @@ -35,7 +33,7 @@ - + MindSpore Signal+ 使用手册
@@ -46,19 +44,8 @@
@@ -67,15 +54,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/appdevelop/autocodegen/index.html b/master/html/appdevelop/autocodegen/index.html index 97a9ad2..524ca0b 100644 --- a/master/html/appdevelop/autocodegen/index.html +++ b/master/html/appdevelop/autocodegen/index.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
@@ -45,18 +43,8 @@
@@ -65,15 +53,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/appdevelop/dsp/index.html b/master/html/appdevelop/dsp/index.html index 871b6bd..dd1d89c 100644 --- a/master/html/appdevelop/dsp/index.html +++ b/master/html/appdevelop/dsp/index.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
@@ -45,19 +43,8 @@
@@ -66,15 +53,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/appdevelop/dsp/rdsar.html b/master/html/appdevelop/dsp/rdsar.html index 2438cc6..b736448 100644 --- a/master/html/appdevelop/dsp/rdsar.html +++ b/master/html/appdevelop/dsp/rdsar.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
@@ -45,26 +43,8 @@
@@ -73,16 +53,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/appdevelop/index.html b/master/html/appdevelop/index.html index 84ecc90..482cd4e 100644 --- a/master/html/appdevelop/index.html +++ b/master/html/appdevelop/index.html @@ -22,9 +22,7 @@ - - - + @@ -35,7 +33,7 @@ - + MindSpore Signal+ 使用手册
@@ -46,16 +44,8 @@
@@ -64,14 +54,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/custom_op/complex_abs.html b/master/html/functionlib/custom_op/complex_abs.html index 4d0b85d..78ad0a3 100644 --- a/master/html/functionlib/custom_op/complex_abs.html +++ b/master/html/functionlib/custom_op/complex_abs.html @@ -22,9 +22,7 @@ - - - + @@ -35,7 +33,7 @@ - + MindSpore Signal+ 使用手册
@@ -46,24 +44,8 @@
@@ -72,16 +54,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/custom_op/fft.html b/master/html/functionlib/custom_op/fft.html index c6b66bb..00db727 100644 --- a/master/html/functionlib/custom_op/fft.html +++ b/master/html/functionlib/custom_op/fft.html @@ -22,9 +22,7 @@ - - - + @@ -35,7 +33,7 @@ - + MindSpore Signal+ 使用手册
@@ -46,24 +44,8 @@
@@ -72,16 +54,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/custom_op/ifft.html b/master/html/functionlib/custom_op/ifft.html index 9dacbed..9d94c40 100644 --- a/master/html/functionlib/custom_op/ifft.html +++ b/master/html/functionlib/custom_op/ifft.html @@ -22,9 +22,7 @@ - - - + @@ -35,7 +33,7 @@ - + MindSpore Signal+ 使用手册
@@ -46,24 +44,8 @@
@@ -72,16 +54,14 @@
-
+

-

© 版权所有 2025 - 2025, NUDT-674。

+

© 版权所有 2025 - 2026, NUDT-674。

利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/custom_op/index.html b/master/html/functionlib/custom_op/index.html index 4d3affd..8d9d78d 100644 --- a/master/html/functionlib/custom_op/index.html +++ b/master/html/functionlib/custom_op/index.html @@ -22,9 +22,7 @@ - - - + @@ -35,7 +33,7 @@ - + MindSpore Signal+ 使用手册
@@ -46,21 +44,8 @@
@@ -69,15 +54,14 @@
-
@@ -112,15 +277,15 @@
    -
  • - +
  • +
  • @@ -151,7 +316,7 @@
  • Clip - 将输入裁剪到区间 [min_val, max_val]

    -\[output_i = \min(\max(input_i, \text{min_val}), \text{max_val})\]
    +\[output_i = \min(\max(input_i, \text{min\_val}), \text{max\_val})\]
  • LRelu - 带泄露的线性整流单元(Leaky Rectified Linear Unit),它在输入为正时保持线性,在输入为负时也保留一个很小的斜率,以避免标准 ReLU 中的“死亡神经元”问题。

    @@ -227,7 +392,7 @@ input_i \,\Phi(input_i)
    \[\begin{split}output_i = \begin{cases} - input_i, & input_i \gt 88.0 \\ + input_i, & input_i > 88.0 \\ \ln(1 + e^{input_i}), & \text{otherwise} \end{cases}\end{split}\]
    @@ -237,7 +402,7 @@ input_i \,\Phi(input_i)
    \[\begin{split}output_i = \begin{cases} - input_i, & input_i \ge 0 \\ + input_i, & input_i >= 0 \\ \alpha (e^{input_i} - 1), & input_i < 0 \end{cases}\end{split}\]
    @@ -248,7 +413,7 @@ input_i \,\Phi(input_i)
    \[\begin{split}output_i = \begin{cases} -input_i, & input_i \ge 0 \\ +input_i, & input_i >= 0 \\ \alpha (e^{\frac{input_i}{alpha}} - 1), & input_i < 0 \end{cases}\end{split}\]

    其中 \(\alpha\) 为可调超参数,用于控制负区间的平滑程度。

    @@ -864,7 +1029,7 @@ input_i + \lambda, & \text{if } input_i < -\lambda \\
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/activation_grad.html b/master/html/functionlib/dsplib/activation_grad.html new file mode 100644 index 0000000..bed1682 --- /dev/null +++ b/master/html/functionlib/dsplib/activation_grad.html @@ -0,0 +1,542 @@ + + + + + + + + + ActivationGrad — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    ActivationGrad

    +

    激活函数梯度计算算子系列。该系列算子根据激活函数的输入(或输出)以及上一层传回的梯度,计算并输出当前层的梯度。

    +
    +\[\text{通用公式:}\quad dst_i = src0_i \cdot f'(src1_i)\]
    +

    其中 \(src0\) 为梯度输入(Output Gradient),\(src1\) 为激活函数的原始输入或输出(取决于具体激活函数类型),\(f'\) 为激活函数的导数。

    +
    +
    包含算子列表:
      +
    • ReluGrad: ReLU 激活梯度。

    • +
    • Relu6Grad: ReLU6 激活梯度。

    • +
    • LeakyReluGrad (l_relu_grad): Leaky ReLU 激活梯度,需传入参数 alpha

    • +
    • SigmoidGrad: Sigmoid 激活梯度。

    • +
    • TanhGrad: Tanh 激活梯度。

    • +
    • HSigmoidGrad: Hard Sigmoid 激活梯度。

    • +
    • HSwishGrad: Hard Swish 激活梯度。

    • +
    • GeluGrad: GELU 激活梯度。

    • +
    • EluGrad: ELU 激活梯度,需传入参数 alpha

    • +
    • SoftplusGrad: Softplus 激活梯度。

    • +
    • HardShrinkGrad: Hard Shrink 激活梯度,需传入参数 lambd

    • +
    • SoftshrinkGrad: Softshrink 激活梯度,需传入参数 lambd

    • +
    +
    +
    输入:
      +
    • src0 - 梯度输入数据地址。

    • +
    • src1 - 原始输入/输出数据地址。

    • +
    • length - 计算长度。

    • +
    • alpha / lambd (float, 可选) - 特定激活函数所需的系数。

    • +
    • core_mask (int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • dst - 梯度计算结果输出地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 仅支持 fp32。

    • +
    • MT7004 支持 fp32, fp16。

    • +
    • 对于不同的算子,src1 的含义可能不同(例如 SigmoidGrad 通常使用前向传播的输出作为 src1,而 ReluGrad 使用前向传播的输入作为 src1),需确保上层传入地址正确。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_relu_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void hp_relu_grad_s(half *src0, half *src1, half *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_relu6_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_l_relu_grad_s(float *src0, float *src1, float *dst, int length, float alpha, int core_mask)
    +
    + +
    +
    +void fp_sigmoid_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_tanh_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_h_sigmoid_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_h_swish_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_gelu_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_elu_grad_s(float *src0, float *src1, float *dst, int length, float alpha, int core_mask)
    +
    + +
    +
    +void fp_softplus_grad_s(float *src0, float *src1, float *dst, int length, int core_mask)
    +
    + +
    +
    +void fp_hard_shrink_grad_s(float *src0, float *src1, float *dst, int length, float lambd, int core_mask)
    +
    + +
    +
    +void fp_softshrink_grad_s(float *src0, float *src1, float *dst, int length, float lambd, int core_mask)
    +

    C调用示例:

    +
     1//FT78NE示例(共享存储)
    + 2#include <stdio.h>
    + 3#include "78NE/utils.h"
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *src0 = (float *)0xA0000000;   // 梯度输入在共享存储
    + 7    float *src1 = (float *)0xA1000000;   // 原始数据在共享存储
    + 8    float *dst  = (float *)0xB0000000;   // 结果输出到共享存储
    + 9    int length = 1024;
    +10    int core_mask = 0xff;
    +11    fp_relu_grad_s(src0, src1, dst, length, core_mask);
    +12    return 0;
    +13}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void fp_relu_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void hp_relu_grad_p(half *src0, half *src1, half *dst, int length)
    +
    + +
    +
    +void fp_relu6_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void fp_l_relu_grad_p(float *src0, float *src1, float *dst, int length, float alpha)
    +
    + +
    +
    +void fp_sigmoid_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void fp_tanh_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void fp_h_sigmoid_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void fp_h_swish_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void fp_gelu_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void fp_elu_grad_p(float *src0, float *src1, float *dst, int length, float alpha)
    +
    + +
    +
    +void fp_softplus_grad_p(float *src0, float *src1, float *dst, int length)
    +
    + +
    +
    +void fp_hard_shrink_grad_p(float *src0, float *src1, float *dst, int length, float lambd)
    +
    + +
    +
    +void fp_softshrink_grad_p(float *src0, float *src1, float *dst, int length, float lambd)
    +

    C调用示例:

    +
     1//MT7004 示例
    + 2#include <stdio.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    float *src0 = (float *)0x10000000;   // 私有存储空间地址
    + 6    float *src1 = (float *)0x10001000;
    + 7    float *dst  = (float *)0x10002000;
    + 8    int length = 139;
    + 9    float alpha = 0.01f;
    +10    fp_l_relu_grad_p(src0, src1, dst, length, alpha);
    +11    return 0;
    +12}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/adam.html b/master/html/functionlib/dsplib/adam.html new file mode 100644 index 0000000..f6827e7 --- /dev/null +++ b/master/html/functionlib/dsplib/adam.html @@ -0,0 +1,454 @@ + + + + + + + + + Adam — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Adam

    +
    +

    使用 Adam 算法更新参数权重。支持 Nesterov 动量。

    +

    算法逻辑如下:

    +
    +\[\begin{split}m_t = m_{t-1} + (g_t - m_{t-1}) \cdot (1 - \beta_1) \\ +v_t = v_{t-1} + (g_t^2 - v_{t-1}) \cdot (1 - \beta_2) \\ +\hat{lr} = lr \cdot \frac{\sqrt{1 - \beta_2^t}}{1 - \beta_1^t}\end{split}\]
    +

    如果不启用 Nesterov:

    +
    +\[w_t = w_{t-1} - \hat{lr} \cdot \frac{m_t}{\sqrt{v_t} + \epsilon}\]
    +

    如果启用 Nesterov:

    +
    +\[w_t = w_{t-1} - \hat{lr} \cdot \frac{m_t \cdot \beta_1 + (1 - \beta_1) \cdot g_t}{\sqrt{v_t} + \epsilon}\]
    +
    +
    输入:
      +
    • m - 一阶矩向量地址(输入/输出)。

    • +
    • v - 二阶矩向量地址(输入/输出)。

    • +
    • gradient - 梯度向量地址。

    • +
    • weight - 权重向量地址(输入/输出)。

    • +
    • beta1 - 一阶矩估计的指数衰减率。

    • +
    • beta2 - 二阶矩估计的指数衰减率。

    • +
    • beta1_power - \(\beta_1^t\) 的值(指针形式传入)。

    • +
    • beta2_power - \(\beta_2^t\) 的值(指针形式传入)。

    • +
    • eps - 数值稳定性项 epsilon。

    • +
    • learning_rate - 学习率。

    • +
    • nesterov - 是否启用 Nesterov 动量(0: 不启用, 1: 启用)。

    • +
    • start - 计算的起始索引(包含)。

    • +
    • end - 计算的结束索引(不包含)。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • m - 更新后的一阶矩。

    • +
    • v - 更新后的二阶矩。

    • +
    • weight - 更新后的权重。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • 仅支持 fp32 数据类型。

    • +
    +
    +
    +

    共享存储版本:

    +
    +
    +void fp_adam_s(float *m, float *v, const float *gradient, float *weight, float beta1, float beta2, float *beta1_power, float *beta2_power, float eps, float learning_rate, int nesterov, int start, int end, int core_mask)
    +

    C调用示例:

    +
     1#include <stdio.h>
    + 2
    + 3int main(int argc, char* argv[]) {
    + 4    // 假设所有数据均位于DDR空间
    + 5    float* m = (float*)0xC0000000;
    + 6    float* v = (float*)0xC1000000;
    + 7    float* gradient = (float*)0xC2000000;
    + 8    float* weight = (float*)0xC3000000;
    + 9
    +10    // 标量参数
    +11    float beta1 = 0.9f;
    +12    float beta2 = 0.999f;
    +13    float beta1_power_val = 0.9f;   // beta1^1
    +14    float beta2_power_val = 0.999f; // beta2^1
    +15    float* beta1_power = &beta1_power_val;
    +16    float* beta2_power = &beta2_power_val;
    +17    float eps = 1e-8f;
    +18    float learning_rate = 0.001f;
    +19    int nesterov = 1;
    +20
    +21    int start = 0;
    +22    int end = 800000; // 元素总数
    +23    int core_mask = 0xff; // 使用所有核心
    +24
    +25    fp_adam_s(m, v, gradient, weight, beta1, beta2, beta1_power, beta2_power,
    +26              eps, learning_rate, nesterov, start, end, core_mask);
    +27
    +28    return 0;
    +29}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void fp_adam_p(float *m, float *v, const float *gradient, float *weight, float beta1, float beta2, float *beta1_power, float *beta2_power, float eps, float learning_rate, int nesterov, int start, int end)
    +

    C调用示例:

    +
     1#include <stdio.h>
    + 2
    + 3int main(int argc, char* argv[]) {
    + 4    // 假设所有数据均位于L2/AM空间
    + 5    float* m = (float*)0x10820000;
    + 6    float* v = (float*)0x10830000;
    + 7    float* gradient = (float*)0x10840000;
    + 8    float* weight = (float*)0x10850000;
    + 9
    +10    float beta1 = 0.9f;
    +11    float beta2 = 0.999f;
    +12    float beta1_power_val = 0.9f;
    +13    float beta2_power_val = 0.999f;
    +14    float eps = 1e-8f;
    +15    float learning_rate = 0.001f;
    +16    int nesterov = 0;
    +17    int start = 0;
    +18    int end = 2000;
    +19
    +20    fp_adam_p(m, v, gradient, weight, beta1, beta2, &beta1_power_val, &beta2_power_val,
    +21              eps, learning_rate, nesterov, start, end);
    +22
    +23    return 0;
    +24}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/adamweightdecay.html b/master/html/functionlib/dsplib/adamweightdecay.html index 2bbd947..c5e4762 100644 --- a/master/html/functionlib/dsplib/adamweightdecay.html +++ b/master/html/functionlib/dsplib/adamweightdecay.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
    @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -204,9 +369,9 @@ var_t &= var_{t-1} - lr \cdot (\hat{m}_t + decay \cdot var_{t-1}) 15 float epsilon = 1e-8f; 16 float decay = 1e-2f; 17 fp_adamweightdecay_s(var, m, v, gradient, lr, -18 beta1, beta2, epsilon, decay, -19 start, end, core_mask); -20 return 0; +18 beta1, beta2, epsilon, decay, +19 start, end, core_mask); +20 return 0; 21}
    @@ -237,9 +402,9 @@ var_t &= var_{t-1} - lr \cdot (\hat{m}_t + decay \cdot var_{t-1}) 13 float epsilon = 1e-6f; 14 float decay = 5e-3f; 15 hp_adamweightdecay_p(var, m, v, gradient, lr, -16 beta1, beta2, epsilon, decay, -17 length); -18 return 0; +16 beta1, beta2, epsilon, decay, +17 length); +18 return 0; 19}
  • @@ -258,7 +423,7 @@ var_t &= var_{t-1} - lr \cdot (\hat{m}_t + decay \cdot var_{t-1})
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/adder.html b/master/html/functionlib/dsplib/adder.html index 54116a2..389f38a 100644 --- a/master/html/functionlib/dsplib/adder.html +++ b/master/html/functionlib/dsplib/adder.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
    @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -200,9 +365,9 @@ weight_t &= weight_{t-1} - learning\_rate \cdot update_t 13 float moment = 0.99f; 14 bool nesterov = false; 15 fp_applymomentum_s(weight, accumulate, gradient, -16 learning_rate, moment, nesterov, -17 start, end, core_mask); -18 return 0; +16 learning_rate, moment, nesterov, +17 start, end, core_mask); +18 return 0; 19}
    @@ -231,9 +396,9 @@ weight_t &= weight_{t-1} - learning\_rate \cdot update_t 11 float moment = 0.9f; 12 bool nesterov = true; 13 hp_applymomentum_p(weight, accumulate, gradient, -14 learning_rate, moment, nesterov, -15 length); -16 return 0; +14 learning_rate, moment, nesterov, +15 length); +16 return 0; 17}
    @@ -252,7 +417,7 @@ weight_t &= weight_{t-1} - learning\_rate \cdot update_t
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/argmax.html b/master/html/functionlib/dsplib/argmax.html new file mode 100644 index 0000000..001fe61 --- /dev/null +++ b/master/html/functionlib/dsplib/argmax.html @@ -0,0 +1,446 @@ + + + + + + + + + Argmax — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Argmax

    +

    沿指定轴查找最大 topk 个值的索引。当 topk=1 时,该算子等价于 ArgMax

    +
    +\[Y_i = \underset{k}{\operatorname{argmax}} (X_{slice_i})\]
    +

    其中 \(X_{slice_i}\) 是输入张量中沿指定轴的一个切片,函数返回该切片中最大值的索引 \(k\)

    +
    +
    输入:
      +
    • input - 输入数据地址。

    • +
    • output - 输出索引的数据地址,数据类型通常为int32。

    • +
    • output_value - (可选) 输出值的数据地址。

    • +
    • in_shape - 输入张量的维度信息数组。

    • +
    • in_strides - 输入张量的步长信息数组。

    • +
    • out_strides - 输出张量的步长信息数组。

    • +
    • arg_elements - 用于存放候选值的临时工作空间地址。

    • +
    • index - 用于存放候选索引的临时工作空间地址。

    • +
    • topk - 需要查找的最大值的数量。设置为1以执行ArgMax操作。

    • +
    • out_value - 是否返回数值的标志。若为非0,则 output_value 必须提供有效地址。

    • +
    • input_shape_size - 输入张量的维度数 (即 in_shape 数组的长度)。

    • +
    • axis - 执行查找操作的轴。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • output - 存储索引的输出张量。

    • +
    • output_value - 如果 return_values 为 true,则此处存储找到的值。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32

    • +
    • MT7004 支持fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_argmax_s(float *input, void *output, float *output_value, int32_t *in_shape, int *in_strides, int *out_strides, float *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis, int core_mask)
    +
    + +
    +
    +void hp_argmax_s(half *input, void *output, half *output_value, int32_t *in_shape, int *in_strides, int *out_strides, half *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <argmax.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0xA0000000;          // input在DDR空间
    + 6    int *output = (int *)0xB0000000;             // output indices
    + 7    float *output_value = (float *)0xC0000000;     // output values
    + 8    float *arg_elements = (float *)0xD0000000;   // temp workspace 1
    + 9    int *index = (int *)0xE0000000;               // temp workspace 2
    +10
    +11    int in_shape[] = {2, 3, 4};                  // input shape: (2, 3, 4)
    +12    int in_strides[] = {12, 4, 1};               // input strides for contiguous layout
    +13    int out_strides[] = {4, 1};                  // output strides, shape is (2, 4)
    +14    int input_shape_size = 3;
    +15
    +16    int axis = 1;                                // 沿第1轴操作
    +17    int topk = 1;                                // ArgMax
    +18    int out_value = 1;                           // 同时返回值
    +19    int core_mask = 0xff;
    +20
    +21    fp_argmax_s(input, output, output_value, in_shape, in_strides, out_strides, arg_elements, index, topk, out_value, input_shape_size, axis, core_mask);
    +22    return 0;
    +23}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_argmax_p(float *input, void *output, float *output_value, int32_t *in_shape, int *in_strides, int *out_strides, float *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis)
    +
    + +
    +
    +void hp_argmax_p(half *input, void *output, half *output_value, int32_t *in_shape, int *in_strides, int *out_strides, half *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <argmax.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0x10001000;          // input在DDR空间
    + 6    int *output = (int *)0x10002000;             // output indices
    + 7    float *output_value = (float *)0x10003000;     // output values
    + 8    float *arg_elements = (float *)0x10004000;   // temp workspace 1
    + 9    int *index = (int *)0x10005000;               // temp workspace 2
    +10
    +11    int in_shape[] = {2, 3, 4};                  // input shape: (2, 3, 4)
    +12    int in_strides[] = {12, 4, 1};               // input strides for contiguous layout
    +13    int out_strides[] = {4, 1};                  // output strides, shape is (2, 4)
    +14    int input_shape_size = 3;
    +15
    +16    int axis = 1;                                // 沿第1轴操作
    +17    int topk = 1;                                // ArgMax
    +18    int out_value = 1;                           // 同时返回值
    +19
    +20    fp_argmax_p(input, output, output_value, in_shape, in_strides, out_strides,
    +21                arg_elements, index, topk, out_value, input_shape_size, axis);
    +22    return 0;
    +23}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/argmin.html b/master/html/functionlib/dsplib/argmin.html new file mode 100644 index 0000000..9de45d1 --- /dev/null +++ b/master/html/functionlib/dsplib/argmin.html @@ -0,0 +1,447 @@ + + + + + + + + + Argmin — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Argmin

    +

    沿指定轴查找最小 topk 个值的索引。当 topk=1 时,该算子等价于 ArgMin

    +
    +\[Y_i = \underset{k}{\operatorname{argmin}} (X_{slice_i})\]
    +

    其中 \(X_{slice_i}\) 是输入张量中沿指定轴的一个切片,函数返回该切片中最小值的索引 \(k\)

    +
    +
    输入:
      +
    • input - 输入数据地址。

    • +
    • output - 输出索引的数据地址,数据类型通常为int32。

    • +
    • output_value - (可选) 输出值的数据地址。

    • +
    • in_shape - 输入张量的维度信息数组。

    • +
    • in_strides - 输入张量的步长信息数组。

    • +
    • out_strides - 输出张量的步长信息数组。

    • +
    • arg_elements - 用于存放候选值的临时工作空间地址。

    • +
    • index - 用于存放候选索引的临时工作空间地址。

    • +
    • topk - 需要查找的最小值的数量。设置为1以执行ArgMin操作。

    • +
    • out_value - 是否返回数值的标志。若为非0,则 output_value 必须提供有效地址。

    • +
    • input_shape_size - 输入张量的维度数 (即 in_shape 数组的长度)。

    • +
    • axis - 执行查找操作的轴。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • output - 存储索引的输出张量。

    • +
    • output_value - 如果 return_values 为 true,则此处存储找到的值。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32

    • +
    • MT7004 支持fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_argmin_s(float *input, void *output, float *output_value, int *in_shape, int *in_strides, int *out_strides, float *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis, int core_mask)
    +
    + +
    +
    +void hp_argmin_s(half *input, void *output, half *output_value, int *in_shape, int *in_strides, int *out_strides, half *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <argmin.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0xA0000000;          // input在DDR空间
    + 6    int *output = (int *)0xB0000000;             // output indices
    + 7    float *output_value = (float *)0xC0000000;     // output values
    + 8    float *arg_elements = (float *)0xD0000000;   // temp workspace 1
    + 9    int *index = (int *)0xE0000000;               // temp workspace 2
    +10
    +11    int in_shape[] = {2, 3, 4};                  // input shape: (2, 3, 4)
    +12    int in_strides[] = {12, 4, 1};               // input strides for contiguous layout
    +13    int out_strides[] = {4, 1};                  // output strides, shape is (2, 4)
    +14    int input_shape_size = 3;
    +15
    +16    int axis = 1;                                // 沿第1轴操作
    +17    int topk = 1;                                // ArgMax
    +18    int out_value = 1;                           // 同时返回值
    +19    int core_mask = 0xff;
    +20
    +21    fp_argmin_s(input, output, output_value, in_shape, in_strides, out_strides,
    +22                arg_elements, index, topk, out_value, input_shape_size, axis, core_mask);
    +23    return 0;
    +24}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_argmin_p(float *input, void *output, float *output_value, int32_t *in_shape, int *in_strides, int *out_strides, float *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis)
    +
    + +
    +
    +void hp_argmin_p(half *input, void *output, half *output_value, int32_t *in_shape, int *in_strides, int *out_strides, half *arg_elements, int *index, int topk, int out_value, int input_shape_size, int axis)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <argmin.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0x10001000;          // input在DDR空间
    + 6    int *output = (int *)0x10002000;             // output indices
    + 7    float *output_value = (float *)0x10003000;     // output values
    + 8    float *arg_elements = (float *)0x10004000;   // temp workspace 1
    + 9    int *index = (int *)0x10005000;               // temp workspace 2
    +10
    +11    int in_shape[] = {2, 3, 4};                  // input shape: (2, 3, 4)
    +12    int in_strides[] = {12, 4, 1};               // input strides for contiguous layout
    +13    int out_strides[] = {4, 1};                  // output strides, shape is (2, 4)
    +14    int input_shape_size = 3;
    +15
    +16    int axis = 1;                                // 沿第1轴操作
    +17    int topk = 1;                                // ArgMin
    +18    int out_value = 1;                           // 同时返回值
    +19
    +20    fp_argmin_p(input, output, output_value, in_shape, in_strides, out_strides,
    +21                arg_elements, index, topk, out_value, input_shape_size, axis);
    +22    return 0;
    +23}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/assert.html b/master/html/functionlib/dsplib/assert.html index 6f2d34a..939ea2c 100644 --- a/master/html/functionlib/dsplib/assert.html +++ b/master/html/functionlib/dsplib/assert.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
    @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -204,12 +369,12 @@ 18 int window_h = 2; 19 int core_mask = 0xff; 20 fp_avgpoolinggrad_s(input, output, batch, output_h, output_w, -21 channel, input_w, input_h, -22 stride_w, stride_h, -23 pad_l, pad_u, -24 window_w, window_h, -25 core_mask); -26 return 0; +21 channel, input_w, input_h, +22 stride_w, stride_h, +23 pad_l, pad_u, +24 window_w, window_h, +25 core_mask); +26 return 0; 27}
    @@ -246,12 +411,12 @@ 19 int start_idx = 0; 20 int end_idx = batch * output_h * output_w * channel; 21 hp_avgpoolinggrad_p(input, output, batch, output_h, output_w, -22 channel, input_w, input_h, -23 stride_w, stride_h, -24 pad_l, pad_u, -25 window_w, window_h, -26 start_idx, end_idx); -27 return 0; +22 channel, input_w, input_h, +23 stride_w, stride_h, +24 pad_l, pad_u, +25 window_w, window_h, +26 start_idx, end_idx); +27 return 0; 28}
    @@ -270,7 +435,7 @@
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/batchnorm.html b/master/html/functionlib/dsplib/batchnorm.html new file mode 100644 index 0000000..a9efeea --- /dev/null +++ b/master/html/functionlib/dsplib/batchnorm.html @@ -0,0 +1,444 @@ + + + + + + + + + BatchNorm — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    BatchNorm

    +

    对输入数组按通道执行批归一化(Batch Normalization)计算。 +该算子使用给定的均值与方差对输入进行标准化,并通过 epsilon 保证数值稳定性。

    +
    +\[dst_{i,j} = \frac{src_{i,j} - mean_j}{\sqrt{variance_j + \epsilon}}\]
    +

    其中:

    +
      +
    • \(i\) 表示第 unit 个样本

    • +
    • \(j\) 表示通道索引

    • +
    • \(mean_j\)\(variance_j\) 为第 \(j\) 个通道的统计量

    • +
    +

    对于 int8 类型输入,内部以浮点方式计算,最终结果按实现规则取整并输出为 int8

    +
    +
    输入:
      +
    • input - 输入数据地址,形状为 [unit, channel]

    • +
    • mean - 均值数组地址,长度为 channel

    • +
    • variance - 方差数组地址,长度为 channel

    • +
    • unit - 样本数(或展开后的空间维度)。

    • +
    • channel - 通道数。

    • +
    • epsilon - 数值稳定因子。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 批归一化后的输出数据地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 int8fp32 类型

    • +
    • MT7004 支持 fp16fp32 类型

    • +
    • 当前实现不包含 scale 与 bias,仅执行标准化操作

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_batchnorm_s(int8_t *input, int8_t *output, float *mean, float *variance, int unit, int channel, float epsilon, int core_mask)
    +
    + +
    +
    +void fp_batchnorm_s(float *input, float *output, float *mean, float *variance, int unit, int channel, float epsilon, int core_mask)
    +
    + +
    +
    +void hp_batchnorm_s(half *input, half *output, float *mean, float *variance, int unit, int channel, float epsilon, int core_mask)
    +
    + +

    C调用示例:

    +
     1// FT78NE 示例
    + 2#include <stdio.h>
    + 3#include <batchnorm.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    int8_t *input  = (int8_t *)0xA0000000;   // input 在 DDR 空间
    + 7    int8_t *output = (int8_t *)0xC0000000;
    + 8    float *mean    = (float *)0xA1000000;
    + 9    float *var     = (float *)0xA2000000;
    +10    int unit = 128;
    +11    int channel = 64;
    +12    float epsilon = 1e-5f;
    +13    int core_mask = 0xff;
    +14
    +15    i8_batchnorm_s(input, output, mean, var, unit, channel, epsilon, core_mask);
    +16    return 0;
    +17}
    +
    +
    +

    私有存储版本:

    +
    +
    +void i8_batchnorm_p(int8_t *input, int8_t *output, float *mean, float *variance, int unit, int channel, float epsilon)
    +
    + +
    +
    +void fp_batchnorm_p(float *input, float *output, float *mean, float *variance, int unit, int channel, float epsilon)
    +
    + +
    +
    +void hp_batchnorm_p(half *input, half *output, float *mean, float *variance, int unit, int channel, float epsilon)
    +
    + +

    C调用示例:

    +
     1// FT78NE 示例
    + 2#include <stdio.h>
    + 3#include <batchnorm.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    int8_t *input  = (int8_t *)0x10810000;   // input 在 L2 空间
    + 7    int8_t *output = (int8_t *)0x10820000;
    + 8    float *mean    = (float *)0x10830000;
    + 9    float *var     = (float *)0x10840000;
    +10    int unit = 128;
    +11    int channel = 64;
    +12    float epsilon = 1e-5f;
    +13
    +14    i8_batchnorm_p(input, output, mean, var, unit, channel, epsilon);
    +15    return 0;
    +16}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/batchnormgrad.html b/master/html/functionlib/dsplib/batchnormgrad.html new file mode 100644 index 0000000..7a098b5 --- /dev/null +++ b/master/html/functionlib/dsplib/batchnormgrad.html @@ -0,0 +1,445 @@ + + + + + + + + + Batchnormgrad — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Batchnormgrad

    +

    逐元素计算加法梯度

    +

    计算批标准化 (Batch Normalization) 的梯度。

    +

    该算子计算损失函数 L 分别对输入 x、缩放因子 scale (γ) 和偏置 bias (β) 的梯度。其中 bias 的梯度为 dbias

    +
    +\[\begin{split}dscale(\gamma) &= \sum_{i=1}^{m} dy_i \cdot \hat{x}_i \\ +dbias(\beta) &= \sum_{i=1}^{m} dy_i\end{split}\]
    +
    +\[dx_i = \frac{\gamma}{m\sqrt{\sigma^2 + \epsilon}} \left[ m \cdot dy_i - \sum_{j=1}^{m}dy_j - \hat{x}_i \sum_{j=1}^{m}dy_j \hat{x}_j \right]\]
    +

    其中 \(m\) 是批处理大小 (batch),\(\hat{x}\) 是归一化后的 \(x\)

    +
    +
    输入:
      +
    • x - 前向传播时的输入张量。

    • +
    • dy - 来自后一层的上游梯度。

    • +
    • mean - 前向传播时计算的均值。

    • +
    • invar - 前向传播时计算的逆方差 (1 / sqrt(variance + epsilon))。

    • +
    • scale - 前向传播时使用的缩放因子 (gamma, γ)。

    • +
    • batch - 批处理大小。

    • +
    • channel - 通道数。

    • +
    • is_train - 是否为训练模式。梯度计算通常在训练时进行。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • dx - 对输入 x 的梯度。

    • +
    • dbias - 对偏置 bias (β) 的梯度。

    • +
    • dscale - 对缩放因子 scale (γ) 的梯度。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32

    • +
    • MT7004 支持fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_batchnormgrad_s(float *x, float *dy, float *mean, float *invar, float *scale, int batch, int channel, int is_train, float *dx, float *dbias, float *dscale, int core_mask)
    +
    + +
    +
    +void hp_batchnormgrad_s(half *x, half *dy, half *mean, half *invar, half *scale, int batch, int channel, int is_train, half *dx, half *dbias, half *dscale, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <batchnormgrad.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *x = (float *)0xA0000000;          // forward input x
    + 6    float *dy = (float *)0xB0000000;         // upstream gradient dy
    + 7    float *mean = (float *)0xC0000000;       // forward mean
    + 8    float *invar = (float *)0xD0000000;      // forward inverse variance
    + 9    float *scale = (float *)0xE0000000;      // forward scale (gamma)
    +10
    +11    float *dx = (float *)0xA1000000;         // output gradient dx
    +12    float *dbias = (float *)0xB1000000;      // output gradient dbias
    +13    float *dscale = (float *)0xC1000000;     // output gradient dscale
    +14
    +15    int batch = 4;
    +16    int channel = 64;
    +17    int is_train = true;
    +18    int core_mask = 0xff;
    +19
    +20    fp_batchnormgrad_s(x, dy, mean, invar, scale, batch, channel, is_train, dx, dbias, dscale, core_mask);
    +21    return 0;
    +22}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_batchnormgrad_p(float *x, float *dy, float *mean, float *invar, float *scale, int batch, int channel, int is_train, float *dx, float *dbias, float *dscale)
    +
    + +
    +
    +void hp_batchnormgrad_p(half *x, half *dy, half *mean, half *invar, half *scale, int batch, int channel, int is_train, half *dx, half *dbias, half *dscale)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <batchnormgrad.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *x = (float *)0x10000000;          // forward input x in L2 space
    + 6    float *dy = (float *)0x10100000;         // upstream gradient dy
    + 7    float *mean = (float *)0x10200000;       // forward mean
    + 8    float *invar = (float *)0x10300000;      // forward inverse variance
    + 9    float *scale = (float *)0x10400000;      // forward scale (gamma)
    +10
    +11    float *dx = (float *)0x10500000;         // output gradient dx
    +12    float *dbias = (float *)0x10600000;      // output gradient dbias
    +13    float *dscale = (float *)0x10700000;     // output gradient dscale
    +14
    +15    int batch = 4;
    +16    int channel = 32;
    +17    int is_train = true;
    +18
    +19    fp_batchnormgrad_p(x, dy, mean, invar, scale, batch, channel, is_train, dx, dbias, dscale);
    +20    return 0;
    +21}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/batchtospace.html b/master/html/functionlib/dsplib/batchtospace.html index 1734c10..8e9df43 100644 --- a/master/html/functionlib/dsplib/batchtospace.html +++ b/master/html/functionlib/dsplib/batchtospace.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
    @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -295,7 +460,7 @@ W_{\text{out}} &= b_w \times W - c_{\text{left}} - c_{\text{right}}, \\
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/batchtospacend.html b/master/html/functionlib/dsplib/batchtospacend.html index 3998a3d..db736db 100644 --- a/master/html/functionlib/dsplib/batchtospacend.html +++ b/master/html/functionlib/dsplib/batchtospacend.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
      @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
    @@ -111,15 +276,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -144,7 +309,7 @@
    • \(j\) 对应输出通道,其范围为 \([0, C_{out}-1]\),其中 \(C_{out}\) 为输出通道数,该值也等于卷积核的个数。

    • \(k\) 对应输入通道数,其范围为 \([0, C_{in}-1]\),其中 \(C_{in}\) 为输入通道数,该值也等于卷积核的通道数。

    -

    因此,上面的公式中,\(bias(C_{out_j})\) 为第 \(j\) 个输出通道的偏置,\(weight(C_{out_j}, k)\) 表示第 \(j\) 个卷积核在第 \(k\) 个输入通道的卷积核切片,\(X(N_i, k)\) 为特征图第 \(i\) 个 batch 第 \(k\) 个输入通道的切片。卷积核 shape 为 \((\text{kernel_size}[0], \text{kernel_size}[1])\),其中 kernel_size[0] 和 kernel_size[1] 是卷积核的高度和宽度。若考虑到输入输出通道以及 group,则完整卷积核的 shape 为 \((C_{out}, \text{kernel_size}[0], \text{kernel_size}[1], C_{in}/\text{group})\),其中 group 是分组卷积时在通道上分割输入 \(x\) 的组数。

    +

    因此,上面的公式中,\(bias(C_{out_j})\) 为第 \(j\) 个输出通道的偏置,\(weight(C_{out_j}, k)\) 表示第 \(j\) 个卷积核在第 \(k\) 个输入通道的卷积核切片,\(X(N_i, k)\) 为特征图第 \(i\) 个 batch 第 \(k\) 个输入通道的切片。卷积核 shape 为 \((\text{kernel\_size}[0], \text{kernel\_size}[1])\),其中 kernel_size[0] 和 kernel_size[1] 是卷积核的高度和宽度。若考虑到输入输出通道以及 group,则完整卷积核的 shape 为 \((C_{out}, \text{kernel\_size}[0], \text{kernel\_size}[1], C_{in}/\text{group})\),其中 group 是分组卷积时在通道上分割输入 \(x\) 的组数。

    输入:
    • input_x - 输入数据的地址

    • @@ -349,7 +514,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/conv2d_transpose.html b/master/html/functionlib/dsplib/conv2d_transpose.html index b1b4313..1bbb4fa 100644 --- a/master/html/functionlib/dsplib/conv2d_transpose.html +++ b/master/html/functionlib/dsplib/conv2d_transpose.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
      @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -341,7 +506,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/conv2dbackpropfilterfusion.html b/master/html/functionlib/dsplib/conv2dbackpropfilterfusion.html index 94ed8a0..3af15a5 100644 --- a/master/html/functionlib/dsplib/conv2dbackpropfilterfusion.html +++ b/master/html/functionlib/dsplib/conv2dbackpropfilterfusion.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
      @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -291,7 +456,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/conv2dbackpropinputfusion.html b/master/html/functionlib/dsplib/conv2dbackpropinputfusion.html index 15757f4..bb97e53 100644 --- a/master/html/functionlib/dsplib/conv2dbackpropinputfusion.html +++ b/master/html/functionlib/dsplib/conv2dbackpropinputfusion.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
      @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    @@ -111,15 +276,15 @@
      -
    • - +
    • +
    • @@ -210,7 +375,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/crop_and_resize.html b/master/html/functionlib/dsplib/crop_and_resize.html index ea9bc2c..f253be0 100644 --- a/master/html/functionlib/dsplib/crop_and_resize.html +++ b/master/html/functionlib/dsplib/crop_and_resize.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
      @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
    @@ -111,15 +276,15 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -138,14 +303,14 @@
      \[\begin{split}\forall k \in [1, ids\_size], \quad \begin{cases} -\text{if } \textbf{is_regulated}[i_k] = 0, & +\text{if } \textbf{is\_regulated}[i\_k] = 0, & \begin{cases} - \displaystyle X_{i_k} \leftarrow - X_{i_k} \cdot \frac{\text{max_norm}} - {\sum_{j=1}^{layer\_size\_} X_{i_k, j}} \\[10pt] - \textbf{is_regulated}[i_k] \leftarrow 1 + \displaystyle X_{i\_k} \leftarrow + X_{i\_k} \cdot \frac{\text{max\_norm}} + {\sum_{j=1}^{layer\_size\_} X_{i\_k, j}} \\[10pt] + \textbf{is\_regulated}[i\_k] \leftarrow 1 \end{cases} \\[12pt] -\text{输出向量 } Y_k \leftarrow X_{i_k} +\text{输出向量 } Y_k \leftarrow X_{i\_k} \end{cases}\end{split}\]
      输入:
        @@ -250,7 +415,7 @@
        -

        © 版权所有 2025 - 2025, NUDT-674。

        +

        © 版权所有 2025 - 2026, NUDT-674。

        利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/equal.html b/master/html/functionlib/dsplib/equal.html index 8ce0def..84e7b96 100644 --- a/master/html/functionlib/dsplib/equal.html +++ b/master/html/functionlib/dsplib/equal.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
        @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    @@ -111,15 +276,15 @@
      -
    • - +
    • +
    • @@ -188,7 +353,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/expfusion.html b/master/html/functionlib/dsplib/expfusion.html index 94b0902..ad59184 100644 --- a/master/html/functionlib/dsplib/expfusion.html +++ b/master/html/functionlib/dsplib/expfusion.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
      @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -134,8 +299,7 @@

      ExpFusion

      -
      -

      传入一个数组,逐元素计算其乘上输入因子(可选择)后的指数值,再将指数值乘上输出因子后输出。

      +

      传入一个数组,逐元素计算其乘上输入因子(可选择)后的指数值,再将指数值乘上输出因子后输出。

      \[ \begin{align}\begin{aligned}\begin{split}dst_i = \exp(src_i \cdot s_{in}) \cdot s_{out} \quad \text{where} \quad @@ -169,7 +333,6 @@ s_{in} =
    • MT7004 支持fp16, fp32, int16, int32, cplx64

    -

    共享存储版本:

    @@ -209,7 +372,9 @@ s_{in} =
    void c128_expfusion_s(double *src_data, double *dst_data, int length, float in_scale, float out_scale, int scale, int core_mask)
    -

    C调用示例:

    +
    + +

    C调用示例:

     1//FT78NE示例
      2#include <stdio.h>
      3#include <expfusion.h>
    @@ -226,8 +391,6 @@ s_{in} =
     14}
     
    -
    -

    私有存储版本:

    @@ -267,7 +430,9 @@ s_{in} =
    void c128_expfusion_p(double *src_data, double *dst_data, int length, float in_scale, float out_scale, int scale)
    -

    C调用示例:

    +
    + +

    C调用示例:

     1//FT78NE示例
      2#include <stdio.h>
      3#include <expfusion.h>
    @@ -282,8 +447,6 @@ s_{in} =
     12}
     
    -
    -
    @@ -297,7 +460,7 @@ s_{in} =
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/fake_quant_with_min_max_vars.html b/master/html/functionlib/dsplib/fake_quant_with_min_max_vars.html new file mode 100644 index 0000000..2830707 --- /dev/null +++ b/master/html/functionlib/dsplib/fake_quant_with_min_max_vars.html @@ -0,0 +1,424 @@ + + + + + + + + + FakeQuantWithMinMaxVars — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    +
    + +
    +
    +
    +
    + +
    +

    FakeQuantWithMinMaxVars

    +

    对输入数据执行逐元素伪量化运算。该算子通过给定的最小/最大值(min_val/max_val)计算缩放因子(scale)和零点(zero_point),将浮点输入模拟量化到指定的整数范围(quant_min/quant_max),然后再将其反量化回浮点数。

    +
    +\[scale = \frac{max\_val - min\_val}{quant\_max - quant\_min}\]
    +
    +\[output_i = \left( \text{round} \left( \frac{\text{clamp}(input_i, nudge\_min, nudge\_max) - nudge\_min}{scale} \right) \right) \times scale + nudge\_min\]
    +
    +
    输入:
      +
    • src - 输入数据地址。

    • +
    • min_val - 浮点范围的最小值。

    • +
    • max_val - 浮点范围的最大值。

    • +
    • length - 计算长度。

    • +
    • quant_min - 量化后的整数最小值(例如 0 或 -128)。

    • +
    • quant_max - 量化后的整数最大值(例如 255 或 127)。

    • +
    • symmetric - 是否使用对称量化(bool 类型)。若为 true,则范围调整为关于 0 对称。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 伪量化后的计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持:fp32 (fp)

    • +
    • MT7004 支持:fp16 (hp), fp32 (fp)

    • +
    • 该算子内部包含 "Nudge" 逻辑,即会自动调整零点(Zero Point)使其为整数,并根据调整后的零点重新计算实际使用的浮点范围(nudge_min/nudge_max)。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_fake_quant_with_min_max_vars_s(float *src, float min_val, float max_val, float *output, int length, int quant_min, int quant_max, bool symmetric, int core_mask)
    +
    + +
    +
    +void hp_fake_quant_with_min_max_vars_s(half *src, half min_val, half max_val, half *output, int length, int quant_min, int quant_max, bool symmetric, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例:fp32 类型共享存储多核计算
    + 2#include <stdio.h>
    + 3#include <stdbool.h>
    + 4#include "78NE/utils.h"
    + 5
    + 6int main(int argc, char* argv[]) {
    + 7    float *input = (float *)0xA0000000;
    + 8    float *output = (float *)0xB0000000;
    + 9    float min_v = -10.0f;
    +10    float max_v = 10.0f;
    +11    int length = 960001;
    +12    int core_mask = 0b1011;
    +13    fp_fake_quant_with_min_max_vars_s(input, min_v, max_v, output, length, 0, 255, false, core_mask);
    +14    return 0;
    +15}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void fp_fake_quant_with_min_max_vars_p(float *src, float min_val, float max_val, float *output, int length, int quant_min, int quant_max, bool symmetric)
    +
    + +
    +
    +void hp_fake_quant_with_min_max_vars_p(half *src, half min_val, half max_val, half *output, int length, int quant_min, int quant_max, bool symmetric)
    +

    C调用示例:

    +
     1// MT7004 示例:fp16 (half) 类型私有存储单核计算
    + 2#include <stdio.h>
    + 3#include <stdbool.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    half *input = (half *)0x10000000;
    + 7    half *output = (half *)0x10001000;
    + 8    half min_v = (half)-5.0f;
    + 9    half max_v = (half)5.0f;
    +10    int length = 1024;
    +11    hp_fake_quant_with_min_max_vars_p(input, min_v, max_v, output, length, 0, 255, true);
    +12    return 0;
    +13}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.html b/master/html/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.html new file mode 100644 index 0000000..40f80bd --- /dev/null +++ b/master/html/functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.html @@ -0,0 +1,434 @@ + + + + + + + + + FakeQuantWithMinMaxVarsPerChannel — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    +
    + +
    +
    +
    +
    + +
    +

    FakeQuantWithMinMaxVarsPerChannel

    +

    对输入数据执行按通道(Per-Channel)的逐元素伪量化运算。该算子将输入数据划分为 channel_num 个等长的通道,每个通道根据其对应的最小/最大值(min_val[c] / max_val[c])独立计算缩放因子和零点,进行模拟量化与反量化。

    +
    +\[channel\_size = \frac{length}{channel\_num}\]
    +
    +\[\text{对于通道 } c: \quad scale_c = \frac{max\_val_c - min\_val_c}{quant\_max - quant\_min}\]
    +
    +\[output_{c,i} = \text{FakeQuant}(input_{c,i}, scale_c, nudge\_min_c)\]
    +
    +
    输入:
      +
    • src - 输入数据地址。

    • +
    • min_val - 每个通道最小值组成的数组地址。

    • +
    • max_val - 每个通道最大值组成的数组地址。

    • +
    • length - 输入数据总长度(需能被通道数整除)。

    • +
    • quant_min - 量化后的整数最小值。

    • +
    • quant_max - 量化后的整数最大值。

    • +
    • symmetric - 是否使用对称量化。

    • +
    • channel_num - 通道数量。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 伪量化后的计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持:fp32 (fp)

    • +
    • MT7004 支持:fp16 (hp), fp32 (fp)

    • +
    • 输入数据的总长度 length 必须可以被 channel_num 整除。

    • +
    • 每个通道的逻辑(包括 Nudge 零点调整)与单变量版本的伪量化一致。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_fake_quant_with_min_max_vars_per_channel_s(float *src, float *min_val, float *max_val, float *output, int length, int quant_min, int quant_max, bool symmetric, int channel_num, int core_mask)
    +
    + +
    +
    +void hp_fake_quant_with_min_max_vars_per_channel_s(half *src, half *min_val, half *max_val, half *output, int length, int quant_min, int quant_max, bool symmetric, int channel_num, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例:fp32 类型共享存储多核计算
    + 2#include <stdio.h>
    + 3#include <stdbool.h>
    + 4#include "78NE/utils.h"
    + 5
    + 6int main(int argc, char* argv[]) {
    + 7    float *input = (float *)0xA0000000;
    + 8    float *min_arr = (float *)0xA1000000;
    + 9    float *max_arr = (float *)0xA1001000;
    +10    float *output = (float *)0xB0000000;
    +11    int length = 960000;
    +12    int channel_num = 100;
    +13    int core_mask = 0b1011;
    +14
    +15    fp_fake_quant_with_min_max_vars_per_channel_s(input, min_arr, max_arr, output,
    +16                                                 length, 0, 255, false, channel_num, core_mask);
    +17    return 0;
    +18}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void fp_fake_quant_with_min_max_vars_per_channel_p(float *src, float *min_val, float *max_val, float *output, int length, int quant_min, int quant_max, bool symmetric, int channel_num)
    +
    + +
    +
    +void hp_fake_quant_with_min_max_vars_per_channel_p(half *src, half *min_val, half *max_val, half *output, int length, int quant_min, int quant_max, bool symmetric, int channel_num)
    +

    C调用示例:

    +
     1// MT7004 示例:fp16 (half) 类型私有存储单核计算
    + 2#include <stdio.h>
    + 3#include <stdbool.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    half *input = (half *)0x10000000;
    + 7    half *min_arr = (half *)0x10008000;
    + 8    half *max_arr = (half *)0x10008100;
    + 9    half *output = (half *)0x10009000;
    +10    int length = 2000;
    +11    int channel_num = 10;
    +12
    +13    hp_fake_quant_with_min_max_vars_per_channel_p(input, min_arr, max_arr, output,
    +14                                                 length, 0, 255, true, channel_num);
    +15    return 0;
    +16}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/fftimag.html b/master/html/functionlib/dsplib/fftimag.html new file mode 100644 index 0000000..b62adf0 --- /dev/null +++ b/master/html/functionlib/dsplib/fftimag.html @@ -0,0 +1,444 @@ + + + + + + + + + FFTImag — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    FFTImag

    +

    对输入复数序列虚数部分执行一维快速傅里叶变换(FFT)或逆变换(IFFT)。 +内部基于分治 FFT 算法, +并使用预计算旋转因子(twiddle factor)以提升性能。 +傅里叶变换,可以对参数进行调整,以实现FFT/IFFT/RFFT/IRFFT。

    +

    数学定义如下:

    +
    +\[X(k) = \sum_{n=0}^{N-1} x(n)\,e^{-j 2\pi kn / N} \quad (\text{Forward FFT})\]
    +
    +\[x(n) = \sum_{k=0}^{N-1} X(k)\,e^{j 2\pi kn / N} \quad (\text{Inverse FFT})\]
    +

    其中 \(N\) 为 FFT 点数。

    +
    +
    输入:
      +
    • input - 输入复数数据地址。

    • +
    • fft_size - FFT 点数。

    • +
    • +
      dir - 变换方向:
        +
      • FFT_FORWARD:正向 FFT

      • +
      • FFT_INVERSE:反向 FFT

      • +
      +
      +
      +
    • +
    • scratch_ptr - 临时缓冲区地址,用于存放旋转因子及中间计算结果。

    • +
    • twiddle - 旋转因子地址(仅共享存储版本使用)。

    • +
    • fft_size1 - 第一阶段 FFT 点数(仅共享存储版本使用)。

    • +
    • fft_size2 - 第二阶段 FFT 点数(仅共享存储版本使用)。

    • +
    • core_mask - 核掩码(仅共享存储版本需要)。

    • +
    +
    +
    输出:
      +
    • output - 输出复数序列地址。

    • +
    +
    +
    +

    内部核心计算公式如下: +对于FFT,它计算以下表达式:

    +
    +\[X[\omega_1, \dots, \omega_d] = + \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{-j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}},\]
    +

    其中, \(d\) = signal_ndim 是信号的维度,\(N_i\) 则是信号第 \(i\) 个维度的大小。

    +

    对于IFFT,它计算以下表达式:

    +
    +\[X[\omega_1, \dots, \omega_d] = + \frac{1}{\prod_{i=1}^d N_i} \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{\ j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}},\]
    +

    其中, \(d\) = signal_ndim 是信号的维度,\(N_i\) 则是信号第 \(i\) 维的大小。

    +
    +

    备注

    +
      +
    • FFT/IFFT要求complex64或complex128类型的输入,返回complex64或complex128类型的输出。

    • +
    • RFFT要求bool, uint8, int8, int16, int32, int64, float32或float64类型的输入, +返回complex64或complex128类型的输出。

    • +
    • IRFFT要求complex64或complex128类型的输入,返回float32或float64类型的输出。

    • +
    • 共享存储版本/私有存储版本支持函数及调用见RFFT。

    • +
    +
    +
    +
    参数:
      +
    • signal_ndim (int) - 表示每个信号中的维数,控制着傅里叶变换的维数,其值只能为1、2或3。

    • +
    • inverse (bool) - 表示该操作是否为逆变换,用以选择FFT 和 RFFT 或 IFFT 和 IRFFT。

      +
        +
      • 如果为 True ,则为IFFT 和 IRFFT。

      • +
      • 如果为 False ,FFT 和 RFFT。

      • +
      +
    • +
    • real (bool) - 表示该操作是否为实变换,与 inverse 共同决定具体的变换模式:

      +
        +
      • inverseFalserealFalse :对应FFT模式。

      • +
      • inverseTruerealFalse :对应IFFT模式。

      • +
      • inverseFalserealTrue :对应RFFT模式。

      • +
      • inverseTruerealTrue :对应IRFFT模式。

      • +
      +
    • +
    • norm (str,可选) - 表示该操作的规范化方式,可选值:[ "backward" , "forward" , "ortho" ]。默认值: "backward"

      +
        +
      • "backward",正向变换不缩放,逆变换按 \(1/n\) 缩放,其中 n 表示输入 x 的元素数量。。

      • +
      • "ortho",正向变换与逆变换均按 \(1/\sqrt n\) 缩放。

      • +
      • "forward",正向变换按 \(1/n\) 缩放,逆变换不缩放。

      • +
      +
    • +
    • onesided (bool,可选) - 控制输入是否减半以避免冗余。默认值: True

    • +
    • signal_sizes (tuple,可选) - 原始信号的大小(RFFT变换之前的信号,不包含batch这一维),只有在IRFFT模式下和设置 onesided 为True时需要该参数,需要满足 +以下条件。默认值: ()

      +
        +
      • signal_sizes 的长度等于IRFFT的 signal_ndim\(len(signal\_sizes)=signal\_ndim\)

      • +
      • signal_sizes 的最后一个维度除以2等于IRFFT输入的最后一个维度: \(signal\_size[-1]/2+1=x.shape[-1]\)

      • +
      • 除了最后一个维度外, signal_sizes 的维度与输入shape完全相同: \(signal\_sizes[:-1]=x.shape[:-1]\)

      • +
      +
    • +
    +
    +
    异常:
      +
    • TypeError - 如果FFT/IFFT/IRFF的输入类型不是以下类型之一:complex64、complex128。

    • +
    • TypeError - 如果输入的类型不是Tensor。

    • +
    • ValueError - 如果输入 x 的维度小于 signal_ndim

    • +
    • ValueError - 如果 signal_ndim 大于3或小于1。

    • +
    • ValueError - 如果 norm 取值不是"backward"、"forward"或"ortho"。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    类型支持:
      +
    • FT78NE:cplx64cplx128

    • +
    • MT7004:cplx64

    • +
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/fftreal.html b/master/html/functionlib/dsplib/fftreal.html new file mode 100644 index 0000000..8358636 --- /dev/null +++ b/master/html/functionlib/dsplib/fftreal.html @@ -0,0 +1,444 @@ + + + + + + + + + FFTReal — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    FFTReal

    +

    对输入复数序列实数部分执行一维快速傅里叶变换(FFT)或逆变换(IFFT)。 +内部基于分治 FFT 算法, +并使用预计算旋转因子(twiddle factor)以提升性能。 +傅里叶变换,可以对参数进行调整,以实现FFT/IFFT/RFFT/IRFFT。

    +

    数学定义如下:

    +
    +\[X(k) = \sum_{n=0}^{N-1} x(n)\,e^{-j 2\pi kn / N} \quad (\text{Forward FFT})\]
    +
    +\[x(n) = \sum_{k=0}^{N-1} X(k)\,e^{j 2\pi kn / N} \quad (\text{Inverse FFT})\]
    +

    其中 \(N\) 为 FFT 点数。

    +
    +
    输入:
      +
    • input - 输入复数数据地址。

    • +
    • fft_size - FFT 点数。

    • +
    • +
      dir - 变换方向:
        +
      • FFT_FORWARD:正向 FFT

      • +
      • FFT_INVERSE:反向 FFT

      • +
      +
      +
      +
    • +
    • scratch_ptr - 临时缓冲区地址,用于存放旋转因子及中间计算结果。

    • +
    • twiddle - 旋转因子地址(仅共享存储版本使用)。

    • +
    • fft_size1 - 第一阶段 FFT 点数(仅共享存储版本使用)。

    • +
    • fft_size2 - 第二阶段 FFT 点数(仅共享存储版本使用)。

    • +
    • core_mask - 核掩码(仅共享存储版本需要)。

    • +
    +
    +
    输出:
      +
    • output - 输出复数序列地址。

    • +
    +
    +
    +

    内部核心计算公式如下: +对于FFT,它计算以下表达式:

    +
    +\[X[\omega_1, \dots, \omega_d] = + \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{-j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}},\]
    +

    其中, \(d\) = signal_ndim 是信号的维度,\(N_i\) 则是信号第 \(i\) 个维度的大小。

    +

    对于IFFT,它计算以下表达式:

    +
    +\[X[\omega_1, \dots, \omega_d] = + \frac{1}{\prod_{i=1}^d N_i} \sum_{n_1=0}^{N_1-1} \dots \sum_{n_d=0}^{N_d-1} x[n_1, \dots, n_d] + e^{\ j\ 2 \pi \sum_{i=0}^d \frac{\omega_i n_i}{N_i}},\]
    +

    其中, \(d\) = signal_ndim 是信号的维度,\(N_i\) 则是信号第 \(i\) 维的大小。

    +
    +

    备注

    +
      +
    • FFT/IFFT要求complex64或complex128类型的输入,返回complex64或complex128类型的输出。

    • +
    • RFFT要求bool, uint8, int8, int16, int32, int64, float32或float64类型的输入, +返回complex64或complex128类型的输出。

    • +
    • IRFFT要求complex64或complex128类型的输入,返回float32或float64类型的输出。

    • +
    • 共享存储版本/私有存储版本支持函数及调用见RFFT。

    • +
    +
    +
    +
    参数:
      +
    • signal_ndim (int) - 表示每个信号中的维数,控制着傅里叶变换的维数,其值只能为1、2或3。

    • +
    • inverse (bool) - 表示该操作是否为逆变换,用以选择FFT 和 RFFT 或 IFFT 和 IRFFT。

      +
        +
      • 如果为 True ,则为IFFT 和 IRFFT。

      • +
      • 如果为 False ,FFT 和 RFFT。

      • +
      +
    • +
    • real (bool) - 表示该操作是否为实变换,与 inverse 共同决定具体的变换模式:

      +
        +
      • inverseFalserealFalse :对应FFT模式。

      • +
      • inverseTruerealFalse :对应IFFT模式。

      • +
      • inverseFalserealTrue :对应RFFT模式。

      • +
      • inverseTruerealTrue :对应IRFFT模式。

      • +
      +
    • +
    • norm (str,可选) - 表示该操作的规范化方式,可选值:[ "backward" , "forward" , "ortho" ]。默认值: "backward"

      +
        +
      • "backward",正向变换不缩放,逆变换按 \(1/n\) 缩放,其中 n 表示输入 x 的元素数量。。

      • +
      • "ortho",正向变换与逆变换均按 \(1/\sqrt n\) 缩放。

      • +
      • "forward",正向变换按 \(1/n\) 缩放,逆变换不缩放。

      • +
      +
    • +
    • onesided (bool,可选) - 控制输入是否减半以避免冗余。默认值: True

    • +
    • signal_sizes (tuple,可选) - 原始信号的大小(RFFT变换之前的信号,不包含batch这一维),只有在IRFFT模式下和设置 onesided 为True时需要该参数,需要满足 +以下条件。默认值: ()

      +
        +
      • signal_sizes 的长度等于IRFFT的 signal_ndim\(len(signal\_sizes)=signal\_ndim\)

      • +
      • signal_sizes 的最后一个维度除以2等于IRFFT输入的最后一个维度: \(signal\_size[-1]/2+1=x.shape[-1]\)

      • +
      • 除了最后一个维度外, signal_sizes 的维度与输入shape完全相同: \(signal\_sizes[:-1]=x.shape[:-1]\)

      • +
      +
    • +
    +
    +
    异常:
      +
    • TypeError - 如果FFT/IFFT/IRFF的输入类型不是以下类型之一:complex64、complex128。

    • +
    • TypeError - 如果输入的类型不是Tensor。

    • +
    • ValueError - 如果输入 x 的维度小于 signal_ndim

    • +
    • ValueError - 如果 signal_ndim 大于3或小于1。

    • +
    • ValueError - 如果 norm 取值不是"backward"、"forward"或"ortho"。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    类型支持:
      +
    • FT78NE:cplx64cplx128

    • +
    • MT7004:cplx64

    • +
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/fill.html b/master/html/functionlib/dsplib/fill.html new file mode 100644 index 0000000..bca93bb --- /dev/null +++ b/master/html/functionlib/dsplib/fill.html @@ -0,0 +1,488 @@ + + + + + + + + + Fill — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Fill

    +
    +

    使用给定的常量值填充输出数组。该算子将输入的 value 按指定数据类型复制,并在多核环境下利用 DMA 高效地完成填充操作。

    +
    +\[dst_i = value +\quad \text{for} \quad i = 0,1,\dots,N-1\]
    +
    +
    输入:
      +
    • value - 待填充的常量值地址。

    • +
    • param - Fill 参数结构体指针,描述输出张量的形状与数据类型信息。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 填充结果地址。

    • +
    +
    +
    FillParameter 说明:

    FillParameter 用于描述 Fill 算子的输出张量信息,其定义如下:

    +
    typedef struct FillParameter {
    +    int* shape_;      // 张量各维度大小
    +    int ndim_;        // 张量维度数
    +    int elem_cnt_;    // 张量元素总数
    +    int type_size_;   // 单个元素字节大小
    +} FillParameter;
    +
    +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp, dp, int8, int16, int32, cplx64, cplx128

    • +
    • MT7004 支持hp, fp, int16, int32, cplx64

    • +
    +
    +
    +

    共享存储版本:

    +
    +
    +void i8_fill_s(int8_t *value, int8_t *output, int core_mask, FillParameter *param)
    +
    + +
    +
    +void i16_fill_s(int16_t *value, int16_t *output, int core_mask, FillParameter *param)
    +
    + +
    +
    +void i32_fill_s(int32_t *value, int32_t *output, int core_mask, FillParameter *param)
    +
    + +
    +
    +void hp_fill_s(half *value, half *output, int core_mask, FillParameter *param)
    +
    + +
    +
    +void fp_fill_s(float *value, float *output, int core_mask, FillParameter *param)
    +
    + +
    +
    +void dp_fill_s(double *value, double *output, int core_mask, FillParameter *param)
    +
    + +
    +
    +void c64_fill_s(float *value, float *output, int core_mask, FillParameter *param)
    +
    + +
    +
    +void c128_fill_s(double *value, double *output, int core_mask, FillParameter *param)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <fill.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float value = 1.0f;
    + 7    float *output = (float *)0xC0000000;
    + 8    FillParameter param;
    + 9    param.elem_cnt_ = 1024;
    +10    param.type_size_ = sizeof(float);
    +11    int core_mask = 0xff;
    +12    fp_fill_s(&value, output, core_mask, &param);
    +13    return 0;
    +14}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_fill_p(int8_t *value, int8_t *output, FillParameter *param)
    +
    + +
    +
    +void i16_fill_p(int16_t *value, int16_t *output, FillParameter *param)
    +
    + +
    +
    +void i32_fill_p(int32_t *value, int32_t *output, FillParameter *param)
    +
    + +
    +
    +void hp_fill_p(half *value, half *output, FillParameter *param)
    +
    + +
    +
    +void fp_fill_p(float *value, float *output, FillParameter *param)
    +
    + +
    +
    +void dp_fill_p(double *value, double *output, FillParameter *param)
    +
    + +
    +
    +void c64_fill_p(float *value, float *output, FillParameter *param)
    +
    + +
    +
    +void c128_fill_p(double *value, double *output, FillParameter *param)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <fill.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float value = 1.0f;
    + 7    float *output = (float *)0x10820000;
    + 8    FillParameter param;
    + 9    param.elem_cnt_ = 1024;
    +10    param.type_size_ = sizeof(float);
    +11    fp_fill_p(&value, output, &param);
    +12    return 0;
    +13}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/fillv2.html b/master/html/functionlib/dsplib/fillv2.html index 998d9a5..83da1c0 100644 --- a/master/html/functionlib/dsplib/fillv2.html +++ b/master/html/functionlib/dsplib/fillv2.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
    @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -234,7 +399,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/floordiv.html b/master/html/functionlib/dsplib/floordiv.html index 80d6887..22add36 100644 --- a/master/html/functionlib/dsplib/floordiv.html +++ b/master/html/functionlib/dsplib/floordiv.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
      @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -192,9 +357,9 @@ 13 float epsilon = 1e-5f; 14 int core_mask = 0xff; 15 fp_fusedbatchnorm_s(input, scale, offset, mean, variance, -16 epsilon, channel, unit, core_mask, -17 output); -18 return 0; +16 epsilon, channel, unit, core_mask, +17 output); +18 return 0; 19}
    @@ -224,8 +389,8 @@ 12 int unit = 512; 13 float epsilon = 1e-4f; 14 hp_fusedbatchnorm_p(input, scale, offset, mean, variance, -15 epsilon, channel, unit, output); -16 return 0; +15 epsilon, channel, unit, output); +16 return 0; 17}
    @@ -244,7 +409,7 @@
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/gather.html b/master/html/functionlib/dsplib/gather.html new file mode 100644 index 0000000..d5968eb --- /dev/null +++ b/master/html/functionlib/dsplib/gather.html @@ -0,0 +1,494 @@ + + + + + + + + + Gather — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Gather

    +

    沿着给定的轴 axis,根据 indices 张量提供的索引值,从 input 张量中收集数据。支持 batch_dims 指定的批处理维度,即在前 batch_dims 个维度上,索引和输入是对应的。

    +
    +\[\begin{split}\text{output}[i_0, ..., i_{axis-1}, j_0, ..., j_{indices\_ndim-batch\_dims-1}, i_{axis+1}, ..., i_{input\_ndim-1}] = \\ +\text{input}[i_0, ..., i_{axis-1}, \text{indices}[i_0, ..., i_{batch\_dims-1}, j_0, ..., j_{indices\_ndim-batch\_dims-1}], i_{axis+1}, ..., i_{input\_ndim-1}]\end{split}\]
    +
    +
    输入:
      +
    • output - 计算结果输出地址。

    • +
    • input - 输入源张量数据地址。

    • +
    • input_shape - 输入张量的形状数组地址。

    • +
    • input_ndim - 输入张量的维度数量。

    • +
    • indices - 索引张量数据地址(通常为 int32 类型)。

    • +
    • indices_shape - 索引张量的形状数组地址。

    • +
    • indices_ndim - 索引张量的维度数量。

    • +
    • axis - 沿着哪个轴进行聚集操作。

    • +
    • batch_dims - 批处理维度数量。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 聚集后的计算结果。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128

    • +
    • MT7004 支持 fp16, fp32, int16, int32, cplx64

    • +
    • 索引张量 indices 内部存储的索引值必须在 [0, input_shape[axis]) 范围内,否则行为未定义。

    • +
    • 聚集操作涉及非连续访存,在大规模数据下建议使用共享存储版本并行处理。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_gather_s(int8_t *output, int8_t *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +
    + +
    +
    +void i16_gather_s(int16_t *output, int16_t *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +
    + +
    +
    +void i32_gather_s(int32_t *output, int32_t *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +
    + +
    +
    +void hp_gather_s(half *output, half *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +
    + +
    +
    +void fp_gather_s(float *output, float *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +
    + +
    +
    +void dp_gather_s(double *output, double *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +
    + +
    +
    +void c64_gather_s(float *output, float *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +
    + +
    +
    +void c128_gather_s(double *output, double *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例:多核并行聚集操作
    + 2#include <stdio.h>
    + 3#include "78NE/utils.h"
    + 4
    + 5int main() {
    + 6    float *input = (float *)0xA0000000;
    + 7    int *indices = (int *)0xB0000000;
    + 8    float *output = (float *)0xC0000000;
    + 9    int input_shape[] = {16, 800, 80};
    +10    int indices_shape[] = {16, 400};
    +11    int input_ndim = 3;
    +12    int indices_ndim = 2;
    +13    int axis = 1;
    +14    int batch_dims = 1;
    +15    int core_mask = 0xFF; // 使用8核并行
    +16
    +17    fp_gather_s(output, input, input_shape, input_ndim, indices, indices_shape, indices_ndim, axis, batch_dims, core_mask);
    +18    return 0;
    +19}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_gather_p(int8_t *output, int8_t *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +
    + +
    +
    +void i16_gather_p(int16_t *output, int16_t *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +
    + +
    +
    +void i32_gather_p(int32_t *output, int32_t *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +
    + +
    +
    +void hp_gather_p(half *output, half *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +
    + +
    +
    +void fp_gather_p(float *output, float *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +
    + +
    +
    +void dp_gather_p(double *output, double *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +
    + +
    +
    +void c64_gather_p(float *output, float *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +
    + +
    +
    +void c128_gather_p(double *output, double *input, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int axis, int batch_dims)
    +

    C调用示例:

    +
     1// MT7004 示例:单核聚集操作
    + 2#include <stdio.h>
    + 3
    + 4int main() {
    + 5    float *input = (float *)0x10000000;
    + 6    int *indices = (int *)0x10010000;
    + 7    float *output = (float *)0x10020000;
    + 8    int input_shape[] = {2, 100, 10};
    + 9    int indices_shape[] = {2, 50};
    +10    int input_ndim = 3;
    +11    int indices_ndim = 2;
    +12    int axis = 1;
    +13    int batch_dims = 1;
    +14
    +15    fp_gather_p(output, input, input_shape, input_ndim, indices, indices_shape, indices_ndim, axis, batch_dims);
    +16    return 0;
    +17}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/gather_nd.html b/master/html/functionlib/dsplib/gather_nd.html new file mode 100644 index 0000000..911a062 --- /dev/null +++ b/master/html/functionlib/dsplib/gather_nd.html @@ -0,0 +1,485 @@ + + + + + + + + + GatherNd — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    GatherNd

    +

    根据索引张量中的坐标,从输入张量中收集切片或元素。

    +

    GatherNd 允许根据 indices 提供的多维索引,在 input 张量的指定轴上进行切片提取。如果 indices 的最后一个维度长度为 K,则它代表从 input 的前 K 个维度中提取对应的子张量(切片)。

    +
    +
    输入:
      +
    • input - 输入数据张量地址。

    • +
    • output - 计算结果张量地址。

    • +
    • input_shape - 输入张量的形状数组地址。

    • +
    • input_ndim - 输入张量的维度数。

    • +
    • indices - 索引张量地址(类型固定为 int32)。

    • +
    • indices_shape - 索引张量的形状数组地址。

    • +
    • indices_ndim - 索引张量的维度数。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 收集后的数据存放地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128

    • +
    • MT7004 支持 fp16, fp32, int16, int32, cplx64

    • +
    • 索引张量 (indices) 在所有平台上均使用 int32 类型。

    • +
    • 坐标索引必须在输入张量维度的合法范围内,否则行为未定义。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_gather_nd_s(int8_t *input, int8_t *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +
    + +
    +
    +void i16_gather_nd_s(int16_t *input, int16_t *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +
    + +
    +
    +void i32_gather_nd_s(int *input, int *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +
    + +
    +
    +void hp_gather_nd_s(half *input, half *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +
    + +
    +
    +void fp_gather_nd_s(float *input, float *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +
    + +
    +
    +void dp_gather_nd_s(double *input, double *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +
    + +
    +
    +void c64_gather_nd_s(float *input, float *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +
    + +
    +
    +void c128_gather_nd_s(double *input, double *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例(共享存储)
    + 2#include <stdio.h>
    + 3#include "78NE/utils.h"
    + 4
    + 5int main() {
    + 6    float *input = (float *)0xA0000000;    // 输入在 DDR 空间
    + 7    float *output = (float *)0xB0000000;   // 输出在 DDR 空间
    + 8    int *indices = (int *)0xC0000000;      // 索引在 DDR 空间
    + 9    int input_shape[] = {10, 10, 5};
    +10    int indices_shape[] = {3, 2};          // 提取3个坐标,每个坐标深度为2
    +11    int input_ndim = 3;
    +12    int indices_ndim = 2;
    +13    int core_mask = 0xFF;
    +14
    +15    fp_gather_nd_s(input, output, input_shape, input_ndim, indices, indices_shape, indices_ndim, core_mask);
    +16    return 0;
    +17}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_gather_nd_p(int8_t *input, int8_t *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +
    + +
    +
    +void i16_gather_nd_p(int16_t *input, int16_t *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +
    + +
    +
    +void i32_gather_nd_p(int *input, int *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +
    + +
    +
    +void hp_gather_nd_p(half *input, half *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +
    + +
    +
    +void fp_gather_nd_p(float *input, float *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +
    + +
    +
    +void dp_gather_nd_p(double *input, double *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +
    + +
    +
    +void c64_gather_nd_p(float *input, float *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +
    + +
    +
    +void c128_gather_nd_p(double *input, double *output, int *input_shape, int input_ndim, int *indices, int *indices_shape, int indices_ndim)
    +

    C调用示例:

    +
     1// MT7004 示例(私有存储)
    + 2#include <stdio.h>
    + 3
    + 4int main() {
    + 5    float *input = (float *)0x10000000;
    + 6    float *output = (float *)0x10010000;
    + 7    int *indices = (int *)0x10020000;
    + 8    int input_shape[] = {4, 5, 6};
    + 9    int indices_shape[] = {2, 2, 2};
    +10    int input_ndim = 3;
    +11    int indices_ndim = 3;
    +12
    +13    fp_gather_nd_p(input, output, input_shape, input_ndim, indices, indices_shape, indices_ndim);
    +14    return 0;
    +15}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/gatherd.html b/master/html/functionlib/dsplib/gatherd.html new file mode 100644 index 0000000..bad40d0 --- /dev/null +++ b/master/html/functionlib/dsplib/gatherd.html @@ -0,0 +1,490 @@ + + + + + + + + + GatherD — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    GatherD

    +

    按照给定的维度 dim 和索引张量 index,从输入张量中按元素位置 +抽取数据,生成新的输出张量。

    +

    该算子等价于在指定维度上执行逐元素 Gather 操作,其输出形状与 +index 张量形状一致。

    +
    +\[\text{output}[i_0, \dots, i_n] += +\text{input}_x[i_0, \dots, i_{dim-1}, \text{index}[i_0, \dots, i_n], i_{dim+1}, \dots, i_n]\]
    +
    +
    输入:
      +
    • input_x - 输入张量的数据地址。 +数据类型需与所调用的 GatherD 接口类型一致。

    • +
    • dim - 指定进行 Gather 操作的维度索引,取值范围为 +[0, input_shape_size)

    • +
    • index - 索引张量的数据地址,类型为 int*, +用于指定在 dim 维度上的取值位置。

    • +
    • input_shape - 输入张量各维度大小数组地址。

    • +
    • input_shape_size - 输入张量的维度数量。

    • +
    • index_shape - 索引张量的形状数组地址, +其维度数量与 input_shape_size 相同。

    • +
    • core_mask - 核掩码(仅共享存储版本使用)。

    • +
    +
    +
    输出:
      +
    • output - 输出张量的数据地址, +其形状与 index_shape 保持一致, +数据类型与 input_x 相同。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • index 中的取值应满足 +0 <= index[...] < input_shape[dim]

    • +
    • 输出张量的元素个数等于 index_shape 各维度之积。

    • +
    • 该算子不对索引顺序进行任何排序或检查。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_gatherd_s(float *input_x, int dim, int *index, float *output, int *input_shape, int input_shape_size, int *index_shape, int core_mask)
    +
    + +
    +
    +void dp_gatherd_s(double *input_x, int dim, int *index, double *output, int *input_shape, int input_shape_size, int *index_shape, int core_mask)
    +
    + +
    +
    +void i8_gatherd_s(int8_t *input_x, int dim, int *index, int8_t *output, int *input_shape, int input_shape_size, int *index_shape, int core_mask)
    +
    + +
    +
    +void i16_gatherd_s(int16_t *input_x, int dim, int *index, int16_t *output, int *input_shape, int input_shape_size, int *index_shape, int core_mask)
    +
    + +
    +
    +void i32_gatherd_s(int32_t *input_x, int dim, int *index, int32_t *output, int *input_shape, int input_shape_size, int *index_shape, int core_mask)
    +
    + +
    +
    +void c64_gatherd_s(float *input_x, int dim, int *index, float *output, int *input_shape, int input_shape_size, int *index_shape, int core_mask)
    +
    + +
    +
    +void c128_gatherd_s(double *input_x, int dim, int *index, double *output, int *input_shape, int input_shape_size, int *index_shape, int core_mask)
    +
    + +

    C调用示例:

    +
     1// FT78NE 示例
    + 2#include <stdio.h>
    + 3#include <gatherd.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input_x = (float *)0xA0000000;   // input_x 在 DDR 空间
    + 7    float *output  = (float *)0xB0000000;
    + 8    int *index     = (int *)0xA1000000;
    + 9
    +10    int input_shape[] = {4, 8, 16};
    +11    int index_shape[] = {4, 8, 16};
    +12    int input_shape_size = 3;
    +13    int dim = 1;
    +14    int core_mask = 0xff;
    +15
    +16    fp_gatherd_s(input_x, dim, index, output, input_shape, input_shape_size, index_shape, core_mask);
    +17    return 0;
    +18}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_gatherd_p(float *input_x, int dim, int *index, float *output, int *input_shape, int input_shape_size, int *index_shape)
    +
    + +
    +
    +void dp_gatherd_p(double *input_x, int dim, int *index, double *output, int *input_shape, int input_shape_size, int *index_shape)
    +
    + +
    +
    +void i8_gatherd_p(int8_t *input_x, int dim, int *index, int8_t *output, int *input_shape, int input_shape_size, int *index_shape)
    +
    + +
    +
    +void i16_gatherd_p(int16_t *input_x, int dim, int *index, int16_t *output, int *input_shape, int input_shape_size, int *index_shape)
    +
    + +
    +
    +void i32_gatherd_p(int32_t *input_x, int dim, int *index, int32_t *output, int *input_shape, int input_shape_size, int *index_shape)
    +
    + +
    +
    +void c64_gatherd_p(float *input_x, int dim, int *index, float *output, int *input_shape, int input_shape_size, int *index_shape)
    +
    + +
    +
    +void c128_gatherd_p(double *input_x, int dim, int *index, double *output, int *input_shape, int input_shape_size, int *index_shape)
    +
    + +

    C调用示例:

    +
     1// MT7004 示例
    + 2#include <stdio.h>
    + 3#include <gatherd.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input_x = (float *)0x10000000;   // input_x 在 L2 空间
    + 7    float *output  = (float *)0x10010000;
    + 8    int *index     = (int *)0x10020000;
    + 9
    +10    int input_shape[] = {2, 4};
    +11    int index_shape[] = {2, 4};
    +12    int input_shape_size = 2;
    +13    int dim = 0;
    +14
    +15    fp_gatherd_p(input_x, dim, index, output, input_shape, input_shape_size, index_shape);
    +16    return 0;
    +17}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/glu.html b/master/html/functionlib/dsplib/glu.html new file mode 100644 index 0000000..1b181b3 --- /dev/null +++ b/master/html/functionlib/dsplib/glu.html @@ -0,0 +1,452 @@ + + + + + + + + + GLU — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    GLU

    +

    Gated Linear Unit(门控线性单元)算子。该算子将输入张量在指定维度(split_dim)上分割成两个相等的部分(A和B),然后对第二部分(B)应用Sigmoid激活函数,并将其与第一部分(A)逐元素相乘。

    +
    +\[\text{out} = A \otimes \sigma(B)\]
    +

    其中,A和B是输入张量沿 split_dim 维度分割后的两个子张量,\(\sigma\) 是Sigmoid函数,\(\otimes\) 表示逐元素乘法。

    +
    +
    输入:
      +
    • in_data - 输入张量的数据地址。

    • +
    • split_data - 一个指针数组,用于存放分割后的子张量地址(作为临时缓冲区)。

    • +
    • out_data - 输出张量的数据地址。

    • +
    • ndim - 输入张量的维度数量。

    • +
    • split_dim - 执行分割操作的目标维度轴。

    • +
    • input_shape - 指向一个整数数组的指针,该数组描述了输入张量的形状。

    • +
    • num_split - 分割的数量,对于GLU操作,此值应为2。

    • +
    • split_sizes - 指向一个整数数组的指针,该数组描述了每个子张量在 split_dim 维度上的大小。

    • +
    • strides - 指向一个整数数组的指针,用于存放为张量索引预先计算好的步长。

    • +
    • len - 输出张量中的元素总数。

    • +
    • core_mask - 核掩码 (仅共享存储版本需要)。

    • +
    +
    +
    输出:
      +
    • out_data - 存储GLU计算结果的张量。其形状与输入张量相同,但在 split_dim 维度上的大小减半。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE支持fp32和int8类型。

    • +
    • MT7004支持fp16和fp32类型。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_glu_s(int8_t *in_data, int8_t **split_data, int8_t *out_data, int ndim, int split_dim, int *input_shape, int num_split, int *split_sizes, int *strides, int len, int core_mask)
    +
    + +
    +
    +void hp_glu_s(half *in_data, half **split_data, half *out_data, int ndim, int split_dim, int *input_shape, int num_split, int *split_sizes, int *strides, int len, int core_mask)
    +
    + +
    +
    +void fp_glu_s(float *in_data, float **split_data, float *out_data, int ndim, int split_dim, int *input_shape, int num_split, int *split_sizes, int *strides, int len, int core_mask)
    +
    + +

    C调用示例:

    +
     1// 7004平台, fp32示例
    + 2#include <stdio.h>
    + 3#include "glu.h"
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *in_data = (float *)0xA0000000;      // input在DDR空间
    + 7    float *out_data = (float *)0xC0000000;     // output在DDR空间
    + 8    float *split_data_buf[2];                  // 临时缓冲区指针
    + 9    // ... (需要为split_data_buf分配内存)
    +10
    +11    // 假设张量属性已定义
    +12    int ndim = 2;
    +13    int split_dim = 1;
    +14    int input_shape[] = {2, 4};
    +15    int num_split = 2;
    +16    int split_sizes[] = {2, 2};
    +17    int len = 4;
    +18    int strides[2]; // 需要预先计算
    +19    int core_mask = 0xff;
    +20
    +21    fp_glu_s(in_data, split_data_buf, out_data, ndim, split_dim, input_shape, num_split, split_sizes, strides, len, core_mask);
    +22    return 0;
    +23}
    +
    +
    +

    私有存储版本:

    +
    +
    +void i8_glu_p(int8_t *in_data, int8_t **split_data, int8_t *out_data, int ndim, int split_dim, int *input_shape, int num_split, int *split_sizes, int *strides, int len)
    +
    + +
    +
    +void hp_glu_p(half *in_data, half **split_data, half *out_data, int ndim, int split_dim, int *input_shape, int num_split, int *split_sizes, int *strides, int len)
    +
    + +
    +
    +void fp_glu_p(float *in_data, float **split_data, float *out_data, int ndim, int split_dim, int *input_shape, int num_split, int *split_sizes, int *strides, int len)
    +
    + +

    C调用示例:

    +
     1// 7004平台, fp32示例
    + 2#include <stdio.h>
    + 3#include "glu.h"
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *in_data = (float *)0x10000000;     // input在L2空间
    + 7    float *out_data = (float *)0x10001000;    // output在L2空间
    + 8    float *split_data_buf[2];                 // 临时缓冲区指针
    + 9    // ... (需要为split_data_buf分配内存)
    +10
    +11    // 假设张量属性已定义
    +12    int ndim = 2;
    +13    int split_dim = 1;
    +14    int input_shape[] = {2, 4};
    +15    int num_split = 2;
    +16    int split_sizes[] = {2, 2};
    +17    int len = 4;
    +18    int strides[2]; // 需要预先计算
    +19
    +20    fp_glu_p(in_data, split_data_buf, out_data, ndim, split_dim, input_shape, num_split, split_sizes, strides, len);
    +21    return 0;
    +22}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/greater.html b/master/html/functionlib/dsplib/greater.html new file mode 100644 index 0000000..cf320fb --- /dev/null +++ b/master/html/functionlib/dsplib/greater.html @@ -0,0 +1,461 @@ + + + + + + + + + Greater — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Greater

    +

    逐元素比较两个输入张量,判断 input1 的元素是否大于 input2 的对应元素。支持广播。

    +
    +\[\text{output}_i = (\text{input1}_i > \text{input2}_i)\]
    +

    输出一个布尔张量,如果比较结果为真,则对应元素为 true,否则为 false

    +
    +
    输入:
      +
    • element_num - 输出张量的总元素数量。在广播情况下,等于较大输入张量的元素数量。

    • +
    • optimize - 广播优化标志。若为非0,则启用广播模式。

    • +
    • in_elements_num0 - 第一个输入张量 input1 的元素数量。用于广播判断。

    • +
    • input1 - 第一个输入张量的数据地址。

    • +
    • input2 - 第二个输入张量的数据地址。

    • +
    • output - (输出) 输出的布尔类型张量的数据地址。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • output - 写入比较结果的布尔张量。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32, fp64, int8, int16, int32

    • +
    • MT7004 支持fp16, fp32, int16, int32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_greater_s(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output, int core_mask)
    +
    + +
    +
    +void hp_greater_s(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output, int core_mask)
    +
    + +
    +
    +void dp_greater_s(int element_num, int optimize, int in_elements_num0, double *input1, double *input2, bool *output, int core_mask)
    +
    + +
    +
    +void i32_greater_s(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output, int core_mask)
    +
    + +
    +
    +void i16_greater_s(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output, int core_mask)
    +
    + +
    +
    +void i8_greater_s(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <greater.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input1 = (float *)0xA0000000;    // Scalar input, DDR
    + 6    float *input2 = (float *)0xB0000000;    // Tensor input
    + 7    bool *output = (bool *)0xC0000000;      // Output
    + 8
    + 9    // Broadcasting input1 (scalar) > input2 (tensor)
    +10    int element_num = 4096; // Size of input2 and output
    +11    int optimize = 1;       // Enable broadcasting
    +12    int in_elements_num0 = 1; // Size of input1
    +13    int core_mask = 0xff;
    +14
    +15    fp_greater_s(element_num, optimize, in_elements_num0, input1, input2, output, core_mask);
    +16    return 0;
    +17}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_greater_p(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output)
    +
    + +
    +
    +void hp_greater_p(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output)
    +
    + +
    +
    +void i32_greater_p(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output)
    +
    + +
    +
    +void i16_greater_p(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output)
    +
    + +
    +
    +void i8_greater_p(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <greater.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input1 = (float *)0x10000000;    // Tensor input, L2
    + 6    float *input2 = (float *)0x11000000;    // Scalar input
    + 7    bool *output = (bool *)0x12000000;      // Output
    + 8
    + 9    // Broadcasting input1 (tensor) > input2 (scalar)
    +10    int element_num = 1024; // Size of input1 and output
    +11    int optimize = 1;       // Enable broadcasting
    +12    int in_elements_num0 = 1024; // Size of input1
    +13
    +14    fp_greater_p(element_num, optimize, in_elements_num0, input1, input2, output);
    +15    return 0;
    +16}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/greaterequal.html b/master/html/functionlib/dsplib/greaterequal.html new file mode 100644 index 0000000..a14e1ec --- /dev/null +++ b/master/html/functionlib/dsplib/greaterequal.html @@ -0,0 +1,461 @@ + + + + + + + + + Greaterequal — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Greaterequal

    +

    逐元素比较两个输入张量,判断 input1 的元素是否大于或等于 input2 的对应元素。支持广播。

    +
    +\[\text{output}_i = (\text{input1}_i \ge \text{input2}_i)\]
    +

    输出一个布尔张量,如果比较结果为真,则对应元素为 true,否则为 false

    +
    +
    输入:
      +
    • element_num - 输出张量的总元素数量。在广播情况下,等于较大输入张量的元素数量。

    • +
    • optimize - 广播优化标志。若为非0,则启用广播模式。

    • +
    • in_elements_num0 - 第一个输入张量 input1 的元素数量。用于广播判断。

    • +
    • input1 - 第一个输入张量的数据地址。

    • +
    • input2 - 第二个输入张量的数据地址。

    • +
    • output - (输出) 输出的布尔类型张量的数据地址。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • output - 写入比较结果的布尔张量。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32, fp64, int8, int16, int32

    • +
    • MT7004 支持fp16, fp32, int16, int32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_greaterequal_s(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output, int core_mask)
    +
    + +
    +
    +void hp_greaterequal_s(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output, int core_mask)
    +
    + +
    +
    +void dp_greaterequal_s(int element_num, int optimize, int in_elements_num0, double *input1, double *input2, bool *output, int core_mask)
    +
    + +
    +
    +void i8_greaterequal_s(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output, int core_mask)
    +
    + +
    +
    +void i16_greaterequal_s(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output, int core_mask)
    +
    + +
    +
    +void i32_greaterequal_s(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <greaterequal.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input1 = (float *)0xA0000000;    // Scalar input, DDR
    + 6    float *input2 = (float *)0xB0000000;    // Tensor input
    + 7    bool *output = (bool *)0xC0000000;      // Output
    + 8
    + 9    // Broadcasting input1 (scalar) >= input2 (tensor)
    +10    int element_num = 4096; // Size of input2 and output
    +11    int optimize = 1;       // Enable broadcasting
    +12    int in_elements_num0 = 1; // Size of input1
    +13    int core_mask = 0xff;
    +14
    +15    fp_greaterequal_s(element_num, optimize, in_elements_num0, input1, input2, output, core_mask);
    +16    return 0;
    +17}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_greaterequal_p(int element_num, int optimize, int in_elements_num0, float *input1, float *input2, bool *output)
    +
    + +
    +
    +void hp_greaterequal_p(int element_num, int optimize, int in_elements_num0, half *input1, half *input2, bool *output)
    +
    + +
    +
    +void i32_greaterequal_p(int element_num, int optimize, int in_elements_num0, int32_t *input1, int32_t *input2, bool *output)
    +
    + +
    +
    +void i16_greaterequal_p(int element_num, int optimize, int in_elements_num0, int16_t *input1, int16_t *input2, bool *output)
    +
    + +
    +
    +void i8_greaterequal_p(int element_num, int optimize, int in_elements_num0, int8_t *input1, int8_t *input2, bool *output)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <greaterequal.h>
    + 4int main(int argc, char* argv[]) {
    + 5    float *input1 = (float *)0x10000000;    // Tensor input, L2
    + 6    float *input2 = (float *)0x11000000;    // Scalar input
    + 7    bool *output = (bool *)0x12000000;      // Output
    + 8
    + 9    // Broadcasting input1 (tensor) >= input2 (scalar)
    +10    int element_num = 1024; // Size of input1 and output
    +11    int optimize = 1;       // Enable broadcasting
    +12    int in_elements_num0 = 1024; // Size of input1
    +13
    +14    fp_greaterequal_p(element_num, optimize, in_elements_num0, input1, input2, output);
    +15    return 0;
    +16}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/groupnormfusion.html b/master/html/functionlib/dsplib/groupnormfusion.html index c165954..6460ce2 100644 --- a/master/html/functionlib/dsplib/groupnormfusion.html +++ b/master/html/functionlib/dsplib/groupnormfusion.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
    @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -197,9 +362,9 @@ 15 float epsilon = 1e-5f; 16 int core_mask = 0xff; 17 fp_groupnormfusion_s(input, scale, offset, mean, variance, -18 epsilon, num_groups, channel, unit, -19 batch, core_mask, output); -20 return 0; +18 epsilon, num_groups, channel, unit, +19 batch, core_mask, output); +20 return 0; 21}
    @@ -231,9 +396,9 @@ 14 int batch = 16; 15 float epsilon = 1e-4f; 16 hp_groupnormfusion_p(input, scale, offset, mean, variance, -17 epsilon, num_groups, channel, unit, -18 batch, output); -19 return 0; +17 epsilon, num_groups, channel, unit, +18 batch, output); +19 return 0; 20}
    @@ -252,7 +417,7 @@
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/gru.html b/master/html/functionlib/dsplib/gru.html index 340e1e5..efc6ddd 100644 --- a/master/html/functionlib/dsplib/gru.html +++ b/master/html/functionlib/dsplib/gru.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
    @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -173,9 +338,9 @@ 6 float end = 100.0f; 7 int length = 1001; 8 int core_mask = 0xff; - 9 fp_linspace_s(output, start, end, length, core_mask); -10 return 0; -11} + 9 fp_linspace_s(output, start, end, length, core_mask); +10 return 0; +11}
    @@ -192,9 +357,9 @@ 5 float start = 1.5f; 6 float step = 0.5f; 7 int num = 1000; - 8 fp_linspace_p(output, start, step, num); - 9 return 0; -10} + 8 fp_linspace_p(output, start, step, num); + 9 return 0; +10}
    @@ -212,7 +377,7 @@
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/log.html b/master/html/functionlib/dsplib/log.html new file mode 100644 index 0000000..e507641 --- /dev/null +++ b/master/html/functionlib/dsplib/log.html @@ -0,0 +1,437 @@ + + + + + + + + + Log — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Log

    +

    逐元素计算自然对数(以 \(e\) 为底)的对数函数。

    +
    +\[\text{output}_i = \ln(\text{input}_i)\]
    +

    其中 \(\ln(\cdot)\) 表示自然对数函数。 +当输入值小于等于 0 时,结果的行为依赖于具体平台和实现(可能产生 -infNaN)。

    +
    +
    输入:
      +
    • input - 输入张量的数据地址。

    • +
    • length - 输入张量的总元素数量。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • output - 输出张量的数据地址,其大小与 input 相同。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持的数据类型:fp32

    • +
    • MT7004 支持的数据类型:fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_log_s(float *input, float *output, int length, int core_mask)
    +
    + +
    +
    +void hp_log_s(half *input, half *output, int length, int core_mask)
    +
    + +
    +
    +void i16_log_s(int16_t *input, float *output, int length, int core_mask)
    +
    + +
    +
    +void i32_log_s(int *input, float *output, int length, int core_mask)
    +
    + +

    C调用示例:

    +
     1// FT78NE 多核示例
    + 2#include <stdio.h>
    + 3#include <log.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input  = (float *)0xA0000000;   // input 在 DDR 空间
    + 7    float *output = (float *)0xB0000000;   // output 在 DDR 空间
    + 8
    + 9    int length = 4096;
    +10    int core_mask = 0xff;
    +11
    +12    fp_log_s(input, output, length, core_mask);
    +13    return 0;
    +14}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_log_p(float *input, float *output, int length)
    +
    + +
    +
    +void hp_log_p(half *input, half *output, int length)
    +
    + +
    +
    +void i16_log_p(int16_t *input, float *output, int length)
    +
    + +
    +
    +void i32_log_p(int *input, float *output, int length)
    +
    + +

    C调用示例:

    +
     1// FT78NE 单核示例
    + 2#include <stdio.h>
    + 3#include <log.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input  = (float *)0x10000000;   // input 在 L2 空间
    + 7    float *output = (float *)0x11000000;   // output 在 L2 空间
    + 8
    + 9    int length = 1024;
    +10
    +11    fp_log_p(input, output, length);
    +12    return 0;
    +13}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/log1p.html b/master/html/functionlib/dsplib/log1p.html new file mode 100644 index 0000000..e1434c2 --- /dev/null +++ b/master/html/functionlib/dsplib/log1p.html @@ -0,0 +1,449 @@ + + + + + + + + + Log1p — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Log1p

    +

    逐元素计算 log(1 + x) 的值,其中 x 是输入元素。该函数在 x 的值接近于零时,比直接计算 log(1 + x) 能够提供更高的精度。

    +
    +\[output_i = \ln(1 + Input_i)\]
    +
    +
    输入:
      +
    • Input - 输入数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask - 核掩码(仅共享存储版本需要)。

    • +
    +
    +
    输出:
      +
    • Output - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持int8, int16, int32, fp32, fp64

    • +
    • MT7004 支持fp16, fp32, int16, int32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_log1p_s(int8_t *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void i16_log1p_s(int16_t *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void i32_log1p_s(int32_t *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void hp_log1p_s(half *Input, half *output, int length, int core_mask)
    +
    + +
    +
    +void fp_log1p_s(float *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void dp_log1p_s(double *Input, double *output, int length, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <log1p.h> // 假设头文件名为 log1p.h
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0xA0000000;   //input在DDR空间
    + 6    float *output = (float *)0xC0000000;
    + 7    int length = 1000;
    + 8    int core_mask = 0xff;
    + 9    fp_log1p_s(input, output, length, core_mask);
    +10    return 0;
    +11}
    +
    +
    +

    私有存储版本:

    +
    +
    +void i8_log1p_p(int8_t *Input, float *output, int length)
    +
    + +
    +
    +void i16_log1p_p(int16_t *Input, float *output, int length)
    +
    + +
    +
    +void i32_log1p_p(int32_t *Input, float *output, int length)
    +
    + +
    +
    +void hp_log1p_p(half *Input, half *output, int length)
    +
    + +
    +
    +void fp_log1p_p(float *Input, float *output, int length)
    +
    + +
    +
    +void dp_log1p_p(double *Input, double *output, int length)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <log1p.h> // 假设头文件名为 log1p.h
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0x10000000;   //input在L2空间
    + 6    float *output = (float *)0x10001000;
    + 7    int length = 1000;
    + 8    fp_log1p_p(input, output, length);
    + 9    return 0;
    +10}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/loggrad.html b/master/html/functionlib/dsplib/loggrad.html new file mode 100644 index 0000000..3cbdc86 --- /dev/null +++ b/master/html/functionlib/dsplib/loggrad.html @@ -0,0 +1,427 @@ + + + + + + + + + LogGrad — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    LogGrad

    +

    计算对数函数(Log)的梯度。

    +

    该算子用于反向传播阶段,依据链式法则, +将上游梯度与 Log 输入张量进行逐元素相除, +其数学表达式如下:

    +
    +\[\text{output}_i = \frac{\text{input0}_i}{\text{input1}_i}\]
    +
    +
    其中:
      +
    • input0 表示上游梯度(dY)

    • +
    • input1 表示 Log 算子的输入(X)

    • +
    +
    +
    输入:
      +
    • input0 - 上游梯度张量的数据地址。

    • +
    • input1 - Log 算子前向输入张量的数据地址。

    • +
    • length - 输入张量的总元素数量。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • output - 输出梯度张量的数据地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持的数据类型:fp32

    • +
    • MT7004 支持的数据类型:fp16, fp32

    • +
    • input1 中存在 0 或极小值时,结果可能产生 ±∞NaN,需由上层框架保证输入合法性

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_log_grad_s(float *input0, float *input1, float *output, int length, int core_mask)
    +
    + +
    +
    +void hp_log_grad_s(half *input0, half *input1, half *output, int length, int core_mask)
    +
    + +

    C调用示例:

    +
     1// FT78NE 多核示例
    + 2#include <stdio.h>
    + 3#include <log_grad.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input0 = (float *)0xA0000000;   // 上游梯度
    + 7    float *input1 = (float *)0xA0100000;   // Log 前向输入
    + 8    float *output = (float *)0xB0000000;   // 输出梯度
    + 9
    +10    int length = 4096;
    +11    int core_mask = 0xff;
    +12
    +13    fp_log_grad_s(input0, input1, output, length, core_mask);
    +14    return 0;
    +15}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_log_grad_p(float *input0, float *input1, float *output, int length)
    +
    + +
    +
    +void hp_log_grad_p(half *input0, half *input1, half *output, int length)
    +
    + +

    C调用示例:

    +
     1// MT7004 单核示例
    + 2#include <stdio.h>
    + 3#include <log_grad.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    half *input0 = (half *)0x10000000;
    + 7    half *input1 = (half *)0x10010000;
    + 8    half *output = (half *)0x10020000;
    + 9
    +10    int length = 1024;
    +11
    +12    hp_log_grad_p(input0, input1, output, length);
    +13    return 0;
    +14}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/logical_not.html b/master/html/functionlib/dsplib/logical_not.html new file mode 100644 index 0000000..7a30f60 --- /dev/null +++ b/master/html/functionlib/dsplib/logical_not.html @@ -0,0 +1,453 @@ + + + + + + + + + LogicalNot — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    LogicalNot

    +

    对输入数据执行逐元素逻辑非(Logical NOT)运算。对于输入中的每个元素,如果元素值为 0(即为 False),则输出结果为 1(或该类型的 True 值);如果元素值为非 0(即为 True),则输出为 0。

    +
    +\[output_i = \neg (input_i \neq 0)\]
    +
    +
    输入:
      +
    • input - 输入数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp)

    • +
    • MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp)

    • +
    • 输出结果的数据类型通常与输入数据类型保持一致。

    • +
    • 逻辑判断准则:0 值视为 False,任何非 0 值均视为 True。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_logical_not_s(int8_t *input, int8_t *output, int length, int core_mask)
    +
    + +
    +
    +void i16_logical_not_s(int16_t *input, int16_t *output, int length, int core_mask)
    +
    + +
    +
    +void i32_logical_not_s(int32_t *input, int32_t *output, int length, int core_mask)
    +
    + +
    +
    +void hp_logical_not_s(half *input, half *output, int length, int core_mask)
    +
    + +
    +
    +void fp_logical_not_s(float *input, float *output, int length, int core_mask)
    +
    + +
    +
    +void dp_logical_not_s(double *input, double *output, int length, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例(共享存储多核并行)
    + 2#include <stdio.h>
    + 3#include "78NE/utils.h"
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *in = (float *)0xA0000000;    // 输入在共享存储空间
    + 7    float *out = (float *)0xB0000000;   // 输出在共享存储空间
    + 8    int length = 960001;
    + 9    int core_mask = 0b1011;             // 使用指定的核心掩码
    +10    fp_logical_not_s(in, out, length, core_mask);
    +11    return 0;
    +12}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_logical_not_p(int8_t *input, int8_t *output, int length)
    +
    + +
    +
    +void i16_logical_not_p(int16_t *input, int16_t *output, int length)
    +
    + +
    +
    +void i32_logical_not_p(int32_t *input, int32_t *output, int length)
    +
    + +
    +
    +void hp_logical_not_p(half *input, half *output, int length)
    +
    + +
    +
    +void fp_logical_not_p(float *input, float *output, int length)
    +
    + +
    +
    +void dp_logical_not_p(double *input, double *output, int length)
    +

    C调用示例:

    +
     1// MT7004 示例(私有存储单核)
    + 2#include <stdio.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    // 输入和输出均位于私有存储空间
    + 6    int *in = (int *)0x10000000;
    + 7    int *out = (int *)0x10001000;
    + 8    int length = 1024;
    + 9    i32_logical_not_p(in, out, length);
    +10    return 0;
    +11}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/logical_or.html b/master/html/functionlib/dsplib/logical_or.html new file mode 100644 index 0000000..cdd95b7 --- /dev/null +++ b/master/html/functionlib/dsplib/logical_or.html @@ -0,0 +1,456 @@ + + + + + + + + + LogicalOr — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    LogicalOr

    +

    对两个输入数据执行逐元素逻辑或(Logical OR)运算。对于输入中的每个元素,如果两个对应的输入元素中至少有一个不为 0(即为 True),则输出结果为 1(或该类型的 True 值);如果两者均为 0,则输出为 0。

    +
    +\[output_i = (input0_i \neq 0) \lor (input1_i \neq 0)\]
    +
    +
    输入:
      +
    • input0 - 第一个输入数据地址。

    • +
    • input1 - 第二个输入数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp)

    • +
    • MT7004 支持:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp)

    • +
    • 输出结果的数据类型通常与输入数据类型保持一致。

    • +
    • 逻辑判断准则:非 0 值视为 True,0 值视为 False。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_logical_or_s(int8_t *input0, int8_t *input1, int8_t *output, int length, int core_mask)
    +
    + +
    +
    +void i16_logical_or_s(int16_t *input0, int16_t *input1, int16_t *output, int length, int core_mask)
    +
    + +
    +
    +void i32_logical_or_s(int32_t *input0, int32_t *input1, int32_t *output, int length, int core_mask)
    +
    + +
    +
    +void hp_logical_or_s(half *input0, half *input1, half *output, int length, int core_mask)
    +
    + +
    +
    +void fp_logical_or_s(float *input0, float *input1, float *output, int length, int core_mask)
    +
    + +
    +
    +void dp_logical_or_s(double *input0, double *input1, double *output, int length, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例(共享存储多核并行)
    + 2#include <stdio.h>
    + 3#include "78NE/utils.h"
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    int32_t *in0 = (int32_t *)0xA0000000;   // 输入0在共享存储空间
    + 7    int32_t *in1 = (int32_t *)0xA1000000;   // 输入1在共享存储空间
    + 8    int32_t *out = (int32_t *)0xB0000000;   // 输出在共享存储空间
    + 9    int length = 960001;
    +10    int core_mask = 0b1011;                 // 指定参加计算的核心
    +11    i32_logical_or_s(in0, in1, out, length, core_mask);
    +12    return 0;
    +13}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_logical_or_p(int8_t *input0, int8_t *input1, int8_t *output, int length)
    +
    + +
    +
    +void i16_logical_or_p(int16_t *input0, int16_t *input1, int16_t *output, int length)
    +
    + +
    +
    +void i32_logical_or_p(int32_t *input0, int32_t *input1, int32_t *output, int length)
    +
    + +
    +
    +void hp_logical_or_p(half *input0, half *input1, half *output, int length)
    +
    + +
    +
    +void fp_logical_or_p(float *input0, float *input1, float *output, int length)
    +
    + +
    +
    +void dp_logical_or_p(double *input0, double *input1, double *output, int length)
    +

    C调用示例:

    +
     1// MT7004 示例(私有存储单核)
    + 2#include <stdio.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    // 输入和输出均位于私有存储空间
    + 6    float *in0 = (float *)0x10000000;
    + 7    float *in1 = (float *)0x10001000;
    + 8    float *out = (float *)0x10002000;
    + 9    int length = 1024;
    +10    fp_logical_or_p(in0, in1, out, length);
    +11    return 0;
    +12}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/logicaland.html b/master/html/functionlib/dsplib/logicaland.html new file mode 100644 index 0000000..a1b8fb8 --- /dev/null +++ b/master/html/functionlib/dsplib/logicaland.html @@ -0,0 +1,466 @@ + + + + + + + + + LogicalAnd — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    LogicalAnd

    +

    逐元素执行逻辑与(Logical AND)运算。

    +

    对输入张量的对应元素先进行布尔化判断,再执行逻辑与运算,输出结果为 0 或 1, +并以与输入相同的数据类型返回。

    +
    +\[\begin{split}\text{output}_i = +\begin{cases} + 1, & \text{if } (\text{input0}_i \neq 0) \land (\text{input1}_i \neq 0) \\ + 0, & \text{otherwise} +\end{cases}\end{split}\]
    +
    +
    输入:
      +
    • input0 - 第一个输入张量的数据地址。

    • +
    • input1 - 第二个输入张量的数据地址。

    • +
    • length - 输入张量的总元素数量。

    • +
    • core_mask - 核掩码。

    • +
    +
    +
    输出:
      +
    • output - 输出张量的数据地址,其大小与 input0input1 相同。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持的数据类型:fp32, fp64, int8, int16, int32

    • +
    • MT7004 支持的数据类型:fp16, fp32, int16, int32

    • +
    • 输入值在参与逻辑运算前会被转换为布尔值(非 0 为 true,0 为 false)

    • +
    • 输出结果仅为 0 或 1

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_and_s(float *input0, float *input1, float *output, int length, int core_mask)
    +
    + +
    +
    +void dp_and_s(double *input0, double *input1, double *output, int length, int core_mask)
    +
    + +
    +
    +void i8_and_s(int8_t *input0, int8_t *input1, int8_t *output, int length, int core_mask)
    +
    + +
    +
    +void i16_and_s(int16_t *input0, int16_t *input1, int16_t *output, int length, int core_mask)
    +
    + +
    +
    +void i32_and_s(int32_t *input0, int32_t *input1, int32_t *output, int length, int core_mask)
    +
    + +
    +
    +void hp_and_s(half *input0, half *input1, half *output, int length, int core_mask)
    +
    + +

    C调用示例:

    +
     1// FT78NE 多核示例
    + 2#include <stdio.h>
    + 3#include <logical_and.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input0 = (float *)0xA0000000;   // input0 在 DDR 空间
    + 7    float *input1 = (float *)0xA0010000;   // input1 在 DDR 空间
    + 8    float *output = (float *)0xB0000000;   // output 在 DDR 空间
    + 9
    +10    int length = 4096;
    +11    int core_mask = 0xff;
    +12
    +13    fp_and_s(input0, input1, output, length, core_mask);
    +14    return 0;
    +15}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_and_p(float *input0, float *input1, float *output, int length)
    +
    + +
    +
    +void dp_and_p(double *input0, double *input1, double *output, int length)
    +
    + +
    +
    +void i8_and_p(int8_t *input0, int8_t *input1, int8_t *output, int length)
    +
    + +
    +
    +void i16_and_p(int16_t *input0, int16_t *input1, int16_t *output, int length)
    +
    + +
    +
    +void i32_and_p(int32_t *input0, int32_t *input1, int32_t *output, int length)
    +
    + +
    +
    +void hp_and_p(half *input0, half *input1, half *output, int length)
    +
    + +

    C调用示例:

    +
     1// MT7004 单核示例
    + 2#include <stdio.h>
    + 3#include <logical_and.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    half *input0 = (half *)0x10000000;   // input0 在 L2 空间
    + 7    half *input1 = (half *)0x10002000;   // input1 在 L2 空间
    + 8    half *output = (half *)0x10010000;   // output 在 L2 空间
    + 9
    +10    int length = 1024;
    +11
    +12    hp_and_p(input0, input1, output, length);
    +13    return 0;
    +14}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/logsoftmax.html b/master/html/functionlib/dsplib/logsoftmax.html new file mode 100644 index 0000000..9585a49 --- /dev/null +++ b/master/html/functionlib/dsplib/logsoftmax.html @@ -0,0 +1,439 @@ + + + + + + + + + LogSoftmax — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    LogSoftmax

    +

    对输入数组沿指定维度进行 Log-Softmax 计算,输出每个元素的对数概率值。

    +
    +\[\text{output}_{i} = \log\frac{\exp(\text{input}_{i})}{\sum_j \exp(\text{input}_j)} +\quad \text{for elements along the given axis}\]
    +
    +
    输入:
      +
    • input_ptr - 输入数据地址。

    • +
    • axis - 归一化的轴。

    • +
    • n_dim - 输入张量维度。

    • +
    • inner_size - 内部尺寸(轴之后的元素个数)。

    • +
    • outter_size - 外部尺寸(轴之前的元素个数)。

    • +
    • axis_size - 归一化轴的元素数量。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    • sum_data - 中间累加存储地址(用于存放指数和)。

    • +
    +
    +
    输出:
      +
    • output_ptr - Log-Softmax 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp, int8

    • +
    • MT7004 支持hp, fp

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_logsoftmax_s(float *input_ptr, float *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask)
    +
    + +
    +
    +void hp_logsoftmax_s(half *input_ptr, half *output_ptr, half *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask)
    +
    + +
    +
    +void i8_logsoftmax_s(int8_t *input_ptr, int8_t *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <logsoftmax.h>
    + 4
    + 5int main() {
    + 6    float *input = (float *)0xA0000000;   // input在DDR空间
    + 7    float *output = (float *)0xC0000000;
    + 8    float *sum_data = (float *)0xD0000000;
    + 9    int axis = 1;
    +10    int n_dim = 3;
    +11    int inner_size = 4;
    +12    int outter_size = 2;
    +13    int axis_size = 3;
    +14    int core_mask = 0xff;
    +15
    +16    fp_logsoftmax_s(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size, core_mask);
    +17    return 0;
    +18}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_logsoftmax_p(float *input_ptr, float *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size)
    +
    + +
    +
    +void hp_logsoftmax_p(half *input_ptr, half *output_ptr, half *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size)
    +
    + +
    +
    +void i8_logsoftmax_p(int8_t *input_ptr, int8_t *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <logsoftmax.h>
    + 4
    + 5int main() {
    + 6    float *input = (float *)0x10810000;   // input在L2空间
    + 7    float *output = (float *)0x10820000;
    + 8    float *sum_data = (float *)0x10830000;
    + 9    int axis = 1;
    +10    int n_dim = 3;
    +11    int inner_size = 4;
    +12    int outter_size = 2;
    +13    int axis_size = 3;
    +14
    +15    fp_logsoftmax_p(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size);
    +16    return 0;
    +17}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/lpnormalization.html b/master/html/functionlib/dsplib/lpnormalization.html new file mode 100644 index 0000000..d62c9ef --- /dev/null +++ b/master/html/functionlib/dsplib/lpnormalization.html @@ -0,0 +1,447 @@ + + + + + + + + + LpNormalization — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    LpNormalization

    +

    对输入张量按 实例(Instance)+ 通道(Channel) 维度执行 Lp 归一化操作。 +该算子在每个样本的每个通道内,基于 inner_size 维度计算 Lp 范数, +并结合可学习参数 gammabeta 完成缩放与偏移。

    +
    +\[ \begin{align}\begin{aligned}\text{norm}_{b,c} = \left( \sum_{i=1}^{N} |x_{b,c,i}|^p + \epsilon \right)^{\frac{1}{p}}\\y_{b,c,i} = \frac{x_{b,c,i}}{\text{norm}_{b,c}} \cdot \gamma_c + \beta_c\end{aligned}\end{align} \]
    +

    其中:

    +
      +
    • \(b\) 表示 batch 维度

    • +
    • \(c\) 表示通道维度

    • +
    • \(i\) 表示 inner_size 维度

    • +
    • \(p\) 为范数阶数

    • +
    • \(\gamma_c\)\(\beta_c\) 为通道级缩放与偏移参数

    • +
    +
    +
    输入:
      +
    • input - 输入数据地址,形状为 [batch, channel, inner_size]

    • +
    • gamma - 缩放参数地址,长度为 channel

    • +
    • beta - 偏移参数地址,长度为 channel

    • +
    • p - Lp 范数阶数。

    • +
    • batch - batch 数。

    • +
    • channel - 通道数。

    • +
    • inner_size - 每个通道内的归一化长度。

    • +
    • epsilon - 数值稳定因子。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - LpNormalization 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 fp32 类型

    • +
    • MT7004 支持 fp16fp32 类型

    • +
    • 归一化统计量仅在单个样本、单个通道内计算

    • +
    • p 为浮点数,可用于 L1、L2 等不同范数形式

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_lpnorm_s(float *input, float *gamma, float *beta, float *output, float p, int batch, int channel, int inner_size, float epsilon, int core_mask)
    +
    + +
    +
    +void hp_lpnorm_s(half *input, half *gamma, half *beta, half *output, float p, int batch, int channel, int inner_size, float epsilon, int core_mask)
    +
    + +

    C调用示例:

    +
     1// FT78NE 示例
    + 2#include <stdio.h>
    + 3#include <lpnorm.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input  = (float *)0xA0000000;   // input 在 DDR 空间
    + 7    float *output = (float *)0xC0000000;
    + 8    float *gamma  = (float *)0xA1000000;
    + 9    float *beta   = (float *)0xA2000000;
    +10
    +11    int batch = 4;
    +12    int channel = 64;
    +13    int inner_size = 256;
    +14    float p = 2.0f;
    +15    float epsilon = 1e-6f;
    +16    int core_mask = 0xff;
    +17
    +18    fp_lpnorm_s(input, gamma, beta, output,
    +19                p, batch, channel, inner_size, epsilon, core_mask);
    +20    return 0;
    +21}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_lpnorm_p(float *input, float *gamma, float *beta, float *output, float p, int batch, int channel, int inner_size, float epsilon)
    +
    + +
    +
    +void hp_lpnorm_p(half *input, half *gamma, half *beta, half *output, float p, int batch, int channel, int inner_size, float epsilon)
    +
    + +

    C调用示例:

    +
     1// FT78NE 示例
    + 2#include <stdio.h>
    + 3#include <lpnorm.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input  = (float *)0x10810000;   // input 在 L2 空间
    + 7    float *output = (float *)0x10820000;
    + 8    float *gamma  = (float *)0x10830000;
    + 9    float *beta   = (float *)0x10840000;
    +10
    +11    int batch = 4;
    +12    int channel = 64;
    +13    int inner_size = 256;
    +14    float p = 2.0f;
    +15    float epsilon = 1e-6f;
    +16
    +17    fp_lpnorm_p(input, gamma, beta, output,
    +18                p, batch, channel, inner_size, epsilon);
    +19    return 0;
    +20}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/lrn.html b/master/html/functionlib/dsplib/lrn.html new file mode 100644 index 0000000..b64d350 --- /dev/null +++ b/master/html/functionlib/dsplib/lrn.html @@ -0,0 +1,425 @@ + + + + + + + + + Lrn — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Lrn

    +

    执行局部响应归一化 (Local Response Normalization)。

    +
    +\[\text{output}_{i,j} = \text{input}_{i,j} \times \left( \text{bias} + \alpha \sum_{k=\max(0, j-d)}^{\min(C-1, j+d)} (\text{input}_{i,k})^2 \right)^{-\beta}\]
    +

    其中,i 遍历每个 out_size 维度,j 表示当前通道,d 是 depth_radius,C 是总通道数 channel, :math:alpha 和 :math:beta 是缩放因子。

    +
    +
    输入:
      +
    • input - 输入张量数据地址,其逻辑布局为 (out_size, channel)。

    • +
    • out_size - 空间/批处理维度的乘积 (例如 N*H*W)。

    • +
    • channel - 通道数 (C)。

    • +
    • depth_radius - 归一化窗口的半径。总的窗口大小为 2 * depth_radius + 1。

    • +
    • alpha - 缩放因子 :math:alpha。

    • +
    • beta - 指数 :math:beta。

    • +
    • bias - 偏置项。

    • +
    • core_mask - 核掩码(仅共享存储版本需要)。

    • +
    +
    +
    输出:
      +
    • Output - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32

    • +
    • MT7004 支持fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void hp_lrn_s(half *input, half *output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias, int core_mask)
    +
    + +
    +
    +void fp_lrn_s(float *input, float *output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <lrn.h> // 假设头文件名为 lrn.h
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0xA0000000;   //input在DDR空间
    + 6    float *output = (float *)0xC0000000;
    + 7    int out_size = 256; // 例如 N*H*W
    + 8    int channel = 96;
    + 9    int depth_radius = 5;
    +10    float alpha = 0.0001f;
    +11    float beta = 0.75f;
    +12    float bias = 1.0f;
    +13    int core_mask = 0xff;
    +14    fp_lrn_s(input, output, out_size, channel, depth_radius, alpha, beta, bias, core_mask);
    +15    return 0;
    +16}
    +
    +
    +

    私有存储版本:

    +
    +
    +void hp_lrn_p(half *input, half *output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias)
    +
    + +
    +
    +void fp_lrn_p(float *input, float *output, int out_size, int channel, int depth_radius, float alpha, float beta, float bias)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <lrn.h> // 假设头文件名为 lrn.h
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0x10000000;   //input在L2空间
    + 6    float *output = (float *)0x10010000;
    + 7    int out_size = 256; // 例如 N*H*W
    + 8    int channel = 96;
    + 9    int depth_radius = 5;
    +10    float alpha = 0.0001f;
    +11    float beta = 0.75f;
    +12    float bias = 1.0f;
    +13    fp_lrn_p(input, output, out_size, channel, depth_radius, alpha, beta, bias);
    +14    return 0;
    +15}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/lsh_projection.html b/master/html/functionlib/dsplib/lsh_projection.html new file mode 100644 index 0000000..6524324 --- /dev/null +++ b/master/html/functionlib/dsplib/lsh_projection.html @@ -0,0 +1,473 @@ + + + + + + + + + LshProjection — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    LshProjection

    +

    局部敏感哈希(Locality Sensitive Hashing, LSH)投影算子。该算子通过多个哈希组(Hash Groups)对输入特征进行处理,每组生成一个指定位宽(bits_per_hash)的哈希签名(int32)。

    +

    内部逻辑基于 FNV1a 哈希算法和加权评分机制,将高维特征映射为低维的离散哈希值。

    +
    +
    计算过程:
      +
    1. 对于每个哈希组 $i$,循环执行 $j$ 次($j < bits_per_hash$)。

    2. +
    3. 每次根据特定的哈希种子 $seed_{i,j}$ 计算特征与权重的加权评分,并通过评分符号决定一个 Bit 位(0 或 1)。

    4. +
    5. 将生成的 Bit 位拼接成一个完整的 32 位整型哈希签名。

    6. +
    +
    +
    输入:
      +
    • hash_seed - 哈希种子数组地址(float 类型)。

    • +
    • feature - 输入特征向量地址(int32 类型)。

    • +
    • weight - 权重向量地址(类型随算子名而定,若为 NULL 则不加权)。

    • +
    • hash_group_num - 生成的哈希组数量(即输出结果的个数)。

    • +
    • bits_per_hash - 每组哈希签名的有效位数(通常 $le 32$)。

    • +
    • feature_num - 特征向量的维度。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 存储生成的哈希签名地址(int32 数组)。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持权重类型:int8 (i8), int16 (i16), int32 (i32), fp32 (fp), fp64 (dp)

    • +
    • MT7004 支持权重类型:int16 (i16), int32 (i32), fp16 (hp), fp32 (fp)

    • +
    • 特征输入(feature)在所有平台上固定为 int32 类型。

    • +
    • 输出结果(output)在所有平台上固定为 int32 类型。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_lsh_projection_s(const float *hash_seed, const int32_t *feature, const int8_t *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask)
    +
    + +
    +
    +void i16_lsh_projection_s(const float *hash_seed, const int32_t *feature, const int16_t *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask)
    +
    + +
    +
    +void i32_lsh_projection_s(const float *hash_seed, const int32_t *feature, const int32_t *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask)
    +
    + +
    +
    +void hp_lsh_projection_s(const float *hash_seed, const int32_t *feature, const half *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask)
    +
    + +
    +
    +void fp_lsh_projection_s(const float *hash_seed, const int32_t *feature, const float *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask)
    +
    + +
    +
    +void dp_lsh_projection_s(const float *hash_seed, const int32_t *feature, const double *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例:fp32 权重,多核共享存储
    + 2#include "78NE/utils.h"
    + 3
    + 4int main() {
    + 5    float* hash_seed = (float*)0xA0000000;
    + 6    int32_t* feature = (int32_t*)0xA1000000;
    + 7    float* weight = (float*)0xA2000000;
    + 8    int32_t* output = (int32_t*)0xB0000000;
    + 9
    +10    int hash_group_num = 128;
    +11    int bits_per_hash = 16;
    +12    int feature_num = 64;
    +13    int core_mask = 0xFF;
    +14
    +15    fp_lsh_projection_s(hash_seed, feature, weight, output,
    +16                        hash_group_num, bits_per_hash, feature_num, core_mask);
    +17    return 0;
    +18}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_lsh_projection_p(const float *hash_seed, const int32_t *feature, const int8_t *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num)
    +
    + +
    +
    +void i16_lsh_projection_p(const float *hash_seed, const int32_t *feature, const int16_t *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num)
    +
    + +
    +
    +void i32_lsh_projection_p(const float *hash_seed, const int32_t *feature, const int32_t *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num)
    +
    + +
    +
    +void hp_lsh_projection_p(const float *hash_seed, const int32_t *feature, const half *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num)
    +
    + +
    +
    +void fp_lsh_projection_p(const float *hash_seed, const int32_t *feature, const float *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num)
    +
    + +
    +
    +void dp_lsh_projection_p(const float *hash_seed, const int32_t *feature, const double *weight, int32_t *output, int hash_group_num, int bits_per_hash, int feature_num)
    +

    C调用示例:

    +
     1// MT7004 示例:fp16 (hp) 权重,私有存储单核
    + 2#include <stdio.h>
    + 3
    + 4int main() {
    + 5    float* hash_seed = (float*)0x10000000;
    + 6    int32_t* feature = (int32_t*)0x10001000;
    + 7    half* weight = (half*)0x10002000;
    + 8    int32_t* output = (int32_t*)0x10003000;
    + 9
    +10    int hash_group_num = 64;
    +11    int bits_per_hash = 8;
    +12    int feature_num = 32;
    +13
    +14    hp_lsh_projection_p(hash_seed, feature, weight, output,
    +15                        hash_group_num, bits_per_hash, feature_num);
    +16    return 0;
    +17}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/lstm.html b/master/html/functionlib/dsplib/lstm.html index 907c803..c6d78b2 100644 --- a/master/html/functionlib/dsplib/lstm.html +++ b/master/html/functionlib/dsplib/lstm.html @@ -23,8 +23,8 @@ - - + + @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
    @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    输入:
      -
    • input - 输入序列数据,形状为 \((seq_len, batch, input_size)\),即每个时间步的输入特征。

    • -
    • weight_i - 输入到各门 \((input、forget、cell、output)\) 的权重矩阵,大小为 4 * hidden_size * input_size。

    • -
    • weight_h - 上一隐藏状态到各门的权重矩阵,大小为 \(4 * hidden_size * hidden_size\)

    • +
    • input - 输入序列数据,形状为 \((seq\_len, batch, input\_size)\),即每个时间步的输入特征。

    • +
    • weight_i - 输入到各门 \((input, forget, cell, output)\) 的权重矩阵,大小为 4 * hidden_size * input_size。

    • +
    • weight_h - 上一隐藏状态到各门的权重矩阵,大小为 \(4 * hidden\_size * hidden\_size\)

    • input_bias - 输入部分的偏置项,对应 4 个门的偏置。

    • -
    • state_bias - 隐藏状态部分的偏置项(也是 \(4 * hidden_size\)),与 input_bias 一起求和形成总偏置。

    • -
    • hidden_state - 当前批次初始隐藏状态输入( \(h₀\) ),执行后更新为最后时刻的隐藏状态输出( \(hₜ\)

    • -
    • cell_state - 当前批次初始细胞状态输入( \(c₀\)),执行后更新为最后时刻的细胞状态输出( \(cₜ\))。

    • +
    • state_bias - 隐藏状态部分的偏置项(也是 \(4 * hidden\_size\)),与 input_bias 一起求和形成总偏置。

    • +
    • hidden_state - 当前批次初始隐藏状态输入( \(h_0\) ),执行后更新为最后时刻的隐藏状态输出( \(h_t\)

    • +
    • cell_state - 当前批次初始细胞状态输入( \(c_0\)),执行后更新为最后时刻的细胞状态输出( \(c_t\))。

    • buffer - 临时工作区指针数组(中间计算缓存,如门值、激活结果、临时矩阵等,用于优化性能)。

    • LstmParameter - LSTM 配置参数结构体,包含输入大小、隐藏层维度、序列长度、是否双向等信息。

    • core_mask - 核掩码(仅适用于共享存储版本)。

    • @@ -204,7 +369,7 @@ h_t &= o_t \odot \tanh(c_t) && \text{(隐藏状态更新)}

      备注

      • FT78NE 支持fp32

      • -
      • MT7004 支持fp32

      • +
      • MT7004 支持fp32、fp16

    @@ -215,6 +380,11 @@ h_t &= o_t \odot \tanh(c_t) && \text{(隐藏状态更新)} void fp_Lstm_s(float *output, const float *input, const float *weight_i, const float *weight_h, const float *input_bias, const float *state_bias, float *hidden_state, float *cell_state, float *buffer[9], const LstmParameter *lstm_param, int core_mask)
    +
    +
    +void hp_Lstm_s(half *output, const half *input, const half *weight_i, const half *weight_h, const half *input_bias, const half *state_bias, half *hidden_state, half *cell_state, half *buffer[9], const LstmParameter *lstm_param, int core_mask)
    +
    +

    C调用示例:

     1//FT78NE示例
    @@ -268,6 +438,11 @@ h_t &= o_t \odot \tanh(c_t) && \text{(隐藏状态更新)}
     
    void fp_Lstm_p(float *output, const float *input, const float *weight_i, const float *weight_h, const float *input_bias, const float *state_bias, float *hidden_state, float *cell_state, float *buffer[9], const LstmParameter *lstm_param)
    +
    + +
    +
    +void hp_Lstm_p(half *output, const half *input, const half *weight_i, const half *weight_h, const half *input_bias, const half *state_bias, half *hidden_state, half *cell_state, half *buffer[9], const LstmParameter *lstm_param)

    C调用示例:

    @@ -323,14 +498,14 @@ h_t &= o_t \odot \tanh(c_t) && \text{(隐藏状态更新)}
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -194,9 +359,9 @@ 7 float delta = 0.5f; 8 int length = 1000; 9 int core_mask = 0xff; -10 fp_range_s(output, start, delta, length, core_mask); -11 return 0; -12} +10 fp_range_s(output, start, delta, length, core_mask); +11 return 0; +12}
    @@ -234,9 +399,9 @@ 6 float start = 1.5f; 7 float delta = 0.5f; 8 int length = 1000; - 9 fp_range_p(output, start, delta, length); -10 return 0; -11} + 9 fp_range_p(output, start, delta, length); +10 return 0; +11}
    @@ -254,7 +419,7 @@
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/rank.html b/master/html/functionlib/dsplib/rank.html new file mode 100644 index 0000000..6358262 --- /dev/null +++ b/master/html/functionlib/dsplib/rank.html @@ -0,0 +1,398 @@ + + + + + + + + + Rank — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Rank

    +
    +

    获取输入张量的秩(维数),并将该值写入输出地址。

    +
    +
    输入:
      +
    • output - 输出数据的地址,用于存储秩的结果。

    • +
    • n - 输入张量的秩(维数)。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 存储秩数值的地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • 由于该算子对于不同数据类型的具体实现一致,因此统一使用 rank_srank_p 命名,不再区分数据类型前缀(如 fp_, i8_ 等)。

    • +
    • 支持的数据类型包括:int8, int16, int32, fp32, fp64, cplx64, cplx128。

    • +
    +
    +
    +

    共享存储版本:

    +
    +
    +void rank_s(int *output, int n, int core_mask)
    +

    C调用示例:

    +
     1#include <stdio.h>
    + 2
    + 3int main(int argc, char* argv[]) {
    + 4    int n = 4; // 假设张量的秩为4
    + 5    int *output = (int *)0xA0000000;
    + 6    int core_mask = 0xff;
    + 7
    + 8    rank_s(output, n, core_mask);
    + 9
    +10    return 0;
    +11}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void rank_p(int *output, int n)
    +

    C调用示例:

    +
     1#include <stdio.h>
    + 2
    + 3int main(int argc, char* argv[]) {
    + 4    int n = 3; // 假设张量的秩为3
    + 5    int *output = (int *)0x10810000;
    + 6
    + 7    rank_p(output, n);
    + 8
    + 9    return 0;
    +10}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/real_div.html b/master/html/functionlib/dsplib/real_div.html new file mode 100644 index 0000000..5e8f15c --- /dev/null +++ b/master/html/functionlib/dsplib/real_div.html @@ -0,0 +1,473 @@ + + + + + + + + + RealDiv — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    RealDiv

    +

    逐元素计算两个输入的除法。

    +
    +\[output_i = \frac{input0_i}{input1_i}\]
    +
    +
    输入:
      +
    • input0 - 被除数输入数据地址。

    • +
    • input1 - 除数输入数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128

    • +
    • MT7004 支持 fp16, fp32, int16, int32, cplx64

    • +
    • 若除数元素为 0,则输出结果为无穷大或未定义值,需由上层逻辑处理。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_real_div_s(int8_t *input0, int8_t *input1, int8_t *output, int length, int core_mask)
    +
    + +
    +
    +void i16_real_div_s(int16_t *input0, int16_t *input1, int16_t *output, int length, int core_mask)
    +
    + +
    +
    +void i32_real_div_s(int32_t *input0, int32_t *input1, int32_t *output, int length, int core_mask)
    +
    + +
    +
    +void hp_real_div_s(half *input0, half *input1, half *output, int length, int core_mask)
    +
    + +
    +
    +void fp_real_div_s(float *input0, float *input1, float *output, int length, int core_mask)
    +
    + +
    +
    +void dp_real_div_s(double *input0, double *input1, double *output, int length, int core_mask)
    +
    + +
    +
    +void c64_real_div_s(float *input0, float *input1, float *output, int length, int core_mask)
    +
    + +
    +
    +void c128_real_div_s(double *input0, double *input1, double *output, int length, int core_mask)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    float *input0 = (float *)0xA0000000;   // input0 在 DDR 空间
    + 6    float *input1 = (float *)0xA1000000;   // input1 在 DDR 空间
    + 7    float *output = (float *)0xB0000000;   // 输出结果在 DDR 空间
    + 8    int length = 1024;
    + 9    int core_mask = 0xff;
    +10    fp_real_div_s(input0, input1, output, length, core_mask);
    +11    return 0;
    +12}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_real_div_p(int8_t *input0, int8_t *input1, int8_t *output, int length)
    +
    + +
    +
    +void i16_real_div_p(int16_t *input0, int16_t *input1, int16_t *output, int length)
    +
    + +
    +
    +void i32_real_div_p(int32_t *input0, int32_t *input1, int32_t *output, int length)
    +
    + +
    +
    +void hp_real_div_p(half *input0, half *input1, half *output, int length)
    +
    + +
    +
    +void fp_real_div_p(float *input0, float *input1, float *output, int length)
    +
    + +
    +
    +void dp_real_div_p(double *input0, double *input1, double *output, int length)
    +
    + +
    +
    +void c64_real_div_p(float *input0, float *input1, float *output, int length)
    +
    + +
    +
    +void c128_real_div_p(double *input0, double *input1, double *output, int length)
    +

    C调用示例:

    +
     1//MT7004 示例
    + 2#include <stdio.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    float *input0 = (float *)0x10000000;
    + 6    float *input1 = (float *)0x10001000;
    + 7    float *output = (float *)0x10002000;
    + 8    int length = 1024;
    + 9    fp_real_div_p(input0, input1, output, length);
    +10    return 0;
    +11}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/reciprocal.html b/master/html/functionlib/dsplib/reciprocal.html new file mode 100644 index 0000000..c151740 --- /dev/null +++ b/master/html/functionlib/dsplib/reciprocal.html @@ -0,0 +1,470 @@ + + + + + + + + + Reciprocal — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Reciprocal

    +

    逐元素计算输入数据的倒数。

    +
    +\[output_i = \frac{1}{Input_i}\]
    +
    +
    输入:
      +
    • Input - 输入数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128

    • +
    • MT7004 支持 fp16, fp32, int16, int32, cplx64

    • +
    • 当输入为 0 时,输出结果为无穷大或未定义值,需由上层逻辑处理。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_reciprocal_s(int8_t *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void i16_reciprocal_s(int16_t *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void i32_reciprocal_s(int32_t *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void hp_reciprocal_s(half *Input, half *output, int length, int core_mask)
    +
    + +
    +
    +void fp_reciprocal_s(float *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void dp_reciprocal_s(double *Input, double *output, int length, int core_mask)
    +
    + +
    +
    +void c64_reciprocal_s(float *Input, float *output, int length, int core_mask)
    +
    + +
    +
    +void c128_reciprocal_s(double *Input, double *output, int length, int core_mask)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0xA0000000;   //input在DDR空间
    + 6    float *output = (float *)0xB0000000;
    + 7    int length = 1000;
    + 8    int core_mask = 0xff;
    + 9    fp_reciprocal_s(input, output, length, core_mask);
    +10    return 0;
    +11}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_reciprocal_p(int8_t *Input, float *output, int length)
    +
    + +
    +
    +void i16_reciprocal_p(int16_t *Input, float *output, int length)
    +
    + +
    +
    +void i32_reciprocal_p(int32_t *Input, float *output, int length)
    +
    + +
    +
    +void hp_reciprocal_p(half *Input, half *output, int length)
    +
    + +
    +
    +void fp_reciprocal_p(float *Input, float *output, int length)
    +
    + +
    +
    +void dp_reciprocal_p(double *Input, double *output, int length)
    +
    + +
    +
    +void c64_reciprocal_p(float *Input, float *output, int length)
    +
    + +
    +
    +void c128_reciprocal_p(double *Input, double *output, int length)
    +

    C调用示例:

    +
     1//MT7004示例
    + 2#include <stdio.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    float *input = (float *)0x10000000;
    + 6    float *output = (float *)0x10001000;
    + 7    int length = 1000;
    + 8    fp_reciprocal_p(input, output, length);
    + 9    return 0;
    +10}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/reduce.html b/master/html/functionlib/dsplib/reduce.html index 2c42918..191bcaa 100644 --- a/master/html/functionlib/dsplib/reduce.html +++ b/master/html/functionlib/dsplib/reduce.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
    @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
    @@ -111,15 +276,15 @@
    @@ -111,15 +276,15 @@
      -
    • - +
    • +
    • @@ -250,7 +415,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/reversev2.html b/master/html/functionlib/dsplib/reversev2.html index ae66b94..d8a8ccb 100644 --- a/master/html/functionlib/dsplib/reversev2.html +++ b/master/html/functionlib/dsplib/reversev2.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
      @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
    @@ -112,15 +277,15 @@

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/scatter_elements.html b/master/html/functionlib/dsplib/scatter_elements.html index 550a22f..2238044 100644 --- a/master/html/functionlib/dsplib/scatter_elements.html +++ b/master/html/functionlib/dsplib/scatter_elements.html @@ -23,7 +23,7 @@ - + @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
    @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
    @@ -112,15 +277,15 @@
      -
    • - +
    • +
    • @@ -136,6 +301,17 @@

      SGD

      对权重张量执行带动量与权重衰减的随机梯度下降更新。

      +
      +\[\begin{split}\begin{aligned} +g'_t &= g_t + weight\_decay \cdot w_{t-1} \\ +m_t &= moment \cdot m_{t-1} + (1 - dampening) \cdot g'_t \\ +u_t &= +\begin{cases} + m_t \cdot moment + g'_t, & \text{if nesterov = True} \\ + m_t, & \text{otherwise} +\end{cases} \\ +w_t &= w_{t-1} - learning\_rate \cdot u_t +\end{aligned}\end{split}\]
      输入:
      • weight - 待更新权重张量首地址。

      • @@ -177,18 +353,7 @@
        void fp_sgd_s(float *weight, float *accumulate, const float *gradient, float learning_rate, float dampening, float moment, bool nesterov, float weight_decay, int start, int end, int core_mask)
        -
        -\[\begin{split}\begin{aligned} -g'_t &= g_t + weight\_decay \cdot w_{t-1} \\ -m_t &= moment \cdot m_{t-1} + (1 - dampening) \cdot g'_t \\ -u_t &= -\begin{cases} - m_t \cdot moment + g'_t, & \text{if nesterov = True} \\ - m_t, & \text{otherwise} -\end{cases} \\ -w_t &= w_{t-1} - learning\_rate \cdot u_t -\end{aligned}\end{split}\]
        -

        C调用示例:

        +

        C调用示例:

         1// FT78NE 多核示例
          2#include <stdio.h>
          3#include <stdbool.h>
        @@ -206,9 +371,9 @@ w_t &= w_{t-1} - learning\_rate \cdot u_t
         15    bool nesterov = true;
         16    float weight_decay = 1e-2f;
         17    fp_sgd_s(weight, accumulate, gradient, learning_rate,
        -18             dampening, moment, nesterov, weight_decay,
        -19             start, end, core_mask);
        -20    return 0;
        +18             dampening, moment, nesterov, weight_decay,
        +19             start, end, core_mask);
        +20    return 0;
         21}
         
        @@ -239,9 +404,9 @@ w_t &= w_{t-1} - learning\_rate \cdot u_t 13 bool nesterov = false; 14 float weight_decay = 5e-3f; 15 hp_sgd_p(weight, accumulate, gradient, learning_rate, -16 dampening, moment, nesterov, weight_decay, -17 length); -18 return 0; +16 dampening, moment, nesterov, weight_decay, +17 length); +18 return 0; 19}
    @@ -260,7 +425,7 @@ w_t &= w_{t-1} - learning\_rate \cdot u_t
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/shape.html b/master/html/functionlib/dsplib/shape.html new file mode 100644 index 0000000..41c5874 --- /dev/null +++ b/master/html/functionlib/dsplib/shape.html @@ -0,0 +1,416 @@ + + + + + + + + + Shape — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Shape

    +

    形状推断函数(Shape Inference Function)。该函数根据输入张量的形状和算子参数,推断输出张量的形状。该函数不区分数据类型,只处理张量的形状信息。

    +

    如果所有输出张量都是常量(ConstTensor 或 ConstScalar),则直接返回,不进行形状推断。否则,根据算子类型调用相应的形状推断函数。

    +
    +
    支持的算子类型:
      +
    • Arithmetic_InferShape - 算术运算的形状推断

    • +
    • Common_InferShape - 通用算子的形状推断

    • +
    • Softmax_InferShape - Softmax 算子的形状推断

    • +
    • MaxMinGrad_InferShape - MaxMin 梯度算子的形状推断

    • +
    • Dropout_InferShape - Dropout 算子的形状推断

    • +
    • DynamicQuant_InferShape - 动态量化算子的形状推断

    • +
    • Fft_InferShape - FFT 算子的形状推断

    • +
    • Flatten_InferShape - Flatten 算子的形状推断

    • +
    • LayerNorm_InferShape - LayerNorm 算子的形状推断

    • +
    • LogSoftmax_InferShape - LogSoftmax 算子的形状推断

    • +
    +
    +
    输入:
      +
    • inputs - 输入张量数组(TensorC** 类型)。

    • +
    • inputs_size - 输入张量的数量。

    • +
    • outputs - 输出张量数组(TensorC** 类型)。

    • +
    • outputs_size - 输出张量的数量。

    • +
    • param - 算子参数(OpParameter* 类型),包含算子类型和其他参数信息。

    • +
    +
    +
    输出:
      +
    • outputs - 输出张量数组,其中的形状信息会被更新。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • 该函数不区分数据类型,适用于所有数据类型

    • +
    • 函数会自动检查输出是否为常量,如果是常量则跳过形状推断

    • +
    +
    +

    共享存储/私有存储版本:

    +
    +
    +void shape(TensorC **inputs, int inputs_size, TensorC **outputs, int outputs_size, OpParameter *param)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <shape.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    TensorC** input_tensors_ptrs = (TensorC**)0x10010000;
    + 7    TensorC** output_tensors_ptrs = (TensorC**)0x10011000;
    + 8
    + 9    TensorC input0;
    +10    TensorC input1;
    +11    TensorC output;
    +12
    +13    int input0_shape[4] = {1,2,3,4};
    +14    int input1_shape[4] = {1,3,4};
    +15    int output_shape[4]; //不用初始化
    +16    memcpy(input0.shape_, input0_shape, 4 * sizeof(int));
    +17    input0.shape_size_ = 4;
    +18    memcpy(input1.shape_, input1_shape, 4 * sizeof(int));
    +19    input1.shape_size_ = 3;
    +20    input0.data_type_ = kNumberTypeFloat32;
    +21    input1.data_type_ = kNumberTypeFloat32;
    +22    input0.format_ = Format_NCHW;
    +23    input1.format_ = Format_NCHW;
    +24
    +25    input_tensors_ptrs[0] = &input0;
    +26    input_tensors_ptrs[1] = &input1;
    +27    output_tensors_ptrs[0] = &output;
    +28
    +29    ArithmeticParameter param;
    +30    param.op_parameter_.type_ = Arithmetic_InferShape;
    +31
    +32    shape(input_tensors_ptrs, 2, output_tensors_ptrs, 1, (OpParameter*)&param);
    +33    return 0;
    +34}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.html b/master/html/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.html new file mode 100644 index 0000000..4dc3a44 --- /dev/null +++ b/master/html/functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.html @@ -0,0 +1,437 @@ + + + + + + + + + SigmoidCrossEntropyWithLogitsGrad — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    +
    + +
    +
    +
    +
    + +
    +

    SigmoidCrossEntropyWithLogitsGrad

    +

    计算 Sigmoid Cross Entropy With Logits 的梯度。 +该算子以 logits 和 labels 为输入,输出对 logits 的梯度值。

    +

    数学表达式为:

    +
    +\[ \begin{align}\begin{aligned}\sigma(x) = \frac{1}{1 + e^{-x}}\\\quad\\dst_i = \sigma(x_i) - y_i\end{aligned}\end{align} \]
    +
    +
    其中:
      +
    • \(x_i\) 表示第 i 个 logit(Input0)

    • +
    • \(y_i\) 表示第 i 个标签(Input1)

    • +
    +
    +
    +

    为提高数值稳定性,计算中对正负 logits 采用不同形式:

    +
    +\[\begin{split}\sigma(x) = +\begin{cases} + \dfrac{1}{1 + e^{-x}}, & x > 0 \\ + \dfrac{e^{x}}{1 + e^{x}}, & x \le 0 +\end{cases}\end{split}\]
    +
    +
    输入:
      +
    • Input0 - logits 输入数据地址。

    • +
    • Input1 - 标签(labels)数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 梯度计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 fp32 类型

    • +
    • MT7004 支持 fp16fp32 类型

    • +
    • 输入 logits 与 labels 必须具有相同长度

    • +
    • 输出为对 logits 的梯度,不包含对 labels 的梯度

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_sigmoidcrossentropywithlogitsgrad_s(float *Input0, float *Input1, float *output, int length, int core_mask)
    +
    + +
    +
    +void hp_sigmoidcrossentropywithlogitsgrad_s(half *Input0, half *Input1, half *output, int length, int core_mask)
    +
    + +

    C调用示例:

    +
     1// FT78NE 示例
    + 2#include <stdio.h>
    + 3#include <sigmoid_cross_entropy.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *logits = (float *)0xA0000000;   // DDR 空间
    + 7    float *labels = (float *)0xA0100000;
    + 8    float *output = (float *)0xC0000000;
    + 9
    +10    int length = 1024;
    +11    int core_mask = 0xff;
    +12
    +13    fp_sigmoidcrossentropywithlogitsgrad_s(logits, labels, output, length, core_mask);
    +14
    +15    return 0;
    +16}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_sigmoidcrossentropywithlogitsgrad_p(float *Input0, float *Input1, float *output, int length)
    +
    + +
    +
    +void hp_sigmoidcrossentropywithlogitsgrad_p(half *Input0, half *Input1, half *output, int length)
    +
    + +

    C调用示例:

    +
     1// FT78NE 示例(私有存储)
    + 2#include <stdio.h>
    + 3#include <sigmoid_cross_entropy.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *logits = (float *)0x10810000;   // L2 空间
    + 7    float *labels = (float *)0x10820000;
    + 8    float *output = (float *)0x10830000;
    + 9
    +10    int length = 1024;
    +11    fp_sigmoidcrossentropywithlogitsgrad_p( logits, labels, output, length);
    +12
    +13    return 0;
    +14}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/sigmoidcrossentropywithlogits.html b/master/html/functionlib/dsplib/sigmoidcrossentropywithlogits.html new file mode 100644 index 0000000..ccabb4d --- /dev/null +++ b/master/html/functionlib/dsplib/sigmoidcrossentropywithlogits.html @@ -0,0 +1,437 @@ + + + + + + + + + SigmoidCrossEntropyWithLogits — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    +
    + +
    +
    +
    +
    + +
    +

    SigmoidCrossEntropyWithLogits

    +

    计算预测值与真实值之间的sigmoid交叉熵。

    +

    测量离散分类任务中的分布误差,每个类相互独立,且计算出各个类的交叉熵损失。

    +

    将输入 logits 设置为 \(X\),输入 label 为 \(Y\),输出为 \(loss\)。然后,

    +
    +\[\begin{split}\begin{aligned} +p &= \text{sigmoid}(X) = \frac{1}{1 + e^{-X}} \\ +loss &= -[Y \cdot \ln(p) + (1 - Y) \cdot \ln(1 - p)] +\end{aligned}\end{split}\]
    +
    +
    输入:
      +
    • input0 - 输入 logits 张量地址。

    • +
    • input1 - 输入标签张量地址,与 logits 形状相同。

    • +
    • length - 张量元素总数。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 输出损失张量地址,与输入张量形状相同。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持的数据类型:int8, fp32

    • +
    • MT7004 支持的数据类型:fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_sigmoidcrossentropywithlogits_s(int8_t *input0, int8_t *input1, int8_t *output, int length, int core_mask)
    +
    + +
    +
    +void fp_sigmoidcrossentropywithlogits_s(float *input0, float *input1, float *output, int length, int core_mask)
    +
    + +
    +
    +void hp_sigmoidcrossentropywithlogits_s(half *input0, half *input1, half *output, int length, int core_mask)
    +
    + +

    C调用示例:

    +
    +
     1// FT78NE 多核示例
    + 2#include <stdio.h>
    + 3#include <sigmoidcrossentropywithlogits.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input0 = (float *)0xA0000000;   // logits在DDR空间
    + 7    float *input1 = (float *)0xB0000000;    // label在DDR空间
    + 8    float *output = (float *)0xC0000000;   // 输出损失在DDR空间
    + 9    int length = 1000;
    +10    int core_mask = 0xff;
    +11
    +12    // 计算 sigmoid 交叉熵损失
    +13    fp_sigmoidcrossentropywithlogits_s(input0, input1, output, length, core_mask);
    +14    return 0;
    +15}
    +
    +
    +
    +

    私有存储版本:

    +
    +
    +void i8_sigmoidcrossentropywithlogits_p(int8_t *input0, int8_t *input1, int8_t *output, int length)
    +
    + +
    +
    +void fp_sigmoidcrossentropywithlogits_p(float *input0, float *input1, float *output, int length)
    +
    + +
    +
    +void hp_sigmoidcrossentropywithlogits_p(half *input0, half *input1, half *output, int length)
    +
    + +

    C调用示例:

    +
    +
     1// MT7004 单核示例
    + 2#include <stdio.h>
    + 3#include <sigmoidcrossentropywithlogits.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    half *input0 = (half *)0x10000000;   // logits在L2空间
    + 7    half *input1 = (half *)0x10004000;    // label在L2空间
    + 8    half *output = (half *)0x10008000;   // 输出损失在L2空间
    + 9    int length = 1000;
    +10
    +11    // 计算 sigmoid 交叉熵损失
    +12    hp_sigmoidcrossentropywithlogits_p(input0, input1, output, length);
    +13    return 0;
    +14}
    +
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/sin.html b/master/html/functionlib/dsplib/sin.html new file mode 100644 index 0000000..200b51c --- /dev/null +++ b/master/html/functionlib/dsplib/sin.html @@ -0,0 +1,453 @@ + + + + + + + + + Sin — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Sin

    +

    传入一个数组,对每个元素逐元素计算其正弦值并输出。

    +
    +\[dst_i = \sin(src_i)\]
    +

    输入角度单位为弧度。

    +
    +
    输入:
      +
    • src_data - 输入数据地址。

    • +
    • length - 计算长度。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • dst_data - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 fp, dp, int8, int16, int32

    • +
    • MT7004 支持 hp, fp, int16, int32

    • +
    • 整数类型在计算时会先转换为浮点数,再按对应类型输出

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_sin_s(int8_t *src_data, float *dst_data, int length, int core_mask)
    +
    + +
    +
    +void i16_sin_s(int16_t *src_data, float *dst_data, int length, int core_mask)
    +
    + +
    +
    +void i32_sin_s(int *src_data, float *dst_data, int length, int core_mask)
    +
    + +
    +
    +void hp_sin_s(half *src_data, half *dst_data, int length, int core_mask)
    +
    + +
    +
    +void fp_sin_s(float *src_data, float *dst_data, int length, int core_mask)
    +
    + +
    +
    +void dp_sin_s(double *src_data, double *dst_data, int length, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <sin.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input = (float *)0xA0000000;   // input在DDR空间
    + 7    float *output = (float *)0xC0000000;
    + 8    int length = 1024;
    + 9    int core_mask = 0xff;
    +10    fp_sin_s(input, output, length, core_mask);
    +11    return 0;
    +12}
    +
    +
    +

    私有存储版本:

    +
    +
    +void i8_sin_p(int8_t *src_data, float *dst_data, int length)
    +
    + +
    +
    +void i16_sin_p(int16_t *src_data, float *dst_data, int length)
    +
    + +
    +
    +void i32_sin_p(int *src_data, float *dst_data, int length)
    +
    + +
    +
    +void hp_sin_p(half *src_data, half *dst_data, int length)
    +
    + +
    +
    +void fp_sin_p(float *src_data, float *dst_data, int length)
    +
    + +
    +
    +void dp_sin_p(double *src_data, double *dst_data, int length)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <sin.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *input = (float *)0x10810000;   // input在L2空间
    + 7    float *output = (float *)0x10820000;
    + 8    int length = 1024;
    + 9    fp_sin_p(input, output, length);
    +10    return 0;
    +11}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/size.html b/master/html/functionlib/dsplib/size.html new file mode 100644 index 0000000..1823985 --- /dev/null +++ b/master/html/functionlib/dsplib/size.html @@ -0,0 +1,408 @@ + + + + + + + + + Size — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Size

    +
    +

    计算输入张量或向量的元素总个数(即形状各维度的乘积),并将结果写入输出地址。

    +
    +
    输入:
      +
    • output - 输出数据的地址,用于存储计算结果。

    • +
    • shape - 输入张量的形状(维度)数组地址。

    • +
    • n - 输入张量的维度数(Rank)。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 存储元素总个数的地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • 由于该算子对于不同数据类型的具体实现一致,因此统一使用 size_ssize_p 命名,不再区分数据类型前缀(如 fp_, i8_ 等)。

    • +
    • 支持的数据类型包括:int8, int16, int32, fp32, fp64, cplx64, cplx128。

    • +
    +
    +
    +

    共享存储版本:

    +
    +
    +void size_s(int *output, int *shape, int n, int core_mask)
    +

    C调用示例:

    +
     1#include <stdio.h>
    + 2#include <size.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    int n = 4;
    + 6    // 假设 shape 为 {2, 3, 4, 5},元素总数为 120
    + 7    int *output = (int *)0xA0000000;
    + 8    int *shape = (int *)0xA0000100;
    + 9
    +10    // 实际使用中需确保 shape 地址处已存入维度数据
    +11    int core_mask = 0xff;
    +12
    +13    size_s(output, shape, n, core_mask);
    +14
    +15    return 0;
    +16}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void size_p(int *output, int *shape, int n)
    +

    C调用示例:

    +
     1#include <stdio.h>
    + 2#include <size.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    int n = 3;
    + 6    // 假设 shape 为 {10, 10, 4},元素总数为 400
    + 7    int *output = (int *)0x10000000;
    + 8    int *shape = (int *)0x10000040;
    + 9
    +10    // 实际使用中需确保 shape 地址处已存入维度数据
    +11    size_p(output, shape, n);
    +12
    +13    return 0;
    +14}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/skipgram.html b/master/html/functionlib/dsplib/skipgram.html new file mode 100644 index 0000000..558fbc7 --- /dev/null +++ b/master/html/functionlib/dsplib/skipgram.html @@ -0,0 +1,454 @@ + + + + + + + + + Skipgram — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Skipgram

    +

    从输入语句中生成skip-gram。skip-gram是从句子中提取的词语序列,其中相邻词语之间的距离(跳过的词语数量)不超过指定的最大值。此算子主要用于自然语言处理(NLP)任务。

    +
    +
    输入:
      +
    • sentence - StringPack* 类型,指向输入语句的指针。

    • +
    • words - StringPack* 类型,用于存储从语句中解析出的词语的工作空间。

    • +
    • ngram_size - int 类型,指定每个n-gram中的词语数量。

    • +
    • max_skip_size - int 类型,指定在构成n-gram时可以跳过的最大词语数量。

    • +
    • include_all_ngrams - int 类型,布尔标志。如果为非零,则生成长度从1到 ngram_size 的所有n-gram;否则,仅生成长度为 ngram_size 的n-gram。

    • +
    • grams - StringPack** 类型,用于存储生成的gram的工作空间。

    • +
    • grams_word_count - int* 类型,用于存储每个gram中词语数量的工作空间。

    • +
    • stack - int* 类型,长度为 ngram_size 的临时工作空间。

    • +
    • blank - char* 类型,指向空格字符的指针,用于连接gram中的词语。

    • +
    • output_tensor - char* 类型,最终输出张量的数据地址。

    • +
    • offset - int* 类型,输出参数,用于存储 output_tensor 中每个gram的偏移量。

    • +
    • len - int* 类型,输出参数,用于存储 output_tensor 中每个gram的长度。

    • +
    • shape - int* 类型,输出参数,用于存储输出张量的维度信息。

    • +
    • core_mask - (仅限共享版本) 核掩码。

    • +
    +
    +
    输出:
      +
    • output_tensoroffsetlenshape - 这些指针指向的内存区域将被填充,以表示包含生成结果的最终张量。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • 该算子不区分数据类型,直接操作字符数据。

    • +
    +
    +

    共享存储版本:

    +
    +
    +int skipgram_s(StringPack *sentence, StringPack *words, int ngram_size, int max_skip_size, int include_all_ngrams, StringPack **grams, int *grams_word_count, int *stack, char *blank, char *output_tensor, int *offset, int *len, int *shape, int core_mask)
    +
    + +

    C调用示例:

    +
     1#include <stdio.h>
    + 2#include "skipgram.h" // 假设头文件名
    + 3
    + 4typedef struct StringPack {
    + 5  long long len;
    + 6  char *data;
    + 7} StringPack;
    + 8
    + 9int main(int argc, char* argv[]) {
    +10    // 假设输入和工作空间已在DDR中分配
    +11    StringPack* sentence = (StringPack*)0xA0000000;
    +12    StringPack* words = (StringPack*)0xA0010000;
    +13    StringPack** grams = (StringPack**)0xA0020000;
    +14    int* grams_word_count = (int*)0xA0030000;
    +15    int* stack = (int*)0xA0040000;
    +16    char* output_tensor = (char*)0xB0000000;
    +17    int* offset = (int*)0xB0010000;
    +18    int* len = (int*)0xB0020000;
    +19    int* shape = (int*)0xB0030000;
    +20    char blank_char = ' ';
    +21
    +22    // 填充sentence内容
    +23    sentence->data = "mindspore signal processing library";
    +24    sentence->len = 33;
    +25
    +26    int ngram_size = 2;
    +27    int max_skip_size = 1;
    +28    int include_all_ngrams = 1;
    +29    int core_mask = 0xff;
    +30
    +31    skipgram_s(sentence, words, ngram_size, max_skip_size, include_all_ngrams,
    +32               grams, grams_word_count, stack, &blank_char,
    +33               output_tensor, offset, len, shape, core_mask);
    +34    return 0;
    +35}
    +
    +
    +

    私有存储版本:

    +
    +
    +int skipgram_p(StringPack *sentence, StringPack *words, int ngram_size, int max_skip_size, int include_all_ngrams, StringPack **grams, int *grams_word_count, int *stack, char *blank, char *output_tensor, int *offset, int *len, int *shape)
    +
    + +

    C调用示例:

    +
     1#include <stdio.h>
    + 2#include "skipgram.h" // 假设头文件名
    + 3
    + 4typedef struct StringPack {
    + 5  long long len;
    + 6  char *data;
    + 7} StringPack;
    + 8
    + 9int main(int argc, char* argv[]) {
    +10    // 假设输入和工作空间已在私有内存中分配
    +11    StringPack* sentence = (StringPack*)0x10001000;
    +12    StringPack* words = (StringPack*)0x10002000;
    +13    StringPack** grams = (StringPack**)0x10003000;
    +14    int* grams_word_count = (int*)0x10004000;
    +15    int* stack = (int*)0x10005000;
    +16    char* output_tensor = (char*)0x10006000;
    +17    int* offset = (int*)0x10007000;
    +18    int* len = (int*)0x10008000;
    +19    int* shape = (int*)0x10009000;
    +20    char blank_char = ' ';
    +21
    +22    // 填充sentence内容
    +23    sentence->data = "mindspore signal processing library";
    +24    sentence->len = 33;
    +25
    +26    int ngram_size = 2;
    +27    int max_skip_size = 1;
    +28    int include_all_ngrams = 1;
    +29
    +30    skipgram_p(sentence, words, ngram_size, max_skip_size, include_all_ngrams,
    +31               grams, grams_word_count, stack, &blank_char,
    +32               output_tensor, offset, len, shape);
    +33    return 0;
    +34}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/slice.html b/master/html/functionlib/dsplib/slice.html new file mode 100644 index 0000000..31fffa6 --- /dev/null +++ b/master/html/functionlib/dsplib/slice.html @@ -0,0 +1,483 @@ + + + + + + + + + Slice — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Slice

    +

    从输入张量中提取一个子集(切片)。根据给定的起始索引(begin)和切片大小(size),在每个维度上截取数据。

    +
    +\[output[i_0, i_1, \dots, i_{n-1}] = input[begin_0 + i_0, begin_1 + i_1, \dots, begin_{n-1} + i_{n-1}]\]
    +

    其中 $n$ 为维度数(ndim),且满足 $0 le i_k < size_k$。

    +
    +
    输入:
      +
    • input - 输入张量数据地址。

    • +
    • input_shape - 输入张量的形状数组地址。

    • +
    • ndim - 输入张量的维度数。

    • +
    • begin - 切片起始索引数组地址(长度为 ndim)。

    • +
    • size - 切片大小数组地址(长度为 ndim)。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output - 计算结果存储地址(输出张量形状即为 size)。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 int8, int16, int32, fp32, fp64, cplx64, cplx128

    • +
    • MT7004 支持 fp16, fp32, int16, int32, cplx64

    • +
    • 切片操作不改变数据数值,仅改变数据的空间排布,常用于特征提取或张量分解。

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_slice_s(int8_t *input, int8_t *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +
    + +
    +
    +void i16_slice_s(int16_t *input, int16_t *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +
    + +
    +
    +void i32_slice_s(int32_t *input, int32_t *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +
    + +
    +
    +void hp_slice_s(half *input, half *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +
    + +
    +
    +void fp_slice_s(float *input, float *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +
    + +
    +
    +void dp_slice_s(double *input, double *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +
    + +
    +
    +void c64_slice_s(float *input, float *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +
    + +
    +
    +void c128_slice_s(double *input, double *output, int *input_shape, int ndim, int *begin, int *size, int core_mask)
    +

    C调用示例:

    +
     1// FT78NE 示例(多核并行切片)
    + 2#include <stdio.h>
    + 3#include "78NE/utils.h"
    + 4
    + 5int main() {
    + 6    float *input = (float *)0xA0000000;
    + 7    float *output = (float *)0xB0000000;
    + 8    int input_shape[] = {16, 32, 64, 128};
    + 9    int begin[] = {2, 4, 8, 16};
    +10    int size[] = {8, 12, 20, 32};
    +11    int ndim = 4;
    +12    int core_mask = 0xFF; // 使用8核并行
    +13
    +14    fp_slice_s(input, output, input_shape, ndim, begin, size, core_mask);
    +15    return 0;
    +16}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_slice_p(int8_t *input, int8_t *output, int *input_shape, int ndim, int *begin, int *size)
    +
    + +
    +
    +void i16_slice_p(int16_t *input, int16_t *output, int *input_shape, int ndim, int *begin, int *size)
    +
    + +
    +
    +void i32_slice_p(int32_t *input, int32_t *output, int *input_shape, int ndim, int *begin, int *size)
    +
    + +
    +
    +void hp_slice_p(half *input, half *output, int *input_shape, int ndim, int *begin, int *size)
    +
    + +
    +
    +void fp_slice_p(float *input, float *output, int *input_shape, int ndim, int *begin, int *size)
    +
    + +
    +
    +void dp_slice_p(double *input, double *output, int *input_shape, int ndim, int *begin, int *size)
    +
    + +
    +
    +void c64_slice_p(float *input, float *output, int *input_shape, int ndim, int *begin, int *size)
    +
    + +
    +
    +void c128_slice_p(double *input, double *output, int *input_shape, int ndim, int *begin, int *size)
    +

    C调用示例:

    +
     1// MT7004 示例(单核私有存储切片)
    + 2#include <stdio.h>
    + 3
    + 4int main() {
    + 5    float *input = (float *)0x10000000;
    + 6    float *output = (float *)0x10010000;
    + 7    int input_shape[] = {4, 8, 16, 32};
    + 8    int begin[] = {1, 2, 3, 4};
    + 9    int size[] = {2, 3, 6, 8};
    +10    int ndim = 4;
    +11
    +12    fp_slice_p(input, output, input_shape, ndim, begin, size);
    +13    return 0;
    +14}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/smooth1loss.html b/master/html/functionlib/dsplib/smooth1loss.html new file mode 100644 index 0000000..ad08b4a --- /dev/null +++ b/master/html/functionlib/dsplib/smooth1loss.html @@ -0,0 +1,431 @@ + + + + + + + + + SmoothL1Loss — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    SmoothL1Loss

    +

    计算平滑L1损失函数

    +
    +\[\begin{split}\text{loss}(x_1, x_2) = \begin{cases} + \frac{(x_1 - x_2)^2}{2\beta}, & \text{if } |x_1 - x_2| < \beta \\ + |x_1 - x_2| - \frac{\beta}{2}, & \text{otherwise} +\end{cases}\end{split}\]
    +

    其中 \(x_1\) 为预测值,\(x_2\) 为目标值,\(\beta\) 为平滑参数。

    +
    +
    输入:
      +
    • predict - 预测值数据地址。

    • +
    • target - 目标值数据地址。

    • +
    • length - 计算长度。

    • +
    • beta - 平滑参数(FT78NE平台为float值,MT7004平台为half*指针)。

    • +
    • core_mask(int, 可选) - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • out - 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持int8, fp32

    • +
    • MT7004 支持fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void i8_smoothl1loss_s(int8_t *out, int8_t *predict, int8_t *target, int length, float beta, int core_mask)
    +
    + +
    +
    +void fp_smoothl1loss_s(float *out, float *predict, float *target, int length, float beta, int core_mask)
    +
    + +
    +
    +void hp_smoothl1loss_s(half *out, half *predict, half *target, int length, half *beta, int core_mask)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <smoothl1loss.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *predict = (float *)0xA0000000;   //predict在DDR空间
    + 7    float *target = (float *)0xB0000000;    //target在DDR空间
    + 8    float *out = (float *)0xC0000000;       //out在DDR空间
    + 9    int length = 1000;
    +10    float beta = 1.0f;
    +11    int core_mask = 0xff;
    +12    fp_smoothl1loss_s(out, predict, target, length, beta, core_mask);
    +13    return 0;
    +14}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void i8_smoothl1loss_p(int8_t *out, int8_t *predict, int8_t *target, int length, float beta)
    +
    + +
    +
    +void fp_smoothl1loss_p(float *out, float *predict, float *target, int length, float beta)
    +
    + +
    +
    +void hp_smoothl1loss_p(half *out, half *predict, half *target, int length, half *beta)
    +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <smoothl1loss.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    float *predict = (float *)0x10810000;   //predict在L2空间
    + 7    float *target = (float *)0x10850000;    //target在L2空间
    + 8    float *out = (float *)0x108A0000;       //out在L2空间
    + 9    int length = 1000;
    +10    float beta = 1.0f;
    +11    fp_smoothl1loss_p(out, predict, target, length, beta);
    +12    return 0;
    +13}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/smoothl1lossgrad.html b/master/html/functionlib/dsplib/smoothl1lossgrad.html new file mode 100644 index 0000000..0f93e9d --- /dev/null +++ b/master/html/functionlib/dsplib/smoothl1lossgrad.html @@ -0,0 +1,434 @@ + + + + + + + + + Smoothl1lossgrad — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    +
    + +
    +
    +
    +
    + +
    +

    Smoothl1lossgrad

    +

    计算 Smooth L1 Loss 操作的梯度。该算子是 Smooth L1 Loss 算子的反向传播(backward pass)部分。

    +

    Smooth L1 Loss 是 L1 Loss 和 L2 Loss 的平滑组合,在损失值较小时使用 L2 Loss,在损失值较大时使用 L1 Loss,以减少异常值的影响。

    +
    +\[\text{diff}_i = \text{x1}_i - \text{x2}_i\]
    +
    +\[\begin{split}\text{dx1}_i = \begin{cases} + \text{dy}_i, & \text{if } \text{diff}_i > \beta \\ + -\text{dy}_i, & \text{if } \text{diff}_i < -\beta \\ + \frac{\text{diff}_i}{\beta} \times \text{dy}_i, & \text{if } -\beta \leq \text{diff}_i \leq \beta +\end{cases}\end{split}\]
    +

    其中 x1 是预测值(predict),x2 是目标值(target),dy 是来自后一层的上游梯度,dx1 是对预测值 x1 的梯度。beta 是平滑参数,控制从 L2 Loss 到 L1 Loss 的过渡点。

    +
    +
    输入:
      +
    • dy - 来自后一层的上游梯度数据地址。

    • +
    • x1 - 前向传播时的预测值数据地址。

    • +
    • x2 - 前向传播时的目标值数据地址。

    • +
    • length - 计算长度。

    • +
    • beta - 平滑参数,控制从 L2 Loss 到 L1 Loss 的过渡点。通常取值范围为 0.1 到 1.0。

    • +
    • core_mask - 核掩码(仅共享存储版本需要)。

    • +
    +
    +
    输出:
      +
    • dx1 - 计算出的对预测值 x1 的梯度数据地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp32

    • +
    • MT7004 支持fp16, fp32

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_smoothl1lossgrad_s(float *dy, float *dx1, float *x1, float *x2, int length, float beta, int core_mask)
    +
    + +
    +
    +void hp_smoothl1lossgrad_s(half *dy, half *dx1, half *x1, half *x2, int length, half beta, int core_mask)
    +
    + +

    C调用示例:

    +
     1//MT7004示例
    + 2#include <stdio.h>
    + 3#include <smoothl1lossgrad.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    // 假设在DDR空间
    + 7    float *dy = (float *)0xA0000000;   // 上游梯度
    + 8    float *x1 = (float *)0xA1000000;   // 预测值
    + 9    float *x2 = (float *)0xA2000000;   // 目标值
    +10    float *dx1 = (float *)0xB0000000;  // 输出梯度(对 x1 的梯度)
    +11
    +12    int length = 1000;
    +13    float beta = 1.0f;  // 平滑参数
    +14    int core_mask = 0xff;
    +15
    +16    fp_smoothl1lossgrad_s(dy, dx1, x1, x2, length, beta, core_mask);
    +17    return 0;
    +18}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_smoothl1lossgrad_p(float *dy, float *dx1, float *x1, float *x2, int length, float beta)
    +
    + +
    +
    +void hp_smoothl1lossgrad_p(half *dy, half *dx1, half *x1, half *x2, int length, half beta)
    +
    + +

    C调用示例:

    +
     1//MT7004示例
    + 2#include <stdio.h>
    + 3#include <smoothl1lossgrad.h>
    + 4
    + 5int main(int argc, char* argv[]) {
    + 6    // 假设在L2空间
    + 7    float *dy = (float *)0x10000000;   // 上游梯度
    + 8    float *x1 = (float *)0x10001000;   // 预测值
    + 9    float *x2 = (float *)0x10002000;   // 目标值
    +10    float *dx1 = (float *)0x10003000;   // 输出梯度(对 x1 的梯度)
    +11
    +12    int length = 1000;
    +13    float beta = 1.0f;  // 平滑参数
    +14
    +15    fp_smoothl1lossgrad_p(dy, dx1, x1, x2, length, beta);
    +16    return 0;
    +17}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/softmax.html b/master/html/functionlib/dsplib/softmax.html new file mode 100644 index 0000000..303e8c6 --- /dev/null +++ b/master/html/functionlib/dsplib/softmax.html @@ -0,0 +1,439 @@ + + + + + + + + + Softmax — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    + +
    +
    + +
    +

    Softmax

    +

    对输入数组沿指定维度进行 Softmax 计算,输出每个元素的概率值。

    +
    +\[\text{output}_{i} = \frac{\exp(\text{input}_{i})}{\sum_j \exp(\text{input}_j)} +\quad \text{for elements along the given axis}\]
    +
    +
    输入:
      +
    • input_ptr - 输入数据地址。

    • +
    • axis - 归一化的轴。

    • +
    • n_dim - 输入张量维度。

    • +
    • inner_size - 内部尺寸(轴之后的元素个数)。

    • +
    • outter_size - 外部尺寸(轴之前的元素个数)。

    • +
    • axis_size - 归一化轴的元素数量。

    • +
    • sum_data - 中间累加存储地址(用于存放指数和)。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • output_ptr - Softmax 计算结果地址。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持fp, int8

    • +
    • MT7004 支持hp, fp

    • +
    +
    +

    共享存储版本:

    +
    +
    +void fp_softmax_s(float *input_ptr, float *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask)
    +
    + +
    +
    +void hp_softmax_s(half *input_ptr, half *output_ptr, half *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask)
    +
    + +
    +
    +void i8_softmax_s(int8_t *input_ptr, int8_t *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size, int core_mask)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <softmax.h>
    + 4
    + 5int main() {
    + 6    float *input = (float *)0xA0000000;   // input在DDR空间
    + 7    float *output = (float *)0xC0000000;
    + 8    float *sum_data = (float *)0xD0000000;
    + 9    int axis = 1;
    +10    int n_dim = 3;
    +11    int inner_size = 4;
    +12    int outter_size = 2;
    +13    int axis_size = 3;
    +14    int core_mask = 0xff;
    +15
    +16    fp_softmax_s(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size, core_mask);
    +17    return 0;
    +18}
    +
    +
    +

    私有存储版本:

    +
    +
    +void fp_softmax_p(float *input_ptr, float *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size)
    +
    + +
    +
    +void hp_softmax_p(half *input_ptr, half *output_ptr, half *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size)
    +
    + +
    +
    +void i8_softmax_p(int8_t *input_ptr, int8_t *output_ptr, float *sum_data, int axis, int n_dim, int inner_size, int outter_size, int axis_size)
    +
    + +

    C调用示例:

    +
     1//FT78NE示例
    + 2#include <stdio.h>
    + 3#include <softmax.h>
    + 4
    + 5int main() {
    + 6    float *input = (float *)0x10810000;   // input在L2空间
    + 7    float *output = (float *)0x10820000;
    + 8    float *sum_data = (float *)0x10830000;
    + 9    int axis = 1;
    +10    int n_dim = 3;
    +11    int inner_size = 4;
    +12    int outter_size = 2;
    +13    int axis_size = 3;
    +14
    +15    fp_softmax_p(input, output, sum_data, axis, n_dim, inner_size, outter_size, axis_size);
    +16    return 0;
    +17}
    +
    +
    +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/softmax_cross_entropy_with_logits.html b/master/html/functionlib/dsplib/softmax_cross_entropy_with_logits.html new file mode 100644 index 0000000..d5c1403 --- /dev/null +++ b/master/html/functionlib/dsplib/softmax_cross_entropy_with_logits.html @@ -0,0 +1,471 @@ + + + + + + + + + SoftmaxCrossEntropyWithLogits — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    +
    + +
    +
    +
    +
    + +
    +

    SoftmaxCrossEntropyWithLogits

    +
    +

    计算 Softmax 交叉熵损失及梯度。

    +

    该算子首先对输入 logits 进行 Softmax 归一化得到概率,然后计算其与 labels 的交叉熵损失。如果启用了 need_grads,则会计算损失相对于 logits 的梯度。

    +

    算法逻辑如下:

    +
      +
    1. Softmax:

    2. +
    +
    +\[p_{i,j} = \frac{e^{x_{i,j}}}{\sum_{k} e^{x_{i,k}}}\]
    +
      +
    1. Cross Entropy Loss:

    2. +
    +
    +\[loss_i = - \sum_{j} y_{i,j} \log(p_{i,j})\]
    +
      +
    1. Gradients (当 need_grads=1 时):

    2. +
    +
    +\[\frac{\partial loss}{\partial x_{i,j}} = p_{i,j} - y_{i,j}\]
    +
    +
    输入:
      +
    • logits - 输入数据地址(未归一化的对数概率)。形状为 \([batch\_size, num\_of\_classes]\)

    • +
    • labels - 标签数据地址。形状为 \([batch\_size, num\_of\_classes]\)

    • +
    • probs - 输出概率地址(Softmax 结果)。

    • +
    • grads - 输出梯度地址。如果 need_grads 为 0,可忽略。

    • +
    • output - 输出损失值地址(通常为标量或每个样本的损失)。

    • +
    • sum_data - 中间计算缓冲区(Workspace),用于存储行和等临时数据。

    • +
    • batch_size - 批大小。

    • +
    • num_of_classes - 类别数量。

    • +
    • need_grads - 是否需要计算梯度 (0: 不计算, 1: 计算)。

    • +
    • core_mask - 核掩码(仅适用于共享存储版本)。

    • +
    +
    +
    输出:
      +
    • probs - 更新后的概率分布。

    • +
    • grads - 计算得到的梯度(如启用)。

    • +
    • output - 计算得到的交叉熵损失。

    • +
    +
    +
    支持平台:

    FT78NE +MT7004

    +
    +
    +
    +

    备注

    +
      +
    • FT78NE 支持 Fp32 和 Int8 数据类型。

    • +
    • MT7004 支持 Fp32 和 FP16 数据类型。

    • +
    • 输入数据必须是二维矩阵,行优先存储。

    • +
    • Int8 版本通常采用混合精度计算:输入/标签为 int8_t,但输出(概率、梯度、损失)为 float 以保证精度。

    • +
    +
    +
    +

    共享存储版本:

    +
    +
    +void fp_softmax_cross_entropy_with_logits_s(float *logits, float *labels, float *probs, float *grads, float *output, float *sum_data, int batch_size, int num_of_classes, int need_grads, int core_mask)
    +
    + +
    +
    +void hp_softmax_cross_entropy_with_logits_s(float16 *logits, float16 *labels, float16 *probs, float16 *grads, float16 *output, float16 *sum_data, int batch_size, int num_of_classes, int need_grads, int core_mask)
    +
    + +
    +
    +void i8_softmax_cross_entropy_with_logits_s(int8_t *logits, int8_t *labels, float *probs, float *grads, float *output, float *sum_data, int batch_size, int num_of_classes, int need_grads, int core_mask)
    +

    C调用示例(FT78NE - Int8):

    +
     1#include <stdio.h>
    + 2#include <stdint.h>
    + 3
    + 4int main(int argc, char* argv[]) {
    + 5    // 假设所有数据均位于DDR空间
    + 6    int8_t* logits = (int8_t*)0xC0000000;
    + 7    int8_t* labels = (int8_t*)0xC1000000;
    + 8    float* probs   = (float*)0xC2000000;
    + 9    float* grads   = (float*)0xC3000000;
    +10    float* output  = (float*)0xC4000000;
    +11    float* sum_data= (float*)0xC5000000; // Workspace
    +12
    +13    int batch_size = 128;
    +14    int num_of_classes = 1000;
    +15    int need_grads = 1;
    +16    int core_mask = 0xff; // 使用所有核心
    +17
    +18    // i8版本:输入为int8,输出及中间计算为float
    +19    i8_softmax_cross_entropy_with_logits_s(logits, labels, probs, grads, output,
    +20                                           sum_data, batch_size, num_of_classes,
    +21                                           need_grads, core_mask);
    +22
    +23    return 0;
    +24}
    +
    +
    +
    + +

    私有存储版本:

    +
    +
    +void fp_softmax_cross_entropy_with_logits_p(float *logits, float *labels, float *probs, float *grads, float *output, float *sum_data, int batch_size, int num_of_classes, int need_grads)
    +
    + +
    +
    +void hp_softmax_cross_entropy_with_logits_p(float16 *logits, float16 *labels, float16 *probs, float16 *grads, float16 *output, float16 *sum_data, int batch_size, int num_of_classes, int need_grads)
    +
    + +
    +
    +void i8_softmax_cross_entropy_with_logits_p(int8_t *logits, int8_t *labels, float *probs, float *grads, float *output, float *sum_data, int batch_size, int num_of_classes, int need_grads)
    +

    C调用示例(MT7004 - FP16):

    +
     1#include <stdio.h>
    + 2
    + 3int main(int argc, char* argv[]) {
    + 4    // 假设所有数据均位于AM空间
    + 5    float16* logits = (float16*)0x10010000;
    + 6    float16* labels = (float16*)0x10020000;
    + 7    float16* probs  = (float16*)0x10030000;
    + 8    float16* grads  = (float16*)0x10040000;
    + 9    float16* output = (float16*)0x10050000;
    +10    float16* sum_data=(float16*)0x10060000;
    +11
    +12    int batch_size = 32;
    +13    int num_of_classes = 512;
    +14    int need_grads = 1;
    +15
    +16    hp_softmax_cross_entropy_with_logits_p(logits, labels, probs, grads, output,
    +17                                            sum_data, batch_size, num_of_classes,
    +18                                            need_grads);
    +19
    +20    return 0;
    +21}
    +
    +
    +
    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/functionlib/dsplib/spacetobatch.html b/master/html/functionlib/dsplib/spacetobatch.html index c2f9966..3017970 100644 --- a/master/html/functionlib/dsplib/spacetobatch.html +++ b/master/html/functionlib/dsplib/spacetobatch.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
    @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
      -
    • - +
    • +
    • @@ -287,7 +452,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/spacetobatchnd.html b/master/html/functionlib/dsplib/spacetobatchnd.html index 2de5398..108719d 100644 --- a/master/html/functionlib/dsplib/spacetobatchnd.html +++ b/master/html/functionlib/dsplib/spacetobatchnd.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
      @@ -46,11 +46,7 @@
    @@ -111,15 +276,15 @@
      -
    • - +
    • +
    • @@ -287,7 +452,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/functionlib/dsplib/spacetodepth.html b/master/html/functionlib/dsplib/spacetodepth.html index 3ad7df4..f246332 100644 --- a/master/html/functionlib/dsplib/spacetodepth.html +++ b/master/html/functionlib/dsplib/spacetodepth.html @@ -35,7 +35,7 @@ - + MindSpore Signal+ 使用手册
      @@ -47,11 +47,7 @@
    @@ -112,15 +277,15 @@
    @@ -111,15 +276,15 @@
    @@ -111,15 +276,15 @@
    @@ -64,14 +54,14 @@
    @@ -149,15 +309,12 @@
    -
    @@ -63,15 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/genindex.html b/master/html/genindex.html index 8b0b052..8901dbf 100644 --- a/master/html/genindex.html +++ b/master/html/genindex.html @@ -31,7 +31,7 @@ - + MindSpore Signal+ 使用手册
    @@ -43,10 +43,7 @@
    @@ -55,14 +52,14 @@
      -
    • +
    • @@ -83,6 +80,10 @@ | H | I | M + | R + | S + | T + | U

    A

    @@ -114,6 +115,38 @@

    C

    - +
    @@ -218,6 +811,42 @@

    D

    - +
    @@ -298,6 +1227,20 @@

    F

    - +
    @@ -518,6 +2105,14 @@

    H

    - +
    @@ -714,6 +2881,42 @@

    I

    + -
    @@ -1013,6 +4216,92 @@ +

    R

    + + + +
    + +

    S

    + + + +
    + +

    T

    + + + +
    + +

    U

    + + + +
    +
    @@ -1022,7 +4311,7 @@
    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/index.html b/master/html/index.html index 921fd44..593ece6 100644 --- a/master/html/index.html +++ b/master/html/index.html @@ -22,8 +22,7 @@ - - + @@ -34,7 +33,7 @@ - + MindSpore Signal+ 使用手册
    @@ -46,10 +45,7 @@
    @@ -58,14 +54,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/objects.inv b/master/html/objects.inv index f768bef..f062a07 100644 Binary files a/master/html/objects.inv and b/master/html/objects.inv differ diff --git a/master/html/quickstart/hellodsp.html b/master/html/quickstart/hellodsp.html index e456186..b61e61c 100644 --- a/master/html/quickstart/hellodsp.html +++ b/master/html/quickstart/hellodsp.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -45,33 +43,8 @@
    @@ -80,15 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/quickstart/index.html b/master/html/quickstart/index.html index 5f85e73..617ee06 100644 --- a/master/html/quickstart/index.html +++ b/master/html/quickstart/index.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -45,16 +43,8 @@
    @@ -63,14 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/quickstart/installation.html b/master/html/quickstart/installation.html index 8a95270..899676c 100644 --- a/master/html/quickstart/installation.html +++ b/master/html/quickstart/installation.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -45,28 +43,8 @@
    @@ -75,15 +53,14 @@
      -
    • - +
    • 查看页面源码 @@ -256,15 +233,12 @@ The result of multiplication calculation is correct, MindSpore has been installe
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/quickstart/overview.html b/master/html/quickstart/overview.html index 3c1e155..172ed67 100644 --- a/master/html/quickstart/overview.html +++ b/master/html/quickstart/overview.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -45,16 +43,8 @@
    @@ -63,15 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/refdoc/dsplib_description.html b/master/html/refdoc/dsplib_description.html index 8793252..344cc75 100644 --- a/master/html/refdoc/dsplib_description.html +++ b/master/html/refdoc/dsplib_description.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -45,24 +43,8 @@
    @@ -71,15 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/refdoc/index.html b/master/html/refdoc/index.html index 7cf14a3..7387c0f 100644 --- a/master/html/refdoc/index.html +++ b/master/html/refdoc/index.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -45,16 +43,8 @@
    @@ -63,14 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/refdoc/mindspore.html b/master/html/refdoc/mindspore.html index 3097795..b4e61ea 100644 --- a/master/html/refdoc/mindspore.html +++ b/master/html/refdoc/mindspore.html @@ -21,9 +21,7 @@ - - - + @@ -34,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -45,16 +43,8 @@
    @@ -63,15 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/refdoc/signal_scheduling.html b/master/html/refdoc/signal_scheduling.html index 5c5de47..a4bac91 100644 --- a/master/html/refdoc/signal_scheduling.html +++ b/master/html/refdoc/signal_scheduling.html @@ -21,8 +21,7 @@ - - + @@ -33,7 +32,7 @@ - + MindSpore Signal+ 使用手册
    @@ -44,16 +43,8 @@
    @@ -62,15 +53,14 @@
    -
    +

    -

    © 版权所有 2025 - 2025, NUDT-674。

    +

    © 版权所有 2025 - 2026, NUDT-674。

    利用 Sphinx 构建,使用的 diff --git a/master/html/rstfiles.html b/master/html/rstfiles.html new file mode 100644 index 0000000..4d2c19d --- /dev/null +++ b/master/html/rstfiles.html @@ -0,0 +1,118 @@ + + + + + + + + + MindSpore Signal+ 使用手册 — MindSpore Signal+ 使用手册 alpha 文档 + + + + + + + + + + + + + + + + + + + + +
    + + +
    + +
    +
    +
    + +
    +
    +
    +
    + +
    +

    MindSpore Signal+ 使用手册

    + +
    + + +
    +
    + +
    +
    +
    +
    + + + + \ No newline at end of file diff --git a/master/html/search.html b/master/html/search.html index f05e2a0..568abb1 100644 --- a/master/html/search.html +++ b/master/html/search.html @@ -34,7 +34,7 @@ - + MindSpore Signal+ 使用手册
    @@ -46,10 +46,7 @@
    @@ -58,14 +55,14 @@
      -
    • +
    • @@ -95,7 +92,7 @@
      -

      © 版权所有 2025 - 2025, NUDT-674。

      +

      © 版权所有 2025 - 2026, NUDT-674。

      利用 Sphinx 构建,使用的 diff --git a/master/html/searchindex.js b/master/html/searchindex.js index da54625..57d6a2d 100644 --- a/master/html/searchindex.js +++ b/master/html/searchindex.js @@ -1 +1 @@ -Search.setIndex({"alltitles": {"1. \u57fa\u672c\u539f\u5219": [[64, "id1"]], "1. \u5b9a\u4e49\u6a21\u578b": [[0, "id3"]], "1. \u6570\u636e\u8bfb\u53d6": [[4, "id2"]], "1. \u65b0\u5efa\u5de5\u7a0b": [[60, "id4"]], "1.1 \u4e0b\u8f7d\u5b89\u88c5\u5305": [[62, "id3"]], "1.1 \u65b0\u5efaPython\u6587\u4ef6": [[60, "id1"]], "1.2 \u5f00\u59cb\u5b89\u88c5": [[62, "id4"]], "1.2 \u7f16\u5199Python\u4ee3\u7801": [[60, "id2"]], "1.3 \u8fd0\u884cPython\u4ee3\u7801": [[60, "id3"]], "1.3 \u9a8c\u8bc1\u5b89\u88c5": [[62, "id5"]], "1.MindSpore Python\u7aef": [[60, "mindspore-python"]], "1.\u5b89\u88c5\u5305\u65b9\u5f0f\u5b89\u88c5": [[62, "id2"]], "2. \u5177\u4f53\u547d\u540d\u793a\u4f8b": [[64, "id2"]], "2. \u5de5\u7a0b\u76ee\u5f55\u7ed3\u6784": [[60, "id5"]], "2. \u6570\u636e\u9884\u5904\u7406": [[4, "id3"]], "2. \u6a21\u578b\u8bad\u7ec3": [[0, "model-training"]], "2.1 \u52a0\u6cd5\u7b97\u5b50": [[64, "id3"]], "2.Conda\u65b9\u5f0f\u5b89\u88c5": [[62, "conda"]], "3. \u5176\u4ed6\u6ce8\u610f\u4e8b\u9879": [[64, "id4"]], "3. \u66f4\u6539\u8f93\u5165\u6570\u636e": [[60, "id6"]], "3. \u6838\u5fc3\u8ba1\u7b97": [[4, "id4"]], "3. \u91cd\u65b0\u7ec4\u7f51": [[0, "id5"]], "4. main.cc \u529f\u80fd\u4ecb\u7ecd": [[60, "main-cc"]], "4. \u6570\u636e\u540e\u5904\u7406": [[4, "id5"]], "4. \u8f6c\u6362\u6a21\u578b": [[0, "id6"]], "4.1 \u8bfb\u53d6\u6a21\u578b\u6587\u4ef6": [[60, "id7"]], "4.2 \u8bbe\u7f6e\u8fd0\u884c\u540e\u7aef": [[60, "id8"]], "4.3 \u7f16\u8bd1\u6a21\u578b\u56fe": [[60, "id9"]], "4.4 \u83b7\u53d6\u6a21\u578b\u8f93\u5165": [[60, "id10"]], "4.5 \u6267\u884c\u6a21\u578b\u63a8\u7406": [[60, "id11"]], "4.6 \u83b7\u53d6\u6a21\u578b\u7ed3\u679c": [[60, "id12"]], "5. \u7f16\u8bd1\u5de5\u7a0b": [[60, "id13"]], "5. \u90e8\u7f72\u548c\u8fd0\u884c\u7a0b\u5e8f": [[0, "id7"]], "6. \u8fd0\u884c\u5de5\u7a0b": [[60, "id14"]], "6.1 \u8fde\u63a5 MT7004 \u677f\u5361": [[60, "mt7004"]], "6.2 \u90e8\u7f72\u548c\u8fd0\u884c\u7a0b\u5e8f": [[60, "id15"]], "AI+DSP\u5e94\u7528\u793a\u4f8b": [[1, null]], "AI\u8f85\u52a9\u5f00\u53d1\u793a\u4f8b": [[2, null]], "Activation": [[10, null]], "AdamWeightDecay": [[11, null]], "Adder": [[12, null]], "ApplyMomentum": [[13, null]], "Assert": [[14, null]], "Attention": [[15, null]], "AvgPoolingGrad": [[16, null]], "BatchToSpace": [[17, null]], "BatchToSpaceND": [[18, null]], "BroadcastTo": [[19, null]], "ComplexAbs": [[6, null]], "Conv2DBackpropFilterFusion": [[22, null]], "Conv2DBackpropInputFusion": [[23, null]], "Conv2d": [[20, null]], "Conv2dTranspose": [[21, null]], "Crop": [[24, null]], "CropAndResize": [[25, null]], "DSP Library C API Reference": [[27, null]], "DSP\u5e94\u7528\u793a\u4f8b": [[3, null]], "DSP\u7b97\u5b50C\u63a5\u53e3\u547d\u540d\u89c4\u8303": [[64, null]], "DepthToSpace": [[26, null]], "Eltwise": [[28, null]], "EmbeddingLookup": [[29, null]], "Equal": [[30, null]], "ExpFusion": [[32, null]], "ExpandDims": [[31, null]], "FFT": [[7, null]], "FillV2": [[33, null]], "Floor": [[34, null]], "FloorDiv": [[35, null]], "FusedBatchNorm": [[36, null]], "GRAY_CNN(\u56fe\u7247\u7070\u5ea6\u5316\u5904\u7406+\u56fe\u7247\u8bc6\u522b)": [[0, null]], "GRU": [[38, null]], "GroupNormFusion": [[37, null]], "HelloDSP": [[60, null]], "IFFT": [[8, null]], "LSTM": [[41, null]], "LeakyReLu": [[39, null]], "LinSpace": [[40, null]], "MATLAB \u5b9e\u73b0": [[4, "matlab"]], "MatMulFusion": [[42, null]], "MindSpore Lite\u7aef": [[60, "mindspore-lite"]], "MindSpore Signal+ \u4f7f\u7528\u624b\u518c": [[59, null]], "MindSpore Signal+ \u5b9e\u73b0": [[4, "mindspore-signal"]], "MindSpore Signal+ \u8c03\u5ea6\u65b9\u6848": [[67, null]], "RDSAR\uff08\u8ddd\u79bb-\u591a\u666e\u52d2SAR\u6210\u50cf\u7b97\u6cd5\uff09": [[4, null]], "RaggedRange": [[43, null]], "Range": [[44, null]], "Reduce": [[45, null]], "Resize": [[46, null]], "ReverseSequence": [[47, null]], "ReverseV2": [[48, null]], "SGD": [[51, null]], "ScaleFusion": [[49, null]], "ScatterElements": [[50, null]], "SpaceToBatch": [[52, null]], "SpaceToBatchND": [[53, null]], "SpaceToDepth": [[54, null]], "Squeeze": [[55, null]], "UnSqueeze": [[56, null]], "python\u5b8c\u6574\u4ee3\u7801\u793a\u4f8b": [[0, "python"]], "\u521b\u5efa\u5e76\u8fdb\u5165Conda\u865a\u62df\u73af\u5883": [[62, "id6"]], "\u53c2\u8003\u4e0e\u6e90\u7801": [[4, "id7"]], "\u53c2\u8003\u8d44\u6599": [[65, null]], "\u5b89\u88c5CMake": [[62, "cmake"]], "\u5b89\u88c5MindRadar": [[62, "mindradar"]], "\u5b89\u88c5MindSpore": [[62, "mindspore"]], "\u5b89\u88c5MindSpore Radar \u4e0e\u4f9d\u8d56\u8f6f\u4ef6": [[62, "mindspore-radar"]], "\u5b89\u88c5Netron": [[62, "netron"]], "\u5b89\u88c5Python": [[62, "python"]], "\u5b89\u88c5YHFT-IDE": [[62, "yhft-ide"]], "\u5b98\u65b9\u8d44\u6599": [[66, null]], "\u5e94\u7528\u5f00\u53d1\u793a\u4f8b": [[5, null]], "\u5e94\u7528\u6982\u8ff0": [[0, "id1"]], "\u5f00\u53d1\u6d41\u7a0b": [[0, "id2"]], "\u5feb\u901f\u5165\u95e8": [[61, null]], "\u6574\u4f53\u6982\u89c8": [[63, null]], "\u677f\u5361\u90e8\u7f72": [[4, "id6"]], "\u73af\u5883\u5b89\u88c5": [[62, null]], "\u7b97\u5b50\u5e93\u652f\u6301": [[57, null]], "\u7b97\u5b50\u5e93\u652f\u6301\u60c5\u51b5": [[58, null]], "\u7b97\u6cd5\u6982\u8ff0": [[4, "id1"]], "\u81ea\u5b9a\u4e49\u7b97\u5b50\u5217\u8868": [[9, null]], "\u914d\u7f6e\u4ea4\u53c9\u7f16\u8bd1\u5de5\u5177\u94fe": [[62, "toolchain"]]}, "docnames": ["appdevelop/ai_dsp/gray_cnn", "appdevelop/ai_dsp/index", "appdevelop/autocodegen/index", "appdevelop/dsp/index", "appdevelop/dsp/rdsar", "appdevelop/index", "functionlib/custom_op/complex_abs", "functionlib/custom_op/fft", "functionlib/custom_op/ifft", "functionlib/custom_op/index", "functionlib/dsplib/activation", "functionlib/dsplib/adamweightdecay", "functionlib/dsplib/adder", "functionlib/dsplib/applymomentum", "functionlib/dsplib/assert", "functionlib/dsplib/attention", "functionlib/dsplib/avgpoolinggrad", "functionlib/dsplib/batchtospace", "functionlib/dsplib/batchtospacend", "functionlib/dsplib/broadcastto", "functionlib/dsplib/conv2d", "functionlib/dsplib/conv2d_transpose", "functionlib/dsplib/conv2dbackpropfilterfusion", "functionlib/dsplib/conv2dbackpropinputfusion", "functionlib/dsplib/crop", "functionlib/dsplib/crop_and_resize", "functionlib/dsplib/depthtospace", "functionlib/dsplib/dsplib_index", "functionlib/dsplib/eltwise", "functionlib/dsplib/embeddinglookup", "functionlib/dsplib/equal", "functionlib/dsplib/expand_dims", "functionlib/dsplib/expfusion", "functionlib/dsplib/fillv2", "functionlib/dsplib/floor", "functionlib/dsplib/floordiv", "functionlib/dsplib/fusedbatchnorm", "functionlib/dsplib/groupnormfusion", "functionlib/dsplib/gru", "functionlib/dsplib/leaky_relu", "functionlib/dsplib/linspace", "functionlib/dsplib/lstm", "functionlib/dsplib/matmulfusion", "functionlib/dsplib/raggedrange", "functionlib/dsplib/range", "functionlib/dsplib/reduce", "functionlib/dsplib/resize", "functionlib/dsplib/reverse_sequence", "functionlib/dsplib/reversev2", "functionlib/dsplib/scalefusion", "functionlib/dsplib/scatter_elements", "functionlib/dsplib/sgd", "functionlib/dsplib/spacetobatch", "functionlib/dsplib/spacetobatchnd", "functionlib/dsplib/spacetodepth", "functionlib/dsplib/squeeze", "functionlib/dsplib/unsqueeze", "functionlib/index", "functionlib/supported_op", "index", "quickstart/hellodsp", "quickstart/index", "quickstart/installation", "quickstart/overview", "refdoc/dsplib_description", "refdoc/index", "refdoc/mindspore", "refdoc/signal_scheduling"], "envversion": {"sphinx": 64, "sphinx.domains.c": 3, "sphinx.domains.changeset": 1, "sphinx.domains.citation": 1, "sphinx.domains.cpp": 9, "sphinx.domains.index": 1, "sphinx.domains.javascript": 3, "sphinx.domains.math": 2, "sphinx.domains.python": 4, "sphinx.domains.rst": 2, "sphinx.domains.std": 2}, "filenames": ["appdevelop/ai_dsp/gray_cnn.rst", "appdevelop/ai_dsp/index.rst", "appdevelop/autocodegen/index.rst", "appdevelop/dsp/index.rst", "appdevelop/dsp/rdsar.md", "appdevelop/index.rst", "functionlib/custom_op/complex_abs.rst", "functionlib/custom_op/fft.rst", "functionlib/custom_op/ifft.rst", "functionlib/custom_op/index.rst", "functionlib/dsplib/activation.rst", "functionlib/dsplib/adamweightdecay.rst", "functionlib/dsplib/adder.rst", "functionlib/dsplib/applymomentum.rst", "functionlib/dsplib/assert.rst", "functionlib/dsplib/attention.rst", "functionlib/dsplib/avgpoolinggrad.rst", "functionlib/dsplib/batchtospace.rst", "functionlib/dsplib/batchtospacend.rst", "functionlib/dsplib/broadcastto.rst", "functionlib/dsplib/conv2d.rst", "functionlib/dsplib/conv2d_transpose.rst", "functionlib/dsplib/conv2dbackpropfilterfusion.rst", "functionlib/dsplib/conv2dbackpropinputfusion.rst", "functionlib/dsplib/crop.rst", "functionlib/dsplib/crop_and_resize.rst", "functionlib/dsplib/depthtospace.rst", "functionlib/dsplib/dsplib_index.rst", "functionlib/dsplib/eltwise.rst", "functionlib/dsplib/embeddinglookup.rst", "functionlib/dsplib/equal.rst", "functionlib/dsplib/expand_dims.rst", "functionlib/dsplib/expfusion.rst", "functionlib/dsplib/fillv2.rst", "functionlib/dsplib/floor.rst", "functionlib/dsplib/floordiv.rst", "functionlib/dsplib/fusedbatchnorm.rst", "functionlib/dsplib/groupnormfusion.rst", "functionlib/dsplib/gru.rst", "functionlib/dsplib/leaky_relu.rst", "functionlib/dsplib/linspace.rst", "functionlib/dsplib/lstm.rst", "functionlib/dsplib/matmulfusion.rst", "functionlib/dsplib/raggedrange.rst", "functionlib/dsplib/range.rst", "functionlib/dsplib/reduce.rst", "functionlib/dsplib/resize.rst", "functionlib/dsplib/reverse_sequence.rst", "functionlib/dsplib/reversev2.rst", "functionlib/dsplib/scalefusion.rst", "functionlib/dsplib/scatter_elements.rst", "functionlib/dsplib/sgd.rst", "functionlib/dsplib/spacetobatch.rst", "functionlib/dsplib/spacetobatchnd.rst", "functionlib/dsplib/spacetodepth.rst", "functionlib/dsplib/squeeze.rst", "functionlib/dsplib/unsqueeze.rst", "functionlib/index.rst", "functionlib/supported_op.md", "index.rst", "quickstart/hellodsp.md", "quickstart/index.rst", "quickstart/installation.md", "quickstart/overview.md", "refdoc/dsplib_description.md", "refdoc/index.rst", "refdoc/mindspore.md", "refdoc/signal_scheduling.md"], "indexentries": {"anytype_crop_anycore\uff08c function\uff09": [[24, "c.anytype_crop_anycore", false]], "anytype_expand_dims_anycore\uff08c function\uff09": [[31, "c.anytype_expand_dims_anycore", false]], "anytype_fillv2_p\uff08c function\uff09": [[33, "c.anytype_fillv2_p", false]], "anytype_fillv2_s\uff08c function\uff09": [[33, "c.anytype_fillv2_s", false]], "anytype_reverse_sequence_anycore\uff08c function\uff09": [[47, "c.anytype_reverse_sequence_anycore", false]], "anytype_reversev2_anycore\uff08c function\uff09": [[48, "c.anytype_reversev2_anycore", false]], "anytype_squeeze_anycore\uff08c function\uff09": [[55, "c.anytype_squeeze_anycore", false]], "anytype_unsqueeze_anycore\uff08c function\uff09": [[56, "c.anytype_unsqueeze_anycore", false]], "assert\uff08c function\uff09": [[14, "c.assert", false]], "c128_batchtospace_p\uff08c function\uff09": [[17, "c.c128_batchtospace_p", false]], "c128_batchtospace_s\uff08c function\uff09": [[17, "c.c128_batchtospace_s", false]], "c128_batchtospacend_p\uff08c function\uff09": [[18, "c.c128_batchtospacend_p", false]], "c128_batchtospacend_s\uff08c function\uff09": [[18, "c.c128_batchtospacend_s", false]], "c128_broadcastto_p\uff08c function\uff09": [[19, "c.c128_broadcastto_p", false]], "c128_broadcastto_s\uff08c function\uff09": [[19, "c.c128_broadcastto_s", false]], "c128_depthtospace_p\uff08c function\uff09": [[26, "c.c128_depthtospace_p", false]], "c128_depthtospace_s\uff08c function\uff09": [[26, "c.c128_depthtospace_s", false]], "c128_eltwise_p\uff08c function\uff09": [[28, "c.c128_eltwise_p", false]], "c128_eltwise_s\uff08c function\uff09": [[28, "c.c128_eltwise_s", false]], "c128_equal_p\uff08c function\uff09": [[30, "c.c128_equal_p", false]], "c128_equal_s\uff08c function\uff09": [[30, "c.c128_equal_s", false]], "c128_expfusion_p\uff08c function\uff09": [[32, "c.c128_expfusion_p", false]], "c128_expfusion_s\uff08c function\uff09": [[32, "c.c128_expfusion_s", false]], "c128_matmulfusion_p\uff08c function\uff09": [[42, "c.c128_matmulfusion_p", false]], "c128_matmulfusion_s\uff08c function\uff09": [[42, "c.c128_matmulfusion_s", false]], "c128_scatter_elements_p\uff08c function\uff09": [[50, "c.c128_scatter_elements_p", false]], "c128_scatter_elements_s\uff08c function\uff09": [[50, "c.c128_scatter_elements_s", false]], "c128_spacetobatch_p\uff08c function\uff09": [[52, "c.c128_spacetobatch_p", false]], "c128_spacetobatch_s\uff08c function\uff09": [[52, "c.c128_spacetobatch_s", false]], "c128_spacetobatchnd_p\uff08c function\uff09": [[53, "c.c128_spacetobatchnd_p", false]], "c128_spacetobatchnd_s\uff08c function\uff09": [[53, "c.c128_spacetobatchnd_s", false]], "c128_spacetodepth_p\uff08c function\uff09": [[54, "c.c128_spacetodepth_p", false]], "c128_spacetodepth_s\uff08c function\uff09": [[54, "c.c128_spacetodepth_s", false]], "c64_batchtospace_p\uff08c function\uff09": [[17, "c.c64_batchtospace_p", false]], "c64_batchtospace_s\uff08c function\uff09": [[17, "c.c64_batchtospace_s", false]], "c64_batchtospacend_p\uff08c function\uff09": [[18, "c.c64_batchtospacend_p", false]], "c64_batchtospacend_s\uff08c function\uff09": [[18, "c.c64_batchtospacend_s", false]], "c64_broadcastto_p\uff08c function\uff09": [[19, "c.c64_broadcastto_p", false]], "c64_broadcastto_s\uff08c function\uff09": [[19, "c.c64_broadcastto_s", false]], "c64_depthtospace_p\uff08c function\uff09": [[26, "c.c64_depthtospace_p", false]], "c64_depthtospace_s\uff08c function\uff09": [[26, "c.c64_depthtospace_s", false]], "c64_eltwise_p\uff08c function\uff09": [[28, "c.c64_eltwise_p", false]], "c64_eltwise_s\uff08c function\uff09": [[28, "c.c64_eltwise_s", false]], "c64_equal_p\uff08c function\uff09": [[30, "c.c64_equal_p", false]], "c64_equal_s\uff08c function\uff09": [[30, "c.c64_equal_s", false]], "c64_expfusion_p\uff08c function\uff09": [[32, "c.c64_expfusion_p", false]], "c64_expfusion_s\uff08c function\uff09": [[32, "c.c64_expfusion_s", false]], "c64_matmulfusion_p\uff08c function\uff09": [[42, "c.c64_matmulfusion_p", false]], "c64_matmulfusion_s\uff08c function\uff09": [[42, "c.c64_matmulfusion_s", false]], "c64_scatter_elements_p\uff08c function\uff09": [[50, "c.c64_scatter_elements_p", false]], "c64_scatter_elements_s\uff08c function\uff09": [[50, "c.c64_scatter_elements_s", false]], "c64_spacetobatch_p\uff08c function\uff09": [[52, "c.c64_spacetobatch_p", false]], "c64_spacetobatch_s\uff08c function\uff09": [[52, "c.c64_spacetobatch_s", false]], "c64_spacetobatchnd_p\uff08c function\uff09": [[53, "c.c64_spacetobatchnd_p", false]], "c64_spacetobatchnd_s\uff08c function\uff09": [[53, "c.c64_spacetobatchnd_s", false]], "c64_spacetodepth_p\uff08c function\uff09": [[54, "c.c64_spacetodepth_p", false]], "c64_spacetodepth_s\uff08c function\uff09": [[54, "c.c64_spacetodepth_s", false]], "dp_batchtospace_p\uff08c function\uff09": [[17, "c.dp_batchtospace_p", false]], "dp_batchtospace_s\uff08c function\uff09": [[17, "c.dp_batchtospace_s", false]], "dp_batchtospacend_p\uff08c function\uff09": [[18, "c.dp_batchtospacend_p", false]], "dp_batchtospacend_s\uff08c function\uff09": [[18, "c.dp_batchtospacend_s", false]], "dp_broadcastto_p\uff08c function\uff09": [[19, "c.dp_broadcastto_p", false]], "dp_broadcastto_s\uff08c function\uff09": [[19, "c.dp_broadcastto_s", false]], "dp_depthtospace_p\uff08c function\uff09": [[26, "c.dp_depthtospace_p", false]], "dp_depthtospace_s\uff08c function\uff09": [[26, "c.dp_depthtospace_s", false]], "dp_eltwise_p\uff08c function\uff09": [[28, "c.dp_eltwise_p", false]], "dp_eltwise_s\uff08c function\uff09": [[28, "c.dp_eltwise_s", false]], "dp_equal_p\uff08c function\uff09": [[30, "c.dp_equal_p", false]], "dp_equal_s\uff08c function\uff09": [[30, "c.dp_equal_s", false]], "dp_expfusion_p\uff08c function\uff09": [[32, "c.dp_expfusion_p", false]], "dp_expfusion_s\uff08c function\uff09": [[32, "c.dp_expfusion_s", false]], "dp_floor_p\uff08c function\uff09": [[34, "c.dp_floor_p", false]], "dp_floor_s\uff08c function\uff09": [[34, "c.dp_floor_s", false]], "dp_floordiv_p\uff08c function\uff09": [[35, "c.dp_floordiv_p", false]], "dp_floordiv_s\uff08c function\uff09": [[35, "c.dp_floordiv_s", false]], "dp_matmulfusion_p\uff08c function\uff09": [[42, "c.dp_matmulfusion_p", false]], "dp_matmulfusion_s\uff08c function\uff09": [[42, "c.dp_matmulfusion_s", false]], "dp_raggedrange_p\uff08c function\uff09": [[43, "c.dp_raggedrange_p", false]], "dp_raggedrange_s\uff08c function\uff09": [[43, "c.dp_raggedrange_s", false]], "dp_range_p\uff08c function\uff09": [[44, "c.dp_range_p", false]], "dp_range_s\uff08c function\uff09": [[44, "c.dp_range_s", false]], "dp_reduce_p\uff08c function\uff09": [[45, "c.dp_reduce_p", false]], "dp_reduce_s\uff08c function\uff09": [[45, "c.dp_reduce_s", false]], "dp_scalefusion_p\uff08c function\uff09": [[49, "c.dp_scalefusion_p", false]], "dp_scalefusion_s\uff08c function\uff09": [[49, "c.dp_scalefusion_s", false]], "dp_scatter_elements_p\uff08c function\uff09": [[50, "c.dp_scatter_elements_p", false]], "dp_scatter_elements_s\uff08c function\uff09": [[50, "c.dp_scatter_elements_s", false]], "dp_spacetobatch_p\uff08c function\uff09": [[52, "c.dp_spacetobatch_p", false]], "dp_spacetobatch_s\uff08c function\uff09": [[52, "c.dp_spacetobatch_s", false]], "dp_spacetobatchnd_p\uff08c function\uff09": [[53, "c.dp_spacetobatchnd_p", false]], "dp_spacetobatchnd_s\uff08c function\uff09": [[53, "c.dp_spacetobatchnd_s", false]], "dp_spacetodepth_p\uff08c function\uff09": [[54, "c.dp_spacetodepth_p", false]], "dp_spacetodepth_s\uff08c function\uff09": [[54, "c.dp_spacetodepth_s", false]], "fp_adamweightdecay_p\uff08c function\uff09": [[11, "c.fp_adamweightdecay_p", false]], "fp_adamweightdecay_s\uff08c function\uff09": [[11, "c.fp_adamweightdecay_s", false]], "fp_adder_p\uff08c function\uff09": [[12, "c.fp_adder_p", false]], "fp_adder_s\uff08c function\uff09": [[12, "c.fp_adder_s", false]], "fp_applymomentum_p\uff08c function\uff09": [[13, "c.fp_applymomentum_p", false]], "fp_applymomentum_s\uff08c function\uff09": [[13, "c.fp_applymomentum_s", false]], "fp_attention_p\uff08c function\uff09": [[15, "c.fp_attention_p", false]], "fp_attention_s\uff08c function\uff09": [[15, "c.fp_attention_s", false]], "fp_avgpoolinggrad_p\uff08c function\uff09": [[16, "c.fp_avgpoolinggrad_p", false]], "fp_avgpoolinggrad_s\uff08c function\uff09": [[16, "c.fp_avgpoolinggrad_s", false]], "fp_batchtospace_p\uff08c function\uff09": [[17, "c.fp_batchtospace_p", false]], "fp_batchtospace_s\uff08c function\uff09": [[17, "c.fp_batchtospace_s", false]], "fp_batchtospacend_p\uff08c function\uff09": [[18, "c.fp_batchtospacend_p", false]], "fp_batchtospacend_s\uff08c function\uff09": [[18, "c.fp_batchtospacend_s", false]], "fp_broadcastto_p\uff08c function\uff09": [[19, "c.fp_broadcastto_p", false]], "fp_broadcastto_s\uff08c function\uff09": [[19, "c.fp_broadcastto_s", false]], "fp_celu_p\uff08c function\uff09": [[10, "c.fp_celu_p", false]], "fp_celu_s\uff08c function\uff09": [[10, "c.fp_celu_s", false]], "fp_clip_p\uff08c function\uff09": [[10, "c.fp_clip_p", false]], "fp_clip_s\uff08c function\uff09": [[10, "c.fp_clip_s", false]], "fp_conv2d_p\uff08c function\uff09": [[20, "c.fp_conv2d_p", false]], "fp_conv2d_s\uff08c function\uff09": [[20, "c.fp_conv2d_s", false]], "fp_conv2dbackpropfilterfusion_p\uff08c function\uff09": [[22, "c.fp_conv2dbackpropfilterfusion_p", false]], "fp_conv2dbackpropfilterfusion_s\uff08c function\uff09": [[22, "c.fp_conv2dbackpropfilterfusion_s", false]], "fp_conv2dbackpropinputfusion_p\uff08c function\uff09": [[23, "c.fp_conv2dbackpropinputfusion_p", false]], "fp_conv2dbackpropinputfusion_s\uff08c function\uff09": [[23, "c.fp_conv2dbackpropinputfusion_s", false]], "fp_convtranspose_p\uff08c function\uff09": [[21, "c.fp_convtranspose_p", false]], "fp_convtranspose_s\uff08c function\uff09": [[21, "c.fp_convtranspose_s", false]], "fp_crop_and_resize_anycore\uff08c function\uff09": [[25, "c.fp_crop_and_resize_anycore", false]], "fp_depthtospace_p\uff08c function\uff09": [[26, "c.fp_depthtospace_p", false]], "fp_depthtospace_s\uff08c function\uff09": [[26, "c.fp_depthtospace_s", false]], "fp_eltwise_p\uff08c function\uff09": [[28, "c.fp_eltwise_p", false]], "fp_eltwise_s\uff08c function\uff09": [[28, "c.fp_eltwise_s", false]], "fp_elu_p\uff08c function\uff09": [[10, "c.fp_elu_p", false]], "fp_elu_s\uff08c function\uff09": [[10, "c.fp_elu_s", false]], "fp_embeddinglookup_p\uff08c function\uff09": [[29, "c.fp_embeddinglookup_p", false]], "fp_embeddinglookup_s\uff08c function\uff09": [[29, "c.fp_embeddinglookup_s", false]], "fp_equal_p\uff08c function\uff09": [[30, "c.fp_equal_p", false]], "fp_equal_s\uff08c function\uff09": [[30, "c.fp_equal_s", false]], "fp_expfusion_p\uff08c function\uff09": [[32, "c.fp_expfusion_p", false]], "fp_expfusion_s\uff08c function\uff09": [[32, "c.fp_expfusion_s", false]], "fp_floor_p\uff08c function\uff09": [[34, "c.fp_floor_p", false]], "fp_floor_s\uff08c function\uff09": [[34, "c.fp_floor_s", false]], "fp_floordiv_p\uff08c function\uff09": [[35, "c.fp_floordiv_p", false]], "fp_floordiv_s\uff08c function\uff09": [[35, "c.fp_floordiv_s", false]], "fp_fusedbatchnorm_p\uff08c function\uff09": [[36, "c.fp_fusedbatchnorm_p", false]], "fp_fusedbatchnorm_s\uff08c function\uff09": [[36, "c.fp_fusedbatchnorm_s", false]], "fp_gelu_p\uff08c function\uff09": [[10, "c.fp_gelu_p", false]], "fp_gelu_s\uff08c function\uff09": [[10, "c.fp_gelu_s", false]], "fp_groupnormfusion_p\uff08c function\uff09": [[37, "c.fp_groupnormfusion_p", false]], "fp_groupnormfusion_s\uff08c function\uff09": [[37, "c.fp_groupnormfusion_s", false]], "fp_gru_p\uff08c function\uff09": [[38, "c.fp_Gru_p", false]], "fp_gru_s\uff08c function\uff09": [[38, "c.fp_Gru_s", false]], "fp_hardshrink_p\uff08c function\uff09": [[10, "c.fp_hardshrink_p", false]], "fp_hardshrink_s\uff08c function\uff09": [[10, "c.fp_hardshrink_s", false]], "fp_hardtanh_p\uff08c function\uff09": [[10, "c.fp_hardtanh_p", false]], "fp_hardtanh_s\uff08c function\uff09": [[10, "c.fp_hardtanh_s", false]], "fp_hsigmoid_p\uff08c function\uff09": [[10, "c.fp_hsigmoid_p", false]], "fp_hsigmoid_s\uff08c function\uff09": [[10, "c.fp_hsigmoid_s", false]], "fp_hswish_p\uff08c function\uff09": [[10, "c.fp_hswish_p", false]], "fp_hswish_s\uff08c function\uff09": [[10, "c.fp_hswish_s", false]], "fp_leaky_relu_p\uff08c function\uff09": [[39, "c.fp_leaky_relu_p", false]], "fp_leaky_relu_s\uff08c function\uff09": [[39, "c.fp_leaky_relu_s", false]], "fp_linspace_p\uff08c function\uff09": [[40, "c.fp_linspace_p", false]], "fp_linspace_s\uff08c function\uff09": [[40, "c.fp_linspace_s", false]], "fp_lrelu_p\uff08c function\uff09": [[10, "c.fp_lrelu_p", false]], "fp_lrelu_s\uff08c function\uff09": [[10, "c.fp_lrelu_s", false]], "fp_lstm_p\uff08c function\uff09": [[41, "c.fp_Lstm_p", false]], "fp_lstm_s\uff08c function\uff09": [[41, "c.fp_Lstm_s", false]], "fp_matmulfusion_p\uff08c function\uff09": [[42, "c.fp_matmulfusion_p", false]], "fp_matmulfusion_s\uff08c function\uff09": [[42, "c.fp_matmulfusion_s", false]], "fp_raggedrange_p\uff08c function\uff09": [[43, "c.fp_raggedrange_p", false]], "fp_raggedrange_s\uff08c function\uff09": [[43, "c.fp_raggedrange_s", false]], "fp_range_p\uff08c function\uff09": [[44, "c.fp_range_p", false]], "fp_range_s\uff08c function\uff09": [[44, "c.fp_range_s", false]], "fp_reduce_p\uff08c function\uff09": [[45, "c.fp_reduce_p", false]], "fp_reduce_s\uff08c function\uff09": [[45, "c.fp_reduce_s", false]], "fp_relu6_p\uff08c function\uff09": [[10, "c.fp_relu6_p", false]], "fp_relu6_s\uff08c function\uff09": [[10, "c.fp_relu6_s", false]], "fp_relu_p\uff08c function\uff09": [[10, "c.fp_relu_p", false]], "fp_relu_s\uff08c function\uff09": [[10, "c.fp_relu_s", false]], "fp_resize_anycore\uff08c function\uff09": [[46, "c.fp_resize_anycore", false]], "fp_scalefusion_p\uff08c function\uff09": [[49, "c.fp_scalefusion_p", false]], "fp_scalefusion_s\uff08c function\uff09": [[49, "c.fp_scalefusion_s", false]], "fp_scatter_elements_p\uff08c function\uff09": [[50, "c.fp_scatter_elements_p", false]], "fp_scatter_elements_s\uff08c function\uff09": [[50, "c.fp_scatter_elements_s", false]], "fp_sgd_p\uff08c function\uff09": [[51, "c.fp_sgd_p", false]], "fp_sgd_s\uff08c function\uff09": [[51, "c.fp_sgd_s", false]], "fp_sigmoid_p\uff08c function\uff09": [[10, "c.fp_sigmoid_p", false]], "fp_sigmoid_s\uff08c function\uff09": [[10, "c.fp_sigmoid_s", false]], "fp_softplus_p\uff08c function\uff09": [[10, "c.fp_softplus_p", false]], "fp_softplus_s\uff08c function\uff09": [[10, "c.fp_softplus_s", false]], "fp_softshrink_p\uff08c function\uff09": [[10, "c.fp_softshrink_p", false]], "fp_softshrink_s\uff08c function\uff09": [[10, "c.fp_softshrink_s", false]], "fp_softsignopt_p\uff08c function\uff09": [[10, "c.fp_softsignopt_p", false]], "fp_softsignopt_s\uff08c function\uff09": [[10, "c.fp_softsignopt_s", false]], "fp_spacetobatch_p\uff08c function\uff09": [[52, "c.fp_spacetobatch_p", false]], "fp_spacetobatch_s\uff08c function\uff09": [[52, "c.fp_spacetobatch_s", false]], "fp_spacetobatchnd_p\uff08c function\uff09": [[53, "c.fp_spacetobatchnd_p", false]], "fp_spacetobatchnd_s\uff08c function\uff09": [[53, "c.fp_spacetobatchnd_s", false]], "fp_spacetodepth_p\uff08c function\uff09": [[54, "c.fp_spacetodepth_p", false]], "fp_spacetodepth_s\uff08c function\uff09": [[54, "c.fp_spacetodepth_s", false]], "fp_swish_p\uff08c function\uff09": [[10, "c.fp_swish_p", false]], "fp_swish_s\uff08c function\uff09": [[10, "c.fp_swish_s", false]], "fp_tanh_p\uff08c function\uff09": [[10, "c.fp_tanh_p", false]], "fp_tanh_s\uff08c function\uff09": [[10, "c.fp_tanh_s", false]], "hp_adamweightdecay_p\uff08c function\uff09": [[11, "c.hp_adamweightdecay_p", false]], "hp_adamweightdecay_s\uff08c function\uff09": [[11, "c.hp_adamweightdecay_s", false]], "hp_adder_p\uff08c function\uff09": [[12, "c.hp_adder_p", false]], "hp_adder_s\uff08c function\uff09": [[12, "c.hp_adder_s", false]], "hp_applymomentum_p\uff08c function\uff09": [[13, "c.hp_applymomentum_p", false]], "hp_applymomentum_s\uff08c function\uff09": [[13, "c.hp_applymomentum_s", false]], "hp_avgpoolinggrad_p\uff08c function\uff09": [[16, "c.hp_avgpoolinggrad_p", false]], "hp_avgpoolinggrad_s\uff08c function\uff09": [[16, "c.hp_avgpoolinggrad_s", false]], "hp_batchtospace_p\uff08c function\uff09": [[17, "c.hp_batchtospace_p", false]], "hp_batchtospace_s\uff08c function\uff09": [[17, "c.hp_batchtospace_s", false]], "hp_batchtospacend_p\uff08c function\uff09": [[18, "c.hp_batchtospacend_p", false]], "hp_batchtospacend_s\uff08c function\uff09": [[18, "c.hp_batchtospacend_s", false]], "hp_broadcastto_p\uff08c function\uff09": [[19, "c.hp_broadcastto_p", false]], "hp_broadcastto_s\uff08c function\uff09": [[19, "c.hp_broadcastto_s", false]], "hp_celu_p\uff08c function\uff09": [[10, "c.hp_celu_p", false]], "hp_celu_s\uff08c function\uff09": [[10, "c.hp_celu_s", false]], "hp_clip_p\uff08c function\uff09": [[10, "c.hp_clip_p", false]], "hp_clip_s\uff08c function\uff09": [[10, "c.hp_clip_s", false]], "hp_conv2d_p\uff08c function\uff09": [[20, "c.hp_conv2d_p", false]], "hp_conv2d_s\uff08c function\uff09": [[20, "c.hp_conv2d_s", false]], "hp_conv2dbackpropfilterfusion_p\uff08c function\uff09": [[22, "c.hp_conv2dbackpropfilterfusion_p", false]], "hp_conv2dbackpropfilterfusion_s\uff08c function\uff09": [[22, "c.hp_conv2dbackpropfilterfusion_s", false]], "hp_conv2dbackpropinputfusion_p\uff08c function\uff09": [[23, "c.hp_conv2dbackpropinputfusion_p", false]], "hp_conv2dbackpropinputfusion_s\uff08c function\uff09": [[23, "c.hp_conv2dbackpropinputfusion_s", false]], "hp_convtranspose_p\uff08c function\uff09": [[21, "c.hp_convtranspose_p", false]], "hp_convtranspose_s\uff08c function\uff09": [[21, "c.hp_convtranspose_s", false]], "hp_crop_and_resize_anycore\uff08c function\uff09": [[25, "c.hp_crop_and_resize_anycore", false]], "hp_depthtospace_p\uff08c function\uff09": [[26, "c.hp_depthtospace_p", false]], "hp_depthtospace_s\uff08c function\uff09": [[26, "c.hp_depthtospace_s", false]], "hp_eltwise_p\uff08c function\uff09": [[28, "c.hp_eltwise_p", false]], "hp_eltwise_s\uff08c function\uff09": [[28, "c.hp_eltwise_s", false]], "hp_elu_p\uff08c function\uff09": [[10, "c.hp_elu_p", false]], "hp_elu_s\uff08c function\uff09": [[10, "c.hp_elu_s", false]], "hp_embeddinglookup_p\uff08c function\uff09": [[29, "c.hp_embeddinglookup_p", false]], "hp_embeddinglookup_s\uff08c function\uff09": [[29, "c.hp_embeddinglookup_s", false]], "hp_equal_p\uff08c function\uff09": [[30, "c.hp_equal_p", false]], "hp_equal_s\uff08c function\uff09": [[30, "c.hp_equal_s", false]], "hp_expfusion_p\uff08c function\uff09": [[32, "c.hp_expfusion_p", false]], "hp_expfusion_s\uff08c function\uff09": [[32, "c.hp_expfusion_s", false]], "hp_floor_p\uff08c function\uff09": [[34, "c.hp_floor_p", false]], "hp_floor_s\uff08c function\uff09": [[34, "c.hp_floor_s", false]], "hp_floordiv_p\uff08c function\uff09": [[35, "c.hp_floordiv_p", false]], "hp_floordiv_s\uff08c function\uff09": [[35, "c.hp_floordiv_s", false]], "hp_fusedbatchnorm_p\uff08c function\uff09": [[36, "c.hp_fusedbatchnorm_p", false]], "hp_fusedbatchnorm_s\uff08c function\uff09": [[36, "c.hp_fusedbatchnorm_s", false]], "hp_gelu_p\uff08c function\uff09": [[10, "c.hp_gelu_p", false]], "hp_gelu_s\uff08c function\uff09": [[10, "c.hp_gelu_s", false]], "hp_groupnormfusion_p\uff08c function\uff09": [[37, "c.hp_groupnormfusion_p", false]], "hp_groupnormfusion_s\uff08c function\uff09": [[37, "c.hp_groupnormfusion_s", false]], "hp_gru_p\uff08c function\uff09": [[38, "c.hp_Gru_p", false]], "hp_gru_s\uff08c function\uff09": [[38, "c.hp_Gru_s", false]], "hp_hardshrink_p\uff08c function\uff09": [[10, "c.hp_hardshrink_p", false]], "hp_hardshrink_s\uff08c function\uff09": [[10, "c.hp_hardshrink_s", false]], "hp_hardtanh_p\uff08c function\uff09": [[10, "c.hp_hardtanh_p", false]], "hp_hardtanh_s\uff08c function\uff09": [[10, "c.hp_hardtanh_s", false]], "hp_hsigmoid_p\uff08c function\uff09": [[10, "c.hp_hsigmoid_p", false]], "hp_hsigmoid_s\uff08c function\uff09": [[10, "c.hp_hsigmoid_s", false]], "hp_hswish_p\uff08c function\uff09": [[10, "c.hp_hswish_p", false]], "hp_hswish_s\uff08c function\uff09": [[10, "c.hp_hswish_s", false]], "hp_leaky_relu_p\uff08c function\uff09": [[39, "c.hp_leaky_relu_p", false]], "hp_leaky_relu_s\uff08c function\uff09": [[39, "c.hp_leaky_relu_s", false]], "hp_lrelu_p\uff08c function\uff09": [[10, "c.hp_lrelu_p", false]], "hp_lrelu_s\uff08c function\uff09": [[10, "c.hp_lrelu_s", false]], "hp_reduce_p\uff08c function\uff09": [[45, "c.hp_reduce_p", false]], "hp_reduce_s\uff08c function\uff09": [[45, "c.hp_reduce_s", false]], "hp_relu6_p\uff08c function\uff09": [[10, "c.hp_relu6_p", false]], "hp_relu6_s\uff08c function\uff09": [[10, "c.hp_relu6_s", false]], "hp_relu_p\uff08c function\uff09": [[10, "c.hp_relu_p", false]], "hp_relu_s\uff08c function\uff09": [[10, "c.hp_relu_s", false]], "hp_resize_anycore\uff08c function\uff09": [[46, "c.hp_resize_anycore", false]], "hp_scalefusion_p\uff08c function\uff09": [[49, "c.hp_scalefusion_p", false]], "hp_scalefusion_s\uff08c function\uff09": [[49, "c.hp_scalefusion_s", false]], "hp_scatter_elements_p\uff08c function\uff09": [[50, "c.hp_scatter_elements_p", false]], "hp_scatter_elements_s\uff08c function\uff09": [[50, "c.hp_scatter_elements_s", false]], "hp_sgd_p\uff08c function\uff09": [[51, "c.hp_sgd_p", false]], "hp_sgd_s\uff08c function\uff09": [[51, "c.hp_sgd_s", false]], "hp_sigmoid_p\uff08c function\uff09": [[10, "c.hp_sigmoid_p", false]], "hp_sigmoid_s\uff08c function\uff09": [[10, "c.hp_sigmoid_s", false]], "hp_softplus_p\uff08c function\uff09": [[10, "c.hp_softplus_p", false]], "hp_softplus_s\uff08c function\uff09": [[10, "c.hp_softplus_s", false]], "hp_softshrink_p\uff08c function\uff09": [[10, "c.hp_softshrink_p", false]], "hp_softshrink_s\uff08c function\uff09": [[10, "c.hp_softshrink_s", false]], "hp_softsignopt_p\uff08c function\uff09": [[10, "c.hp_softsignopt_p", false]], "hp_softsignopt_s\uff08c function\uff09": [[10, "c.hp_softsignopt_s", false]], "hp_spacetobatch_p\uff08c function\uff09": [[52, "c.hp_spacetobatch_p", false]], "hp_spacetobatch_s\uff08c function\uff09": [[52, "c.hp_spacetobatch_s", false]], "hp_spacetobatchnd_p\uff08c function\uff09": [[53, "c.hp_spacetobatchnd_p", false]], "hp_spacetobatchnd_s\uff08c function\uff09": [[53, "c.hp_spacetobatchnd_s", false]], "hp_spacetodepth_p\uff08c function\uff09": [[54, "c.hp_spacetodepth_p", false]], "hp_spacetodepth_s\uff08c function\uff09": [[54, "c.hp_spacetodepth_s", false]], "hp_swish_p\uff08c function\uff09": [[10, "c.hp_swish_p", false]], "hp_swish_s\uff08c function\uff09": [[10, "c.hp_swish_s", false]], "hp_tanh_p\uff08c function\uff09": [[10, "c.hp_tanh_p", false]], "hp_tanh_s\uff08c function\uff09": [[10, "c.hp_tanh_s", false]], "i16_batchtospace_p\uff08c function\uff09": [[17, "c.i16_batchtospace_p", false]], "i16_batchtospace_s\uff08c function\uff09": [[17, "c.i16_batchtospace_s", false]], "i16_batchtospacend_p\uff08c function\uff09": [[18, "c.i16_batchtospacend_p", false]], "i16_batchtospacend_s\uff08c function\uff09": [[18, "c.i16_batchtospacend_s", false]], "i16_broadcastto_p\uff08c function\uff09": [[19, "c.i16_broadcastto_p", false]], "i16_broadcastto_s\uff08c function\uff09": [[19, "c.i16_broadcastto_s", false]], "i16_depthtospace_p\uff08c function\uff09": [[26, "c.i16_depthtospace_p", false]], "i16_depthtospace_s\uff08c function\uff09": [[26, "c.i16_depthtospace_s", false]], "i16_eltwise_p\uff08c function\uff09": [[28, "c.i16_eltwise_p", false]], "i16_eltwise_s\uff08c function\uff09": [[28, "c.i16_eltwise_s", false]], "i16_equal_p\uff08c function\uff09": [[30, "c.i16_equal_p", false]], "i16_equal_s\uff08c function\uff09": [[30, "c.i16_equal_s", false]], "i16_expfusion_p\uff08c function\uff09": [[32, "c.i16_expfusion_p", false]], "i16_expfusion_s\uff08c function\uff09": [[32, "c.i16_expfusion_s", false]], "i16_matmulfusion_p\uff08c function\uff09": [[42, "c.i16_matmulfusion_p", false]], "i16_matmulfusion_s\uff08c function\uff09": [[42, "c.i16_matmulfusion_s", false]], "i16_raggedrange_p\uff08c function\uff09": [[43, "c.i16_raggedrange_p", false]], "i16_raggedrange_s\uff08c function\uff09": [[43, "c.i16_raggedrange_s", false]], "i16_range_p\uff08c function\uff09": [[44, "c.i16_range_p", false]], "i16_range_s\uff08c function\uff09": [[44, "c.i16_range_s", false]], "i16_reduce_p\uff08c function\uff09": [[45, "c.i16_reduce_p", false]], "i16_reduce_s\uff08c function\uff09": [[45, "c.i16_reduce_s", false]], "i16_scalefusion_p\uff08c function\uff09": [[49, "c.i16_scalefusion_p", false]], "i16_scalefusion_s\uff08c function\uff09": [[49, "c.i16_scalefusion_s", false]], "i16_scatter_elements_p\uff08c function\uff09": [[50, "c.i16_scatter_elements_p", false]], "i16_scatter_elements_s\uff08c function\uff09": [[50, "c.i16_scatter_elements_s", false]], "i16_spacetobatch_p\uff08c function\uff09": [[52, "c.i16_spacetobatch_p", false]], "i16_spacetobatch_s\uff08c function\uff09": [[52, "c.i16_spacetobatch_s", false]], "i16_spacetobatchnd_p\uff08c function\uff09": [[53, "c.i16_spacetobatchnd_p", false]], "i16_spacetobatchnd_s\uff08c function\uff09": [[53, "c.i16_spacetobatchnd_s", false]], "i16_spacetodepth_p\uff08c function\uff09": [[54, "c.i16_spacetodepth_p", false]], "i16_spacetodepth_s\uff08c function\uff09": [[54, "c.i16_spacetodepth_s", false]], "i32_batchtospace_p\uff08c function\uff09": [[17, "c.i32_batchtospace_p", false]], "i32_batchtospace_s\uff08c function\uff09": [[17, "c.i32_batchtospace_s", false]], "i32_batchtospacend_p\uff08c function\uff09": [[18, "c.i32_batchtospacend_p", false]], "i32_batchtospacend_s\uff08c function\uff09": [[18, "c.i32_batchtospacend_s", false]], "i32_broadcastto_p\uff08c function\uff09": [[19, "c.i32_broadcastto_p", false]], "i32_broadcastto_s\uff08c function\uff09": [[19, "c.i32_broadcastto_s", false]], "i32_depthtospace_p\uff08c function\uff09": [[26, "c.i32_depthtospace_p", false]], "i32_depthtospace_s\uff08c function\uff09": [[26, "c.i32_depthtospace_s", false]], "i32_eltwise_p\uff08c function\uff09": [[28, "c.i32_eltwise_p", false]], "i32_eltwise_s\uff08c function\uff09": [[28, "c.i32_eltwise_s", false]], "i32_equal_p\uff08c function\uff09": [[30, "c.i32_equal_p", false]], "i32_equal_s\uff08c function\uff09": [[30, "c.i32_equal_s", false]], "i32_expfusion_p\uff08c function\uff09": [[32, "c.i32_expfusion_p", false]], "i32_expfusion_s\uff08c function\uff09": [[32, "c.i32_expfusion_s", false]], "i32_matmulfusion_p\uff08c function\uff09": [[42, "c.i32_matmulfusion_p", false]], "i32_matmulfusion_s\uff08c function\uff09": [[42, "c.i32_matmulfusion_s", false]], "i32_raggedrange_p\uff08c function\uff09": [[43, "c.i32_raggedrange_p", false]], "i32_raggedrange_s\uff08c function\uff09": [[43, "c.i32_raggedrange_s", false]], "i32_range_p\uff08c function\uff09": [[44, "c.i32_range_p", false]], "i32_range_s\uff08c function\uff09": [[44, "c.i32_range_s", false]], "i32_reduce_p\uff08c function\uff09": [[45, "c.i32_reduce_p", false]], "i32_reduce_s\uff08c function\uff09": [[45, "c.i32_reduce_s", false]], "i32_scalefusion_p\uff08c function\uff09": [[49, "c.i32_scalefusion_p", false]], "i32_scalefusion_s\uff08c function\uff09": [[49, "c.i32_scalefusion_s", false]], "i32_scatter_elements_p\uff08c function\uff09": [[50, "c.i32_scatter_elements_p", false]], "i32_scatter_elements_s\uff08c function\uff09": [[50, "c.i32_scatter_elements_s", false]], "i32_spacetobatch_p\uff08c function\uff09": [[52, "c.i32_spacetobatch_p", false]], "i32_spacetobatch_s\uff08c function\uff09": [[52, "c.i32_spacetobatch_s", false]], "i32_spacetobatchnd_p\uff08c function\uff09": [[53, "c.i32_spacetobatchnd_p", false]], "i32_spacetobatchnd_s\uff08c function\uff09": [[53, "c.i32_spacetobatchnd_s", false]], "i32_spacetodepth_p\uff08c function\uff09": [[54, "c.i32_spacetodepth_p", false]], "i32_spacetodepth_s\uff08c function\uff09": [[54, "c.i32_spacetodepth_s", false]], "i8_adder_p\uff08c function\uff09": [[12, "c.i8_adder_p", false]], "i8_adder_s\uff08c function\uff09": [[12, "c.i8_adder_s", false]], "i8_batchtospace_p\uff08c function\uff09": [[17, "c.i8_batchtospace_p", false]], "i8_batchtospace_s\uff08c function\uff09": [[17, "c.i8_batchtospace_s", false]], "i8_batchtospacend_p\uff08c function\uff09": [[18, "c.i8_batchtospacend_p", false]], "i8_batchtospacend_s\uff08c function\uff09": [[18, "c.i8_batchtospacend_s", false]], "i8_broadcastto_p\uff08c function\uff09": [[19, "c.i8_broadcastto_p", false]], "i8_broadcastto_s\uff08c function\uff09": [[19, "c.i8_broadcastto_s", false]], "i8_celu_p\uff08c function\uff09": [[10, "c.i8_celu_p", false]], "i8_celu_s\uff08c function\uff09": [[10, "c.i8_celu_s", false]], "i8_clip_p\uff08c function\uff09": [[10, "c.i8_clip_p", false]], "i8_clip_s\uff08c function\uff09": [[10, "c.i8_clip_s", false]], "i8_conv2d_p\uff08c function\uff09": [[20, "c.i8_conv2d_p", false]], "i8_conv2d_s\uff08c function\uff09": [[20, "c.i8_conv2d_s", false]], "i8_convtranspose_p\uff08c function\uff09": [[21, "c.i8_convtranspose_p", false]], "i8_convtranspose_s\uff08c function\uff09": [[21, "c.i8_convtranspose_s", false]], "i8_crop_and_resize_anycore\uff08c function\uff09": [[25, "c.i8_crop_and_resize_anycore", false]], "i8_depthtospace_p\uff08c function\uff09": [[26, "c.i8_depthtospace_p", false]], "i8_depthtospace_s\uff08c function\uff09": [[26, "c.i8_depthtospace_s", false]], "i8_eltwise_p\uff08c function\uff09": [[28, "c.i8_eltwise_p", false]], "i8_eltwise_s\uff08c function\uff09": [[28, "c.i8_eltwise_s", false]], "i8_elu_p\uff08c function\uff09": [[10, "c.i8_elu_p", false]], "i8_elu_s\uff08c function\uff09": [[10, "c.i8_elu_s", false]], "i8_equal_p\uff08c function\uff09": [[30, "c.i8_equal_p", false]], "i8_equal_s\uff08c function\uff09": [[30, "c.i8_equal_s", false]], "i8_expfusion_p\uff08c function\uff09": [[32, "c.i8_expfusion_p", false]], "i8_expfusion_s\uff08c function\uff09": [[32, "c.i8_expfusion_s", false]], "i8_gelu_p\uff08c function\uff09": [[10, "c.i8_gelu_p", false]], "i8_gelu_s\uff08c function\uff09": [[10, "c.i8_gelu_s", false]], "i8_gru_p\uff08c function\uff09": [[38, "c.i8_Gru_p", false]], "i8_gru_s\uff08c function\uff09": [[38, "c.i8_Gru_s", false]], "i8_hardshrink_p\uff08c function\uff09": [[10, "c.i8_hardshrink_p", false]], "i8_hardshrink_s\uff08c function\uff09": [[10, "c.i8_hardshrink_s", false]], "i8_hardtanh_p\uff08c function\uff09": [[10, "c.i8_hardtanh_p", false]], "i8_hardtanh_s\uff08c function\uff09": [[10, "c.i8_hardtanh_s", false]], "i8_hsigmoid_p\uff08c function\uff09": [[10, "c.i8_hsigmoid_p", false]], "i8_hsigmoid_s\uff08c function\uff09": [[10, "c.i8_hsigmoid_s", false]], "i8_hswish_p\uff08c function\uff09": [[10, "c.i8_hswish_p", false]], "i8_hswish_s\uff08c function\uff09": [[10, "c.i8_hswish_s", false]], "i8_leaky_relu_p\uff08c function\uff09": [[39, "c.i8_leaky_relu_p", false]], "i8_leaky_relu_s\uff08c function\uff09": [[39, "c.i8_leaky_relu_s", false]], "i8_lrelu_p\uff08c function\uff09": [[10, "c.i8_lrelu_p", false]], "i8_lrelu_s\uff08c function\uff09": [[10, "c.i8_lrelu_s", false]], "i8_raggedrange_p\uff08c function\uff09": [[43, "c.i8_raggedrange_p", false]], "i8_raggedrange_s\uff08c function\uff09": [[43, "c.i8_raggedrange_s", false]], "i8_range_p\uff08c function\uff09": [[44, "c.i8_range_p", false]], "i8_range_s\uff08c function\uff09": [[44, "c.i8_range_s", false]], "i8_reduce_p\uff08c function\uff09": [[45, "c.i8_reduce_p", false]], "i8_reduce_s\uff08c function\uff09": [[45, "c.i8_reduce_s", false]], "i8_relu6_p\uff08c function\uff09": [[10, "c.i8_relu6_p", false]], "i8_relu6_s\uff08c function\uff09": [[10, "c.i8_relu6_s", false]], "i8_relu_p\uff08c function\uff09": [[10, "c.i8_relu_p", false]], "i8_relu_s\uff08c function\uff09": [[10, "c.i8_relu_s", false]], "i8_resize_anycore\uff08c function\uff09": [[46, "c.i8_resize_anycore", false]], "i8_scalefusion_p\uff08c function\uff09": [[49, "c.i8_scalefusion_p", false]], "i8_scalefusion_s\uff08c function\uff09": [[49, "c.i8_scalefusion_s", false]], "i8_scatter_elements_p\uff08c function\uff09": [[50, "c.i8_scatter_elements_p", false]], "i8_scatter_elements_s\uff08c function\uff09": [[50, "c.i8_scatter_elements_s", false]], "i8_sigmoid_p\uff08c function\uff09": [[10, "c.i8_sigmoid_p", false]], "i8_sigmoid_s\uff08c function\uff09": [[10, "c.i8_sigmoid_s", false]], "i8_softplus_p\uff08c function\uff09": [[10, "c.i8_softplus_p", false]], "i8_softplus_s\uff08c function\uff09": [[10, "c.i8_softplus_s", false]], "i8_softshrink_p\uff08c function\uff09": [[10, "c.i8_softshrink_p", false]], "i8_softshrink_s\uff08c function\uff09": [[10, "c.i8_softshrink_s", false]], "i8_softsignopt_p\uff08c function\uff09": [[10, "c.i8_softsignopt_p", false]], "i8_softsignopt_s\uff08c function\uff09": [[10, "c.i8_softsignopt_s", false]], "i8_spacetobatch_p\uff08c function\uff09": [[52, "c.i8_spacetobatch_p", false]], "i8_spacetobatch_s\uff08c function\uff09": [[52, "c.i8_spacetobatch_s", false]], "i8_spacetobatchnd_p\uff08c function\uff09": [[53, "c.i8_spacetobatchnd_p", false]], "i8_spacetobatchnd_s\uff08c function\uff09": [[53, "c.i8_spacetobatchnd_s", false]], "i8_spacetodepth_p\uff08c function\uff09": [[54, "c.i8_spacetodepth_p", false]], "i8_spacetodepth_s\uff08c function\uff09": [[54, "c.i8_spacetodepth_s", false]], "i8_swish_p\uff08c function\uff09": [[10, "c.i8_swish_p", false]], "i8_swish_s\uff08c function\uff09": [[10, "c.i8_swish_s", false]], "i8_tanh_p\uff08c function\uff09": [[10, "c.i8_tanh_p", false]], "i8_tanh_s\uff08c function\uff09": [[10, "c.i8_tanh_s", false]]}, "objects": {"": [[24, 0, 1, "c.anytype_crop_anycore", "anytype_crop_anycore"], [31, 0, 1, "c.anytype_expand_dims_anycore", "anytype_expand_dims_anycore"], [33, 0, 1, "c.anytype_fillv2_p", "anytype_fillv2_p"], [33, 0, 1, "c.anytype_fillv2_s", "anytype_fillv2_s"], [47, 0, 1, "c.anytype_reverse_sequence_anycore", "anytype_reverse_sequence_anycore"], [48, 0, 1, "c.anytype_reversev2_anycore", "anytype_reversev2_anycore"], [55, 0, 1, "c.anytype_squeeze_anycore", "anytype_squeeze_anycore"], [56, 0, 1, "c.anytype_unsqueeze_anycore", "anytype_unsqueeze_anycore"], [14, 0, 1, "c.assert", "assert"], [17, 0, 1, "c.c128_batchtospace_p", "c128_batchtospace_p"], [17, 0, 1, "c.c128_batchtospace_s", "c128_batchtospace_s"], [18, 0, 1, "c.c128_batchtospacend_p", "c128_batchtospacend_p"], [18, 0, 1, "c.c128_batchtospacend_s", "c128_batchtospacend_s"], [19, 0, 1, "c.c128_broadcastto_p", "c128_broadcastto_p"], [19, 0, 1, "c.c128_broadcastto_s", "c128_broadcastto_s"], [26, 0, 1, "c.c128_depthtospace_p", "c128_depthtospace_p"], [26, 0, 1, "c.c128_depthtospace_s", "c128_depthtospace_s"], [28, 0, 1, "c.c128_eltwise_p", "c128_eltwise_p"], [28, 0, 1, "c.c128_eltwise_s", "c128_eltwise_s"], [30, 0, 1, "c.c128_equal_p", "c128_equal_p"], [30, 0, 1, "c.c128_equal_s", "c128_equal_s"], [32, 0, 1, "c.c128_expfusion_p", "c128_expfusion_p"], [32, 0, 1, "c.c128_expfusion_s", "c128_expfusion_s"], [42, 0, 1, "c.c128_matmulfusion_p", "c128_matmulfusion_p"], [42, 0, 1, "c.c128_matmulfusion_s", "c128_matmulfusion_s"], [50, 0, 1, "c.c128_scatter_elements_p", "c128_scatter_elements_p"], [50, 0, 1, "c.c128_scatter_elements_s", "c128_scatter_elements_s"], [52, 0, 1, "c.c128_spacetobatch_p", "c128_spacetobatch_p"], [52, 0, 1, "c.c128_spacetobatch_s", "c128_spacetobatch_s"], [53, 0, 1, "c.c128_spacetobatchnd_p", "c128_spacetobatchnd_p"], [53, 0, 1, "c.c128_spacetobatchnd_s", "c128_spacetobatchnd_s"], [54, 0, 1, "c.c128_spacetodepth_p", "c128_spacetodepth_p"], [54, 0, 1, "c.c128_spacetodepth_s", "c128_spacetodepth_s"], [17, 0, 1, "c.c64_batchtospace_p", "c64_batchtospace_p"], [17, 0, 1, "c.c64_batchtospace_s", "c64_batchtospace_s"], [18, 0, 1, "c.c64_batchtospacend_p", "c64_batchtospacend_p"], [18, 0, 1, "c.c64_batchtospacend_s", "c64_batchtospacend_s"], [19, 0, 1, "c.c64_broadcastto_p", "c64_broadcastto_p"], [19, 0, 1, "c.c64_broadcastto_s", "c64_broadcastto_s"], [26, 0, 1, "c.c64_depthtospace_p", "c64_depthtospace_p"], [26, 0, 1, "c.c64_depthtospace_s", "c64_depthtospace_s"], [28, 0, 1, "c.c64_eltwise_p", "c64_eltwise_p"], [28, 0, 1, "c.c64_eltwise_s", "c64_eltwise_s"], [30, 0, 1, "c.c64_equal_p", "c64_equal_p"], [30, 0, 1, "c.c64_equal_s", "c64_equal_s"], [32, 0, 1, "c.c64_expfusion_p", "c64_expfusion_p"], [32, 0, 1, "c.c64_expfusion_s", "c64_expfusion_s"], [42, 0, 1, "c.c64_matmulfusion_p", "c64_matmulfusion_p"], [42, 0, 1, "c.c64_matmulfusion_s", "c64_matmulfusion_s"], [50, 0, 1, "c.c64_scatter_elements_p", "c64_scatter_elements_p"], [50, 0, 1, "c.c64_scatter_elements_s", "c64_scatter_elements_s"], [52, 0, 1, "c.c64_spacetobatch_p", "c64_spacetobatch_p"], [52, 0, 1, "c.c64_spacetobatch_s", "c64_spacetobatch_s"], [53, 0, 1, "c.c64_spacetobatchnd_p", "c64_spacetobatchnd_p"], [53, 0, 1, "c.c64_spacetobatchnd_s", "c64_spacetobatchnd_s"], [54, 0, 1, "c.c64_spacetodepth_p", "c64_spacetodepth_p"], [54, 0, 1, "c.c64_spacetodepth_s", "c64_spacetodepth_s"], [17, 0, 1, "c.dp_batchtospace_p", "dp_batchtospace_p"], [17, 0, 1, "c.dp_batchtospace_s", "dp_batchtospace_s"], [18, 0, 1, "c.dp_batchtospacend_p", "dp_batchtospacend_p"], [18, 0, 1, "c.dp_batchtospacend_s", "dp_batchtospacend_s"], [19, 0, 1, "c.dp_broadcastto_p", "dp_broadcastto_p"], [19, 0, 1, "c.dp_broadcastto_s", "dp_broadcastto_s"], [26, 0, 1, "c.dp_depthtospace_p", "dp_depthtospace_p"], [26, 0, 1, "c.dp_depthtospace_s", "dp_depthtospace_s"], [28, 0, 1, "c.dp_eltwise_p", "dp_eltwise_p"], [28, 0, 1, "c.dp_eltwise_s", "dp_eltwise_s"], [30, 0, 1, "c.dp_equal_p", "dp_equal_p"], [30, 0, 1, "c.dp_equal_s", "dp_equal_s"], [32, 0, 1, "c.dp_expfusion_p", "dp_expfusion_p"], [32, 0, 1, "c.dp_expfusion_s", "dp_expfusion_s"], [34, 0, 1, "c.dp_floor_p", "dp_floor_p"], [34, 0, 1, "c.dp_floor_s", "dp_floor_s"], [35, 0, 1, "c.dp_floordiv_p", "dp_floordiv_p"], [35, 0, 1, "c.dp_floordiv_s", "dp_floordiv_s"], [42, 0, 1, "c.dp_matmulfusion_p", "dp_matmulfusion_p"], [42, 0, 1, "c.dp_matmulfusion_s", "dp_matmulfusion_s"], [43, 0, 1, "c.dp_raggedrange_p", "dp_raggedrange_p"], [43, 0, 1, "c.dp_raggedrange_s", "dp_raggedrange_s"], [44, 0, 1, "c.dp_range_p", "dp_range_p"], [44, 0, 1, "c.dp_range_s", "dp_range_s"], [45, 0, 1, "c.dp_reduce_p", "dp_reduce_p"], [45, 0, 1, "c.dp_reduce_s", "dp_reduce_s"], [49, 0, 1, "c.dp_scalefusion_p", "dp_scalefusion_p"], [49, 0, 1, "c.dp_scalefusion_s", "dp_scalefusion_s"], [50, 0, 1, "c.dp_scatter_elements_p", "dp_scatter_elements_p"], [50, 0, 1, "c.dp_scatter_elements_s", "dp_scatter_elements_s"], [52, 0, 1, "c.dp_spacetobatch_p", "dp_spacetobatch_p"], [52, 0, 1, "c.dp_spacetobatch_s", "dp_spacetobatch_s"], [53, 0, 1, "c.dp_spacetobatchnd_p", "dp_spacetobatchnd_p"], [53, 0, 1, "c.dp_spacetobatchnd_s", "dp_spacetobatchnd_s"], [54, 0, 1, "c.dp_spacetodepth_p", "dp_spacetodepth_p"], [54, 0, 1, "c.dp_spacetodepth_s", "dp_spacetodepth_s"], [38, 0, 1, "c.fp_Gru_p", "fp_Gru_p"], [38, 0, 1, "c.fp_Gru_s", "fp_Gru_s"], [41, 0, 1, "c.fp_Lstm_p", "fp_Lstm_p"], [41, 0, 1, "c.fp_Lstm_s", "fp_Lstm_s"], [11, 0, 1, "c.fp_adamweightdecay_p", "fp_adamweightdecay_p"], [11, 0, 1, "c.fp_adamweightdecay_s", "fp_adamweightdecay_s"], [12, 0, 1, "c.fp_adder_p", "fp_adder_p"], [12, 0, 1, "c.fp_adder_s", "fp_adder_s"], [13, 0, 1, "c.fp_applymomentum_p", "fp_applymomentum_p"], [13, 0, 1, "c.fp_applymomentum_s", "fp_applymomentum_s"], [15, 0, 1, "c.fp_attention_p", "fp_attention_p"], [15, 0, 1, "c.fp_attention_s", "fp_attention_s"], [16, 0, 1, "c.fp_avgpoolinggrad_p", "fp_avgpoolinggrad_p"], [16, 0, 1, "c.fp_avgpoolinggrad_s", "fp_avgpoolinggrad_s"], [17, 0, 1, "c.fp_batchtospace_p", "fp_batchtospace_p"], [17, 0, 1, "c.fp_batchtospace_s", "fp_batchtospace_s"], [18, 0, 1, "c.fp_batchtospacend_p", "fp_batchtospacend_p"], [18, 0, 1, "c.fp_batchtospacend_s", "fp_batchtospacend_s"], [19, 0, 1, "c.fp_broadcastto_p", "fp_broadcastto_p"], [19, 0, 1, "c.fp_broadcastto_s", "fp_broadcastto_s"], [10, 0, 1, "c.fp_celu_p", "fp_celu_p"], [10, 0, 1, "c.fp_celu_s", "fp_celu_s"], [10, 0, 1, "c.fp_clip_p", "fp_clip_p"], [10, 0, 1, "c.fp_clip_s", "fp_clip_s"], [20, 0, 1, "c.fp_conv2d_p", "fp_conv2d_p"], [20, 0, 1, "c.fp_conv2d_s", "fp_conv2d_s"], [22, 0, 1, "c.fp_conv2dbackpropfilterfusion_p", "fp_conv2dbackpropfilterfusion_p"], [22, 0, 1, "c.fp_conv2dbackpropfilterfusion_s", "fp_conv2dbackpropfilterfusion_s"], [23, 0, 1, "c.fp_conv2dbackpropinputfusion_p", "fp_conv2dbackpropinputfusion_p"], [23, 0, 1, "c.fp_conv2dbackpropinputfusion_s", "fp_conv2dbackpropinputfusion_s"], [21, 0, 1, "c.fp_convtranspose_p", "fp_convtranspose_p"], [21, 0, 1, "c.fp_convtranspose_s", "fp_convtranspose_s"], [25, 0, 1, "c.fp_crop_and_resize_anycore", "fp_crop_and_resize_anycore"], [26, 0, 1, "c.fp_depthtospace_p", "fp_depthtospace_p"], [26, 0, 1, "c.fp_depthtospace_s", "fp_depthtospace_s"], [28, 0, 1, "c.fp_eltwise_p", "fp_eltwise_p"], [28, 0, 1, "c.fp_eltwise_s", "fp_eltwise_s"], [10, 0, 1, "c.fp_elu_p", "fp_elu_p"], [10, 0, 1, "c.fp_elu_s", "fp_elu_s"], [29, 0, 1, "c.fp_embeddinglookup_p", "fp_embeddinglookup_p"], [29, 0, 1, "c.fp_embeddinglookup_s", "fp_embeddinglookup_s"], [30, 0, 1, "c.fp_equal_p", "fp_equal_p"], [30, 0, 1, "c.fp_equal_s", "fp_equal_s"], [32, 0, 1, "c.fp_expfusion_p", "fp_expfusion_p"], [32, 0, 1, "c.fp_expfusion_s", "fp_expfusion_s"], [34, 0, 1, "c.fp_floor_p", "fp_floor_p"], [34, 0, 1, "c.fp_floor_s", "fp_floor_s"], [35, 0, 1, "c.fp_floordiv_p", "fp_floordiv_p"], [35, 0, 1, "c.fp_floordiv_s", "fp_floordiv_s"], [36, 0, 1, "c.fp_fusedbatchnorm_p", "fp_fusedbatchnorm_p"], [36, 0, 1, "c.fp_fusedbatchnorm_s", "fp_fusedbatchnorm_s"], [10, 0, 1, "c.fp_gelu_p", "fp_gelu_p"], [10, 0, 1, "c.fp_gelu_s", "fp_gelu_s"], [37, 0, 1, "c.fp_groupnormfusion_p", "fp_groupnormfusion_p"], [37, 0, 1, "c.fp_groupnormfusion_s", "fp_groupnormfusion_s"], [10, 0, 1, "c.fp_hardshrink_p", "fp_hardshrink_p"], [10, 0, 1, "c.fp_hardshrink_s", "fp_hardshrink_s"], [10, 0, 1, "c.fp_hardtanh_p", "fp_hardtanh_p"], [10, 0, 1, "c.fp_hardtanh_s", "fp_hardtanh_s"], [10, 0, 1, "c.fp_hsigmoid_p", "fp_hsigmoid_p"], [10, 0, 1, "c.fp_hsigmoid_s", "fp_hsigmoid_s"], [10, 0, 1, "c.fp_hswish_p", "fp_hswish_p"], [10, 0, 1, "c.fp_hswish_s", "fp_hswish_s"], [39, 0, 1, "c.fp_leaky_relu_p", "fp_leaky_relu_p"], [39, 0, 1, "c.fp_leaky_relu_s", "fp_leaky_relu_s"], [40, 0, 1, "c.fp_linspace_p", "fp_linspace_p"], [40, 0, 1, "c.fp_linspace_s", "fp_linspace_s"], [10, 0, 1, "c.fp_lrelu_p", "fp_lrelu_p"], [10, 0, 1, "c.fp_lrelu_s", "fp_lrelu_s"], [42, 0, 1, "c.fp_matmulfusion_p", "fp_matmulfusion_p"], [42, 0, 1, "c.fp_matmulfusion_s", "fp_matmulfusion_s"], [43, 0, 1, "c.fp_raggedrange_p", "fp_raggedrange_p"], [43, 0, 1, "c.fp_raggedrange_s", "fp_raggedrange_s"], [44, 0, 1, "c.fp_range_p", "fp_range_p"], [44, 0, 1, "c.fp_range_s", "fp_range_s"], [45, 0, 1, "c.fp_reduce_p", "fp_reduce_p"], [45, 0, 1, "c.fp_reduce_s", "fp_reduce_s"], [10, 0, 1, "c.fp_relu6_p", "fp_relu6_p"], [10, 0, 1, "c.fp_relu6_s", "fp_relu6_s"], [10, 0, 1, "c.fp_relu_p", "fp_relu_p"], [10, 0, 1, "c.fp_relu_s", "fp_relu_s"], [46, 0, 1, "c.fp_resize_anycore", "fp_resize_anycore"], [49, 0, 1, "c.fp_scalefusion_p", "fp_scalefusion_p"], [49, 0, 1, "c.fp_scalefusion_s", "fp_scalefusion_s"], [50, 0, 1, "c.fp_scatter_elements_p", "fp_scatter_elements_p"], [50, 0, 1, "c.fp_scatter_elements_s", "fp_scatter_elements_s"], [51, 0, 1, "c.fp_sgd_p", "fp_sgd_p"], [51, 0, 1, "c.fp_sgd_s", "fp_sgd_s"], [10, 0, 1, "c.fp_sigmoid_p", "fp_sigmoid_p"], [10, 0, 1, "c.fp_sigmoid_s", "fp_sigmoid_s"], [10, 0, 1, "c.fp_softplus_p", "fp_softplus_p"], [10, 0, 1, "c.fp_softplus_s", "fp_softplus_s"], [10, 0, 1, "c.fp_softshrink_p", "fp_softshrink_p"], [10, 0, 1, "c.fp_softshrink_s", "fp_softshrink_s"], [10, 0, 1, "c.fp_softsignopt_p", "fp_softsignopt_p"], [10, 0, 1, "c.fp_softsignopt_s", "fp_softsignopt_s"], [52, 0, 1, "c.fp_spacetobatch_p", "fp_spacetobatch_p"], [52, 0, 1, "c.fp_spacetobatch_s", "fp_spacetobatch_s"], [53, 0, 1, "c.fp_spacetobatchnd_p", "fp_spacetobatchnd_p"], [53, 0, 1, "c.fp_spacetobatchnd_s", "fp_spacetobatchnd_s"], [54, 0, 1, "c.fp_spacetodepth_p", "fp_spacetodepth_p"], [54, 0, 1, "c.fp_spacetodepth_s", "fp_spacetodepth_s"], [10, 0, 1, "c.fp_swish_p", "fp_swish_p"], [10, 0, 1, "c.fp_swish_s", "fp_swish_s"], [10, 0, 1, "c.fp_tanh_p", "fp_tanh_p"], [10, 0, 1, "c.fp_tanh_s", "fp_tanh_s"], [38, 0, 1, "c.hp_Gru_p", "hp_Gru_p"], [38, 0, 1, "c.hp_Gru_s", "hp_Gru_s"], [11, 0, 1, "c.hp_adamweightdecay_p", "hp_adamweightdecay_p"], [11, 0, 1, "c.hp_adamweightdecay_s", "hp_adamweightdecay_s"], [12, 0, 1, "c.hp_adder_p", "hp_adder_p"], [12, 0, 1, "c.hp_adder_s", "hp_adder_s"], [13, 0, 1, "c.hp_applymomentum_p", "hp_applymomentum_p"], [13, 0, 1, "c.hp_applymomentum_s", "hp_applymomentum_s"], [16, 0, 1, "c.hp_avgpoolinggrad_p", "hp_avgpoolinggrad_p"], [16, 0, 1, "c.hp_avgpoolinggrad_s", "hp_avgpoolinggrad_s"], [17, 0, 1, "c.hp_batchtospace_p", "hp_batchtospace_p"], [17, 0, 1, "c.hp_batchtospace_s", "hp_batchtospace_s"], [18, 0, 1, "c.hp_batchtospacend_p", "hp_batchtospacend_p"], [18, 0, 1, "c.hp_batchtospacend_s", "hp_batchtospacend_s"], [19, 0, 1, "c.hp_broadcastto_p", "hp_broadcastto_p"], [19, 0, 1, "c.hp_broadcastto_s", "hp_broadcastto_s"], [10, 0, 1, "c.hp_celu_p", "hp_celu_p"], [10, 0, 1, "c.hp_celu_s", "hp_celu_s"], [10, 0, 1, "c.hp_clip_p", "hp_clip_p"], [10, 0, 1, "c.hp_clip_s", "hp_clip_s"], [20, 0, 1, "c.hp_conv2d_p", "hp_conv2d_p"], [20, 0, 1, "c.hp_conv2d_s", "hp_conv2d_s"], [22, 0, 1, "c.hp_conv2dbackpropfilterfusion_p", "hp_conv2dbackpropfilterfusion_p"], [22, 0, 1, "c.hp_conv2dbackpropfilterfusion_s", "hp_conv2dbackpropfilterfusion_s"], [23, 0, 1, "c.hp_conv2dbackpropinputfusion_p", "hp_conv2dbackpropinputfusion_p"], [23, 0, 1, "c.hp_conv2dbackpropinputfusion_s", "hp_conv2dbackpropinputfusion_s"], [21, 0, 1, "c.hp_convtranspose_p", "hp_convtranspose_p"], [21, 0, 1, "c.hp_convtranspose_s", "hp_convtranspose_s"], [25, 0, 1, "c.hp_crop_and_resize_anycore", "hp_crop_and_resize_anycore"], [26, 0, 1, "c.hp_depthtospace_p", "hp_depthtospace_p"], [26, 0, 1, "c.hp_depthtospace_s", "hp_depthtospace_s"], [28, 0, 1, "c.hp_eltwise_p", "hp_eltwise_p"], [28, 0, 1, "c.hp_eltwise_s", "hp_eltwise_s"], [10, 0, 1, "c.hp_elu_p", "hp_elu_p"], [10, 0, 1, "c.hp_elu_s", "hp_elu_s"], [29, 0, 1, "c.hp_embeddinglookup_p", "hp_embeddinglookup_p"], [29, 0, 1, "c.hp_embeddinglookup_s", "hp_embeddinglookup_s"], [30, 0, 1, "c.hp_equal_p", "hp_equal_p"], [30, 0, 1, "c.hp_equal_s", "hp_equal_s"], [32, 0, 1, "c.hp_expfusion_p", "hp_expfusion_p"], [32, 0, 1, "c.hp_expfusion_s", "hp_expfusion_s"], [34, 0, 1, "c.hp_floor_p", "hp_floor_p"], [34, 0, 1, "c.hp_floor_s", "hp_floor_s"], [35, 0, 1, "c.hp_floordiv_p", "hp_floordiv_p"], [35, 0, 1, "c.hp_floordiv_s", "hp_floordiv_s"], [36, 0, 1, "c.hp_fusedbatchnorm_p", "hp_fusedbatchnorm_p"], [36, 0, 1, "c.hp_fusedbatchnorm_s", "hp_fusedbatchnorm_s"], [10, 0, 1, "c.hp_gelu_p", "hp_gelu_p"], [10, 0, 1, "c.hp_gelu_s", "hp_gelu_s"], [37, 0, 1, "c.hp_groupnormfusion_p", "hp_groupnormfusion_p"], [37, 0, 1, "c.hp_groupnormfusion_s", "hp_groupnormfusion_s"], [10, 0, 1, "c.hp_hardshrink_p", "hp_hardshrink_p"], [10, 0, 1, "c.hp_hardshrink_s", "hp_hardshrink_s"], [10, 0, 1, "c.hp_hardtanh_p", "hp_hardtanh_p"], [10, 0, 1, "c.hp_hardtanh_s", "hp_hardtanh_s"], [10, 0, 1, "c.hp_hsigmoid_p", "hp_hsigmoid_p"], [10, 0, 1, "c.hp_hsigmoid_s", "hp_hsigmoid_s"], [10, 0, 1, "c.hp_hswish_p", "hp_hswish_p"], [10, 0, 1, "c.hp_hswish_s", "hp_hswish_s"], [39, 0, 1, "c.hp_leaky_relu_p", "hp_leaky_relu_p"], [39, 0, 1, "c.hp_leaky_relu_s", "hp_leaky_relu_s"], [10, 0, 1, "c.hp_lrelu_p", "hp_lrelu_p"], [10, 0, 1, "c.hp_lrelu_s", "hp_lrelu_s"], [45, 0, 1, "c.hp_reduce_p", "hp_reduce_p"], [45, 0, 1, "c.hp_reduce_s", "hp_reduce_s"], [10, 0, 1, "c.hp_relu6_p", "hp_relu6_p"], [10, 0, 1, "c.hp_relu6_s", "hp_relu6_s"], [10, 0, 1, "c.hp_relu_p", "hp_relu_p"], [10, 0, 1, "c.hp_relu_s", "hp_relu_s"], [46, 0, 1, "c.hp_resize_anycore", "hp_resize_anycore"], [49, 0, 1, "c.hp_scalefusion_p", "hp_scalefusion_p"], [49, 0, 1, "c.hp_scalefusion_s", "hp_scalefusion_s"], [50, 0, 1, "c.hp_scatter_elements_p", "hp_scatter_elements_p"], [50, 0, 1, "c.hp_scatter_elements_s", "hp_scatter_elements_s"], [51, 0, 1, "c.hp_sgd_p", "hp_sgd_p"], [51, 0, 1, "c.hp_sgd_s", "hp_sgd_s"], [10, 0, 1, "c.hp_sigmoid_p", "hp_sigmoid_p"], [10, 0, 1, "c.hp_sigmoid_s", "hp_sigmoid_s"], [10, 0, 1, "c.hp_softplus_p", "hp_softplus_p"], [10, 0, 1, "c.hp_softplus_s", "hp_softplus_s"], [10, 0, 1, "c.hp_softshrink_p", "hp_softshrink_p"], [10, 0, 1, "c.hp_softshrink_s", "hp_softshrink_s"], [10, 0, 1, "c.hp_softsignopt_p", "hp_softsignopt_p"], [10, 0, 1, "c.hp_softsignopt_s", "hp_softsignopt_s"], [52, 0, 1, "c.hp_spacetobatch_p", "hp_spacetobatch_p"], [52, 0, 1, "c.hp_spacetobatch_s", "hp_spacetobatch_s"], [53, 0, 1, "c.hp_spacetobatchnd_p", "hp_spacetobatchnd_p"], [53, 0, 1, "c.hp_spacetobatchnd_s", "hp_spacetobatchnd_s"], [54, 0, 1, "c.hp_spacetodepth_p", "hp_spacetodepth_p"], [54, 0, 1, "c.hp_spacetodepth_s", "hp_spacetodepth_s"], [10, 0, 1, "c.hp_swish_p", "hp_swish_p"], [10, 0, 1, "c.hp_swish_s", "hp_swish_s"], [10, 0, 1, "c.hp_tanh_p", "hp_tanh_p"], [10, 0, 1, "c.hp_tanh_s", "hp_tanh_s"], [17, 0, 1, "c.i16_batchtospace_p", "i16_batchtospace_p"], [17, 0, 1, "c.i16_batchtospace_s", "i16_batchtospace_s"], [18, 0, 1, "c.i16_batchtospacend_p", "i16_batchtospacend_p"], [18, 0, 1, "c.i16_batchtospacend_s", "i16_batchtospacend_s"], [19, 0, 1, "c.i16_broadcastto_p", "i16_broadcastto_p"], [19, 0, 1, "c.i16_broadcastto_s", "i16_broadcastto_s"], [26, 0, 1, "c.i16_depthtospace_p", "i16_depthtospace_p"], [26, 0, 1, "c.i16_depthtospace_s", "i16_depthtospace_s"], [28, 0, 1, "c.i16_eltwise_p", "i16_eltwise_p"], [28, 0, 1, "c.i16_eltwise_s", "i16_eltwise_s"], [30, 0, 1, "c.i16_equal_p", "i16_equal_p"], [30, 0, 1, "c.i16_equal_s", "i16_equal_s"], [32, 0, 1, "c.i16_expfusion_p", "i16_expfusion_p"], [32, 0, 1, "c.i16_expfusion_s", "i16_expfusion_s"], [42, 0, 1, "c.i16_matmulfusion_p", "i16_matmulfusion_p"], [42, 0, 1, "c.i16_matmulfusion_s", "i16_matmulfusion_s"], [43, 0, 1, "c.i16_raggedrange_p", "i16_raggedrange_p"], [43, 0, 1, "c.i16_raggedrange_s", "i16_raggedrange_s"], [44, 0, 1, "c.i16_range_p", "i16_range_p"], [44, 0, 1, "c.i16_range_s", "i16_range_s"], [45, 0, 1, "c.i16_reduce_p", "i16_reduce_p"], [45, 0, 1, "c.i16_reduce_s", "i16_reduce_s"], [49, 0, 1, "c.i16_scalefusion_p", "i16_scalefusion_p"], [49, 0, 1, "c.i16_scalefusion_s", "i16_scalefusion_s"], [50, 0, 1, "c.i16_scatter_elements_p", "i16_scatter_elements_p"], [50, 0, 1, "c.i16_scatter_elements_s", "i16_scatter_elements_s"], [52, 0, 1, "c.i16_spacetobatch_p", "i16_spacetobatch_p"], [52, 0, 1, "c.i16_spacetobatch_s", "i16_spacetobatch_s"], [53, 0, 1, "c.i16_spacetobatchnd_p", "i16_spacetobatchnd_p"], [53, 0, 1, "c.i16_spacetobatchnd_s", "i16_spacetobatchnd_s"], [54, 0, 1, "c.i16_spacetodepth_p", "i16_spacetodepth_p"], [54, 0, 1, "c.i16_spacetodepth_s", "i16_spacetodepth_s"], [17, 0, 1, "c.i32_batchtospace_p", "i32_batchtospace_p"], [17, 0, 1, "c.i32_batchtospace_s", "i32_batchtospace_s"], [18, 0, 1, "c.i32_batchtospacend_p", "i32_batchtospacend_p"], [18, 0, 1, "c.i32_batchtospacend_s", "i32_batchtospacend_s"], [19, 0, 1, "c.i32_broadcastto_p", "i32_broadcastto_p"], [19, 0, 1, "c.i32_broadcastto_s", "i32_broadcastto_s"], [26, 0, 1, "c.i32_depthtospace_p", "i32_depthtospace_p"], [26, 0, 1, "c.i32_depthtospace_s", "i32_depthtospace_s"], [28, 0, 1, "c.i32_eltwise_p", "i32_eltwise_p"], [28, 0, 1, "c.i32_eltwise_s", "i32_eltwise_s"], [30, 0, 1, "c.i32_equal_p", "i32_equal_p"], [30, 0, 1, "c.i32_equal_s", "i32_equal_s"], [32, 0, 1, "c.i32_expfusion_p", "i32_expfusion_p"], [32, 0, 1, "c.i32_expfusion_s", "i32_expfusion_s"], [42, 0, 1, "c.i32_matmulfusion_p", "i32_matmulfusion_p"], [42, 0, 1, "c.i32_matmulfusion_s", "i32_matmulfusion_s"], [43, 0, 1, "c.i32_raggedrange_p", "i32_raggedrange_p"], [43, 0, 1, "c.i32_raggedrange_s", "i32_raggedrange_s"], [44, 0, 1, "c.i32_range_p", "i32_range_p"], [44, 0, 1, "c.i32_range_s", "i32_range_s"], [45, 0, 1, "c.i32_reduce_p", "i32_reduce_p"], [45, 0, 1, "c.i32_reduce_s", "i32_reduce_s"], [49, 0, 1, "c.i32_scalefusion_p", "i32_scalefusion_p"], [49, 0, 1, "c.i32_scalefusion_s", "i32_scalefusion_s"], [50, 0, 1, "c.i32_scatter_elements_p", "i32_scatter_elements_p"], [50, 0, 1, "c.i32_scatter_elements_s", "i32_scatter_elements_s"], [52, 0, 1, "c.i32_spacetobatch_p", "i32_spacetobatch_p"], [52, 0, 1, "c.i32_spacetobatch_s", "i32_spacetobatch_s"], [53, 0, 1, "c.i32_spacetobatchnd_p", "i32_spacetobatchnd_p"], [53, 0, 1, "c.i32_spacetobatchnd_s", "i32_spacetobatchnd_s"], [54, 0, 1, "c.i32_spacetodepth_p", "i32_spacetodepth_p"], [54, 0, 1, "c.i32_spacetodepth_s", "i32_spacetodepth_s"], [38, 0, 1, "c.i8_Gru_p", "i8_Gru_p"], [38, 0, 1, "c.i8_Gru_s", "i8_Gru_s"], [12, 0, 1, "c.i8_adder_p", "i8_adder_p"], [12, 0, 1, "c.i8_adder_s", "i8_adder_s"], [17, 0, 1, "c.i8_batchtospace_p", "i8_batchtospace_p"], [17, 0, 1, "c.i8_batchtospace_s", "i8_batchtospace_s"], [18, 0, 1, "c.i8_batchtospacend_p", "i8_batchtospacend_p"], [18, 0, 1, "c.i8_batchtospacend_s", "i8_batchtospacend_s"], [19, 0, 1, "c.i8_broadcastto_p", "i8_broadcastto_p"], [19, 0, 1, "c.i8_broadcastto_s", "i8_broadcastto_s"], [10, 0, 1, "c.i8_celu_p", "i8_celu_p"], [10, 0, 1, "c.i8_celu_s", "i8_celu_s"], [10, 0, 1, "c.i8_clip_p", "i8_clip_p"], [10, 0, 1, "c.i8_clip_s", "i8_clip_s"], [20, 0, 1, "c.i8_conv2d_p", "i8_conv2d_p"], [20, 0, 1, "c.i8_conv2d_s", "i8_conv2d_s"], [21, 0, 1, "c.i8_convtranspose_p", "i8_convtranspose_p"], [21, 0, 1, "c.i8_convtranspose_s", "i8_convtranspose_s"], [25, 0, 1, "c.i8_crop_and_resize_anycore", "i8_crop_and_resize_anycore"], [26, 0, 1, "c.i8_depthtospace_p", "i8_depthtospace_p"], [26, 0, 1, "c.i8_depthtospace_s", "i8_depthtospace_s"], [28, 0, 1, "c.i8_eltwise_p", "i8_eltwise_p"], [28, 0, 1, "c.i8_eltwise_s", "i8_eltwise_s"], [10, 0, 1, "c.i8_elu_p", "i8_elu_p"], [10, 0, 1, "c.i8_elu_s", "i8_elu_s"], [30, 0, 1, "c.i8_equal_p", "i8_equal_p"], [30, 0, 1, "c.i8_equal_s", "i8_equal_s"], [32, 0, 1, "c.i8_expfusion_p", "i8_expfusion_p"], [32, 0, 1, "c.i8_expfusion_s", "i8_expfusion_s"], [10, 0, 1, "c.i8_gelu_p", "i8_gelu_p"], [10, 0, 1, "c.i8_gelu_s", "i8_gelu_s"], [10, 0, 1, "c.i8_hardshrink_p", "i8_hardshrink_p"], [10, 0, 1, "c.i8_hardshrink_s", "i8_hardshrink_s"], [10, 0, 1, "c.i8_hardtanh_p", "i8_hardtanh_p"], [10, 0, 1, "c.i8_hardtanh_s", "i8_hardtanh_s"], [10, 0, 1, "c.i8_hsigmoid_p", "i8_hsigmoid_p"], [10, 0, 1, "c.i8_hsigmoid_s", "i8_hsigmoid_s"], [10, 0, 1, "c.i8_hswish_p", "i8_hswish_p"], [10, 0, 1, "c.i8_hswish_s", "i8_hswish_s"], [39, 0, 1, "c.i8_leaky_relu_p", "i8_leaky_relu_p"], [39, 0, 1, "c.i8_leaky_relu_s", "i8_leaky_relu_s"], [10, 0, 1, "c.i8_lrelu_p", "i8_lrelu_p"], [10, 0, 1, "c.i8_lrelu_s", "i8_lrelu_s"], [43, 0, 1, "c.i8_raggedrange_p", "i8_raggedrange_p"], [43, 0, 1, "c.i8_raggedrange_s", "i8_raggedrange_s"], [44, 0, 1, "c.i8_range_p", "i8_range_p"], [44, 0, 1, "c.i8_range_s", "i8_range_s"], [45, 0, 1, "c.i8_reduce_p", "i8_reduce_p"], [45, 0, 1, "c.i8_reduce_s", "i8_reduce_s"], [10, 0, 1, "c.i8_relu6_p", "i8_relu6_p"], [10, 0, 1, "c.i8_relu6_s", "i8_relu6_s"], [10, 0, 1, "c.i8_relu_p", "i8_relu_p"], [10, 0, 1, "c.i8_relu_s", "i8_relu_s"], [46, 0, 1, "c.i8_resize_anycore", "i8_resize_anycore"], [49, 0, 1, "c.i8_scalefusion_p", "i8_scalefusion_p"], [49, 0, 1, "c.i8_scalefusion_s", "i8_scalefusion_s"], [50, 0, 1, "c.i8_scatter_elements_p", "i8_scatter_elements_p"], [50, 0, 1, "c.i8_scatter_elements_s", "i8_scatter_elements_s"], [10, 0, 1, "c.i8_sigmoid_p", "i8_sigmoid_p"], [10, 0, 1, "c.i8_sigmoid_s", "i8_sigmoid_s"], [10, 0, 1, "c.i8_softplus_p", "i8_softplus_p"], [10, 0, 1, "c.i8_softplus_s", "i8_softplus_s"], [10, 0, 1, "c.i8_softshrink_p", "i8_softshrink_p"], [10, 0, 1, "c.i8_softshrink_s", "i8_softshrink_s"], [10, 0, 1, "c.i8_softsignopt_p", "i8_softsignopt_p"], [10, 0, 1, "c.i8_softsignopt_s", "i8_softsignopt_s"], [52, 0, 1, "c.i8_spacetobatch_p", "i8_spacetobatch_p"], [52, 0, 1, "c.i8_spacetobatch_s", "i8_spacetobatch_s"], [53, 0, 1, "c.i8_spacetobatchnd_p", "i8_spacetobatchnd_p"], [53, 0, 1, "c.i8_spacetobatchnd_s", "i8_spacetobatchnd_s"], [54, 0, 1, "c.i8_spacetodepth_p", "i8_spacetodepth_p"], [54, 0, 1, "c.i8_spacetodepth_s", "i8_spacetodepth_s"], [10, 0, 1, "c.i8_swish_p", "i8_swish_p"], [10, 0, 1, "c.i8_swish_s", "i8_swish_s"], [10, 0, 1, "c.i8_tanh_p", "i8_tanh_p"], [10, 0, 1, "c.i8_tanh_s", "i8_tanh_s"]], "anytype_crop_anycore": [[24, 1, 1, "c.anytype_crop_anycore", "axis"], [24, 1, 1, "c.anytype_crop_anycore", "core_mask"], [24, 1, 1, "c.anytype_crop_anycore", "in_shape"], [24, 1, 1, "c.anytype_crop_anycore", "input"], [24, 1, 1, "c.anytype_crop_anycore", "offset"], [24, 1, 1, "c.anytype_crop_anycore", "out_shape"], [24, 1, 1, "c.anytype_crop_anycore", "output"], [24, 1, 1, "c.anytype_crop_anycore", "type_size"]], "anytype_expand_dims_anycore": [[31, 1, 1, "c.anytype_expand_dims_anycore", "core_mask"], [31, 1, 1, "c.anytype_expand_dims_anycore", "dst"], [31, 1, 1, "c.anytype_expand_dims_anycore", "src"], [31, 1, 1, "c.anytype_expand_dims_anycore", "total_copy_size"]], "anytype_fillv2_p": [[33, 1, 1, "c.anytype_fillv2_p", "core_mask"], [33, 1, 1, "c.anytype_fillv2_p", "length"], [33, 1, 1, "c.anytype_fillv2_p", "output"], [33, 1, 1, "c.anytype_fillv2_p", "type_size"], [33, 1, 1, "c.anytype_fillv2_p", "value"]], "anytype_fillv2_s": [[33, 1, 1, "c.anytype_fillv2_s", "core_mask"], [33, 1, 1, "c.anytype_fillv2_s", "length"], [33, 1, 1, "c.anytype_fillv2_s", "output"], [33, 1, 1, "c.anytype_fillv2_s", "type_size"], [33, 1, 1, "c.anytype_fillv2_s", "value"]], "anytype_reverse_sequence_anycore": [[47, 1, 1, "c.anytype_reverse_sequence_anycore", "core_mask"], [47, 1, 1, "c.anytype_reverse_sequence_anycore", "dst"], [47, 1, 1, "c.anytype_reverse_sequence_anycore", "param"], [47, 1, 1, "c.anytype_reverse_sequence_anycore", "seq_lengths"], [47, 1, 1, "c.anytype_reverse_sequence_anycore", "src"]], "anytype_reversev2_anycore": [[48, 1, 1, "c.anytype_reversev2_anycore", "core_mask"], [48, 1, 1, "c.anytype_reversev2_anycore", "dst"], [48, 1, 1, "c.anytype_reversev2_anycore", "param"], [48, 1, 1, "c.anytype_reversev2_anycore", "src"]], "anytype_squeeze_anycore": [[55, 1, 1, "c.anytype_squeeze_anycore", "core_mask"], [55, 1, 1, "c.anytype_squeeze_anycore", "dst"], [55, 1, 1, "c.anytype_squeeze_anycore", "src"], [55, 1, 1, "c.anytype_squeeze_anycore", "total_copy_size"]], "anytype_unsqueeze_anycore": [[56, 1, 1, "c.anytype_unsqueeze_anycore", "core_mask"], [56, 1, 1, "c.anytype_unsqueeze_anycore", "dst"], [56, 1, 1, "c.anytype_unsqueeze_anycore", "src"], [56, 1, 1, "c.anytype_unsqueeze_anycore", "total_copy_size"]], "assert": [[14, 1, 1, "c.assert", "Input"], [14, 1, 1, "c.assert", "output"]], "c128_batchtospace_p": [[17, 1, 1, "c.c128_batchtospace_p", "block_size"], [17, 1, 1, "c.c128_batchtospace_p", "crops"], [17, 1, 1, "c.c128_batchtospace_p", "data_size"], [17, 1, 1, "c.c128_batchtospace_p", "input"], [17, 1, 1, "c.c128_batchtospace_p", "input_shape"], [17, 1, 1, "c.c128_batchtospace_p", "output"]], "c128_batchtospace_s": [[17, 1, 1, "c.c128_batchtospace_s", "block_size"], [17, 1, 1, "c.c128_batchtospace_s", "core_mask"], [17, 1, 1, "c.c128_batchtospace_s", "crops"], [17, 1, 1, "c.c128_batchtospace_s", "data_size"], [17, 1, 1, "c.c128_batchtospace_s", "input"], [17, 1, 1, "c.c128_batchtospace_s", "input_shape"], [17, 1, 1, "c.c128_batchtospace_s", "output"]], "c128_batchtospacend_p": [[18, 1, 1, "c.c128_batchtospacend_p", "block_size"], [18, 1, 1, "c.c128_batchtospacend_p", "crops"], [18, 1, 1, "c.c128_batchtospacend_p", "data_size"], [18, 1, 1, "c.c128_batchtospacend_p", "input"], [18, 1, 1, "c.c128_batchtospacend_p", "input_shape"], [18, 1, 1, "c.c128_batchtospacend_p", "output"]], "c128_batchtospacend_s": [[18, 1, 1, "c.c128_batchtospacend_s", "block_size"], [18, 1, 1, "c.c128_batchtospacend_s", "core_mask"], [18, 1, 1, "c.c128_batchtospacend_s", "crops"], [18, 1, 1, "c.c128_batchtospacend_s", "data_size"], [18, 1, 1, "c.c128_batchtospacend_s", "input"], [18, 1, 1, "c.c128_batchtospacend_s", "input_shape"], [18, 1, 1, "c.c128_batchtospacend_s", "output"]], "c128_broadcastto_p": [[19, 1, 1, "c.c128_broadcastto_p", "data_size"], [19, 1, 1, "c.c128_broadcastto_p", "input"], [19, 1, 1, "c.c128_broadcastto_p", "input_shape"], [19, 1, 1, "c.c128_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.c128_broadcastto_p", "output"], [19, 1, 1, "c.c128_broadcastto_p", "output_shape"], [19, 1, 1, "c.c128_broadcastto_p", "output_shape_size"]], "c128_broadcastto_s": [[19, 1, 1, "c.c128_broadcastto_s", "core_mask"], [19, 1, 1, "c.c128_broadcastto_s", "data_size"], [19, 1, 1, "c.c128_broadcastto_s", "input"], [19, 1, 1, "c.c128_broadcastto_s", "input_shape"], [19, 1, 1, "c.c128_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.c128_broadcastto_s", "output"], [19, 1, 1, "c.c128_broadcastto_s", "output_shape"], [19, 1, 1, "c.c128_broadcastto_s", "output_shape_size"]], "c128_depthtospace_p": [[26, 1, 1, "c.c128_depthtospace_p", "block_size"], [26, 1, 1, "c.c128_depthtospace_p", "data_size"], [26, 1, 1, "c.c128_depthtospace_p", "in_shape"], [26, 1, 1, "c.c128_depthtospace_p", "input"], [26, 1, 1, "c.c128_depthtospace_p", "output"]], "c128_depthtospace_s": [[26, 1, 1, "c.c128_depthtospace_s", "block_size"], [26, 1, 1, "c.c128_depthtospace_s", "core_mask"], [26, 1, 1, "c.c128_depthtospace_s", "data_size"], [26, 1, 1, "c.c128_depthtospace_s", "in_shape"], [26, 1, 1, "c.c128_depthtospace_s", "input"], [26, 1, 1, "c.c128_depthtospace_s", "output"]], "c128_eltwise_p": [[28, 1, 1, "c.c128_eltwise_p", "Input0"], [28, 1, 1, "c.c128_eltwise_p", "Input1"], [28, 1, 1, "c.c128_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.c128_eltwise_p", "length"], [28, 1, 1, "c.c128_eltwise_p", "output"]], "c128_eltwise_s": [[28, 1, 1, "c.c128_eltwise_s", "Input0"], [28, 1, 1, "c.c128_eltwise_s", "Input1"], [28, 1, 1, "c.c128_eltwise_s", "core_mask"], [28, 1, 1, "c.c128_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.c128_eltwise_s", "length"], [28, 1, 1, "c.c128_eltwise_s", "output"]], "c128_equal_p": [[30, 1, 1, "c.c128_equal_p", "Input0"], [30, 1, 1, "c.c128_equal_p", "Input1"], [30, 1, 1, "c.c128_equal_p", "length"], [30, 1, 1, "c.c128_equal_p", "output"]], "c128_equal_s": [[30, 1, 1, "c.c128_equal_s", "Input0"], [30, 1, 1, "c.c128_equal_s", "Input1"], [30, 1, 1, "c.c128_equal_s", "core_mask"], [30, 1, 1, "c.c128_equal_s", "length"], [30, 1, 1, "c.c128_equal_s", "output"]], "c128_expfusion_p": [[32, 1, 1, "c.c128_expfusion_p", "dst_data"], [32, 1, 1, "c.c128_expfusion_p", "in_scale"], [32, 1, 1, "c.c128_expfusion_p", "length"], [32, 1, 1, "c.c128_expfusion_p", "out_scale"], [32, 1, 1, "c.c128_expfusion_p", "scale"], [32, 1, 1, "c.c128_expfusion_p", "src_data"]], "c128_expfusion_s": [[32, 1, 1, "c.c128_expfusion_s", "core_mask"], [32, 1, 1, "c.c128_expfusion_s", "dst_data"], [32, 1, 1, "c.c128_expfusion_s", "in_scale"], [32, 1, 1, "c.c128_expfusion_s", "length"], [32, 1, 1, "c.c128_expfusion_s", "out_scale"], [32, 1, 1, "c.c128_expfusion_s", "scale"], [32, 1, 1, "c.c128_expfusion_s", "src_data"]], "c128_matmulfusion_p": [[42, 1, 1, "c.c128_matmulfusion_p", "A"], [42, 1, 1, "c.c128_matmulfusion_p", "B"], [42, 1, 1, "c.c128_matmulfusion_p", "C"], [42, 1, 1, "c.c128_matmulfusion_p", "K"], [42, 1, 1, "c.c128_matmulfusion_p", "M"], [42, 1, 1, "c.c128_matmulfusion_p", "N"], [42, 1, 1, "c.c128_matmulfusion_p", "activation_type"], [42, 1, 1, "c.c128_matmulfusion_p", "bias"]], "c128_matmulfusion_s": [[42, 1, 1, "c.c128_matmulfusion_s", "A"], [42, 1, 1, "c.c128_matmulfusion_s", "B"], [42, 1, 1, "c.c128_matmulfusion_s", "C"], [42, 1, 1, "c.c128_matmulfusion_s", "K"], [42, 1, 1, "c.c128_matmulfusion_s", "M"], [42, 1, 1, "c.c128_matmulfusion_s", "N"], [42, 1, 1, "c.c128_matmulfusion_s", "activation_type"], [42, 1, 1, "c.c128_matmulfusion_s", "bias"], [42, 1, 1, "c.c128_matmulfusion_s", "core_mask"]], "c128_scatter_elements_p": [[50, 1, 1, "c.c128_scatter_elements_p", "core_mask"], [50, 1, 1, "c.c128_scatter_elements_p", "indices"], [50, 1, 1, "c.c128_scatter_elements_p", "input"], [50, 1, 1, "c.c128_scatter_elements_p", "output"], [50, 1, 1, "c.c128_scatter_elements_p", "param"], [50, 1, 1, "c.c128_scatter_elements_p", "updates"]], "c128_scatter_elements_s": [[50, 1, 1, "c.c128_scatter_elements_s", "core_mask"], [50, 1, 1, "c.c128_scatter_elements_s", "indices"], [50, 1, 1, "c.c128_scatter_elements_s", "input"], [50, 1, 1, "c.c128_scatter_elements_s", "output"], [50, 1, 1, "c.c128_scatter_elements_s", "param"], [50, 1, 1, "c.c128_scatter_elements_s", "updates"]], "c128_spacetobatch_p": [[52, 1, 1, "c.c128_spacetobatch_p", "block_size"], [52, 1, 1, "c.c128_spacetobatch_p", "data_size"], [52, 1, 1, "c.c128_spacetobatch_p", "input"], [52, 1, 1, "c.c128_spacetobatch_p", "input_shape"], [52, 1, 1, "c.c128_spacetobatch_p", "output"], [52, 1, 1, "c.c128_spacetobatch_p", "paddings"]], "c128_spacetobatch_s": [[52, 1, 1, "c.c128_spacetobatch_s", "block_size"], [52, 1, 1, "c.c128_spacetobatch_s", "core_mask"], [52, 1, 1, "c.c128_spacetobatch_s", "data_size"], [52, 1, 1, "c.c128_spacetobatch_s", "input"], [52, 1, 1, "c.c128_spacetobatch_s", "input_shape"], [52, 1, 1, "c.c128_spacetobatch_s", "output"], [52, 1, 1, "c.c128_spacetobatch_s", "paddings"]], "c128_spacetobatchnd_p": [[53, 1, 1, "c.c128_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.c128_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.c128_spacetobatchnd_p", "input"], [53, 1, 1, "c.c128_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.c128_spacetobatchnd_p", "output"], [53, 1, 1, "c.c128_spacetobatchnd_p", "paddings"]], "c128_spacetobatchnd_s": [[53, 1, 1, "c.c128_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.c128_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.c128_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.c128_spacetobatchnd_s", "input"], [53, 1, 1, "c.c128_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.c128_spacetobatchnd_s", "output"], [53, 1, 1, "c.c128_spacetobatchnd_s", "paddings"]], "c128_spacetodepth_p": [[54, 1, 1, "c.c128_spacetodepth_p", "block"], [54, 1, 1, "c.c128_spacetodepth_p", "data_size"], [54, 1, 1, "c.c128_spacetodepth_p", "in_shape"], [54, 1, 1, "c.c128_spacetodepth_p", "input"], [54, 1, 1, "c.c128_spacetodepth_p", "output"]], "c128_spacetodepth_s": [[54, 1, 1, "c.c128_spacetodepth_s", "block"], [54, 1, 1, "c.c128_spacetodepth_s", "core_mask"], [54, 1, 1, "c.c128_spacetodepth_s", "data_size"], [54, 1, 1, "c.c128_spacetodepth_s", "in_shape"], [54, 1, 1, "c.c128_spacetodepth_s", "input"], [54, 1, 1, "c.c128_spacetodepth_s", "output"]], "c64_batchtospace_p": [[17, 1, 1, "c.c64_batchtospace_p", "block_size"], [17, 1, 1, "c.c64_batchtospace_p", "crops"], [17, 1, 1, "c.c64_batchtospace_p", "data_size"], [17, 1, 1, "c.c64_batchtospace_p", "input"], [17, 1, 1, "c.c64_batchtospace_p", "input_shape"], [17, 1, 1, "c.c64_batchtospace_p", "output"]], "c64_batchtospace_s": [[17, 1, 1, "c.c64_batchtospace_s", "block_size"], [17, 1, 1, "c.c64_batchtospace_s", "core_mask"], [17, 1, 1, "c.c64_batchtospace_s", "crops"], [17, 1, 1, "c.c64_batchtospace_s", "data_size"], [17, 1, 1, "c.c64_batchtospace_s", "input"], [17, 1, 1, "c.c64_batchtospace_s", "input_shape"], [17, 1, 1, "c.c64_batchtospace_s", "output"]], "c64_batchtospacend_p": [[18, 1, 1, "c.c64_batchtospacend_p", "block_size"], [18, 1, 1, "c.c64_batchtospacend_p", "crops"], [18, 1, 1, "c.c64_batchtospacend_p", "data_size"], [18, 1, 1, "c.c64_batchtospacend_p", "input"], [18, 1, 1, "c.c64_batchtospacend_p", "input_shape"], [18, 1, 1, "c.c64_batchtospacend_p", "output"]], "c64_batchtospacend_s": [[18, 1, 1, "c.c64_batchtospacend_s", "block_size"], [18, 1, 1, "c.c64_batchtospacend_s", "core_mask"], [18, 1, 1, "c.c64_batchtospacend_s", "crops"], [18, 1, 1, "c.c64_batchtospacend_s", "data_size"], [18, 1, 1, "c.c64_batchtospacend_s", "input"], [18, 1, 1, "c.c64_batchtospacend_s", "input_shape"], [18, 1, 1, "c.c64_batchtospacend_s", "output"]], "c64_broadcastto_p": [[19, 1, 1, "c.c64_broadcastto_p", "data_size"], [19, 1, 1, "c.c64_broadcastto_p", "input"], [19, 1, 1, "c.c64_broadcastto_p", "input_shape"], [19, 1, 1, "c.c64_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.c64_broadcastto_p", "output"], [19, 1, 1, "c.c64_broadcastto_p", "output_shape"], [19, 1, 1, "c.c64_broadcastto_p", "output_shape_size"]], "c64_broadcastto_s": [[19, 1, 1, "c.c64_broadcastto_s", "core_mask"], [19, 1, 1, "c.c64_broadcastto_s", "data_size"], [19, 1, 1, "c.c64_broadcastto_s", "input"], [19, 1, 1, "c.c64_broadcastto_s", "input_shape"], [19, 1, 1, "c.c64_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.c64_broadcastto_s", "output"], [19, 1, 1, "c.c64_broadcastto_s", "output_shape"], [19, 1, 1, "c.c64_broadcastto_s", "output_shape_size"]], "c64_depthtospace_p": [[26, 1, 1, "c.c64_depthtospace_p", "block_size"], [26, 1, 1, "c.c64_depthtospace_p", "data_size"], [26, 1, 1, "c.c64_depthtospace_p", "in_shape"], [26, 1, 1, "c.c64_depthtospace_p", "input"], [26, 1, 1, "c.c64_depthtospace_p", "output"]], "c64_depthtospace_s": [[26, 1, 1, "c.c64_depthtospace_s", "block_size"], [26, 1, 1, "c.c64_depthtospace_s", "core_mask"], [26, 1, 1, "c.c64_depthtospace_s", "data_size"], [26, 1, 1, "c.c64_depthtospace_s", "in_shape"], [26, 1, 1, "c.c64_depthtospace_s", "input"], [26, 1, 1, "c.c64_depthtospace_s", "output"]], "c64_eltwise_p": [[28, 1, 1, "c.c64_eltwise_p", "Input0"], [28, 1, 1, "c.c64_eltwise_p", "Input1"], [28, 1, 1, "c.c64_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.c64_eltwise_p", "length"], [28, 1, 1, "c.c64_eltwise_p", "output"]], "c64_eltwise_s": [[28, 1, 1, "c.c64_eltwise_s", "Input0"], [28, 1, 1, "c.c64_eltwise_s", "Input1"], [28, 1, 1, "c.c64_eltwise_s", "core_mask"], [28, 1, 1, "c.c64_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.c64_eltwise_s", "length"], [28, 1, 1, "c.c64_eltwise_s", "output"]], "c64_equal_p": [[30, 1, 1, "c.c64_equal_p", "Input0"], [30, 1, 1, "c.c64_equal_p", "Input1"], [30, 1, 1, "c.c64_equal_p", "length"], [30, 1, 1, "c.c64_equal_p", "output"]], "c64_equal_s": [[30, 1, 1, "c.c64_equal_s", "Input0"], [30, 1, 1, "c.c64_equal_s", "Input1"], [30, 1, 1, "c.c64_equal_s", "core_mask"], [30, 1, 1, "c.c64_equal_s", "length"], [30, 1, 1, "c.c64_equal_s", "output"]], "c64_expfusion_p": [[32, 1, 1, "c.c64_expfusion_p", "dst_data"], [32, 1, 1, "c.c64_expfusion_p", "in_scale"], [32, 1, 1, "c.c64_expfusion_p", "length"], [32, 1, 1, "c.c64_expfusion_p", "out_scale"], [32, 1, 1, "c.c64_expfusion_p", "scale"], [32, 1, 1, "c.c64_expfusion_p", "src_data"]], "c64_expfusion_s": [[32, 1, 1, "c.c64_expfusion_s", "core_mask"], [32, 1, 1, "c.c64_expfusion_s", "dst_data"], [32, 1, 1, "c.c64_expfusion_s", "in_scale"], [32, 1, 1, "c.c64_expfusion_s", "length"], [32, 1, 1, "c.c64_expfusion_s", "out_scale"], [32, 1, 1, "c.c64_expfusion_s", "scale"], [32, 1, 1, "c.c64_expfusion_s", "src_data"]], "c64_matmulfusion_p": [[42, 1, 1, "c.c64_matmulfusion_p", "A"], [42, 1, 1, "c.c64_matmulfusion_p", "B"], [42, 1, 1, "c.c64_matmulfusion_p", "C"], [42, 1, 1, "c.c64_matmulfusion_p", "K"], [42, 1, 1, "c.c64_matmulfusion_p", "M"], [42, 1, 1, "c.c64_matmulfusion_p", "N"], [42, 1, 1, "c.c64_matmulfusion_p", "activation_type"], [42, 1, 1, "c.c64_matmulfusion_p", "bias"]], "c64_matmulfusion_s": [[42, 1, 1, "c.c64_matmulfusion_s", "A"], [42, 1, 1, "c.c64_matmulfusion_s", "B"], [42, 1, 1, "c.c64_matmulfusion_s", "C"], [42, 1, 1, "c.c64_matmulfusion_s", "K"], [42, 1, 1, "c.c64_matmulfusion_s", "M"], [42, 1, 1, "c.c64_matmulfusion_s", "N"], [42, 1, 1, "c.c64_matmulfusion_s", "activation_type"], [42, 1, 1, "c.c64_matmulfusion_s", "bias"], [42, 1, 1, "c.c64_matmulfusion_s", "core_mask"]], "c64_scatter_elements_p": [[50, 1, 1, "c.c64_scatter_elements_p", "core_mask"], [50, 1, 1, "c.c64_scatter_elements_p", "indices"], [50, 1, 1, "c.c64_scatter_elements_p", "input"], [50, 1, 1, "c.c64_scatter_elements_p", "output"], [50, 1, 1, "c.c64_scatter_elements_p", "param"], [50, 1, 1, "c.c64_scatter_elements_p", "updates"]], "c64_scatter_elements_s": [[50, 1, 1, "c.c64_scatter_elements_s", "core_mask"], [50, 1, 1, "c.c64_scatter_elements_s", "indices"], [50, 1, 1, "c.c64_scatter_elements_s", "input"], [50, 1, 1, "c.c64_scatter_elements_s", "output"], [50, 1, 1, "c.c64_scatter_elements_s", "param"], [50, 1, 1, "c.c64_scatter_elements_s", "updates"]], "c64_spacetobatch_p": [[52, 1, 1, "c.c64_spacetobatch_p", "block_size"], [52, 1, 1, "c.c64_spacetobatch_p", "data_size"], [52, 1, 1, "c.c64_spacetobatch_p", "input"], [52, 1, 1, "c.c64_spacetobatch_p", "input_shape"], [52, 1, 1, "c.c64_spacetobatch_p", "output"], [52, 1, 1, "c.c64_spacetobatch_p", "paddings"]], "c64_spacetobatch_s": [[52, 1, 1, "c.c64_spacetobatch_s", "block_size"], [52, 1, 1, "c.c64_spacetobatch_s", "core_mask"], [52, 1, 1, "c.c64_spacetobatch_s", "data_size"], [52, 1, 1, "c.c64_spacetobatch_s", "input"], [52, 1, 1, "c.c64_spacetobatch_s", "input_shape"], [52, 1, 1, "c.c64_spacetobatch_s", "output"], [52, 1, 1, "c.c64_spacetobatch_s", "paddings"]], "c64_spacetobatchnd_p": [[53, 1, 1, "c.c64_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.c64_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.c64_spacetobatchnd_p", "input"], [53, 1, 1, "c.c64_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.c64_spacetobatchnd_p", "output"], [53, 1, 1, "c.c64_spacetobatchnd_p", "paddings"]], "c64_spacetobatchnd_s": [[53, 1, 1, "c.c64_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.c64_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.c64_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.c64_spacetobatchnd_s", "input"], [53, 1, 1, "c.c64_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.c64_spacetobatchnd_s", "output"], [53, 1, 1, "c.c64_spacetobatchnd_s", "paddings"]], "c64_spacetodepth_p": [[54, 1, 1, "c.c64_spacetodepth_p", "block"], [54, 1, 1, "c.c64_spacetodepth_p", "data_size"], [54, 1, 1, "c.c64_spacetodepth_p", "in_shape"], [54, 1, 1, "c.c64_spacetodepth_p", "input"], [54, 1, 1, "c.c64_spacetodepth_p", "output"]], "c64_spacetodepth_s": [[54, 1, 1, "c.c64_spacetodepth_s", "block"], [54, 1, 1, "c.c64_spacetodepth_s", "core_mask"], [54, 1, 1, "c.c64_spacetodepth_s", "data_size"], [54, 1, 1, "c.c64_spacetodepth_s", "in_shape"], [54, 1, 1, "c.c64_spacetodepth_s", "input"], [54, 1, 1, "c.c64_spacetodepth_s", "output"]], "dp_batchtospace_p": [[17, 1, 1, "c.dp_batchtospace_p", "block_size"], [17, 1, 1, "c.dp_batchtospace_p", "crops"], [17, 1, 1, "c.dp_batchtospace_p", "data_size"], [17, 1, 1, "c.dp_batchtospace_p", "input"], [17, 1, 1, "c.dp_batchtospace_p", "input_shape"], [17, 1, 1, "c.dp_batchtospace_p", "output"]], "dp_batchtospace_s": [[17, 1, 1, "c.dp_batchtospace_s", "block_size"], [17, 1, 1, "c.dp_batchtospace_s", "core_mask"], [17, 1, 1, "c.dp_batchtospace_s", "crops"], [17, 1, 1, "c.dp_batchtospace_s", "data_size"], [17, 1, 1, "c.dp_batchtospace_s", "input"], [17, 1, 1, "c.dp_batchtospace_s", "input_shape"], [17, 1, 1, "c.dp_batchtospace_s", "output"]], "dp_batchtospacend_p": [[18, 1, 1, "c.dp_batchtospacend_p", "block_size"], [18, 1, 1, "c.dp_batchtospacend_p", "crops"], [18, 1, 1, "c.dp_batchtospacend_p", "data_size"], [18, 1, 1, "c.dp_batchtospacend_p", "input"], [18, 1, 1, "c.dp_batchtospacend_p", "input_shape"], [18, 1, 1, "c.dp_batchtospacend_p", "output"]], "dp_batchtospacend_s": [[18, 1, 1, "c.dp_batchtospacend_s", "block_size"], [18, 1, 1, "c.dp_batchtospacend_s", "core_mask"], [18, 1, 1, "c.dp_batchtospacend_s", "crops"], [18, 1, 1, "c.dp_batchtospacend_s", "data_size"], [18, 1, 1, "c.dp_batchtospacend_s", "input"], [18, 1, 1, "c.dp_batchtospacend_s", "input_shape"], [18, 1, 1, "c.dp_batchtospacend_s", "output"]], "dp_broadcastto_p": [[19, 1, 1, "c.dp_broadcastto_p", "data_size"], [19, 1, 1, "c.dp_broadcastto_p", "input"], [19, 1, 1, "c.dp_broadcastto_p", "input_shape"], [19, 1, 1, "c.dp_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.dp_broadcastto_p", "output"], [19, 1, 1, "c.dp_broadcastto_p", "output_shape"], [19, 1, 1, "c.dp_broadcastto_p", "output_shape_size"]], "dp_broadcastto_s": [[19, 1, 1, "c.dp_broadcastto_s", "core_mask"], [19, 1, 1, "c.dp_broadcastto_s", "data_size"], [19, 1, 1, "c.dp_broadcastto_s", "input"], [19, 1, 1, "c.dp_broadcastto_s", "input_shape"], [19, 1, 1, "c.dp_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.dp_broadcastto_s", "output"], [19, 1, 1, "c.dp_broadcastto_s", "output_shape"], [19, 1, 1, "c.dp_broadcastto_s", "output_shape_size"]], "dp_depthtospace_p": [[26, 1, 1, "c.dp_depthtospace_p", "block_size"], [26, 1, 1, "c.dp_depthtospace_p", "data_size"], [26, 1, 1, "c.dp_depthtospace_p", "in_shape"], [26, 1, 1, "c.dp_depthtospace_p", "input"], [26, 1, 1, "c.dp_depthtospace_p", "output"]], "dp_depthtospace_s": [[26, 1, 1, "c.dp_depthtospace_s", "block_size"], [26, 1, 1, "c.dp_depthtospace_s", "core_mask"], [26, 1, 1, "c.dp_depthtospace_s", "data_size"], [26, 1, 1, "c.dp_depthtospace_s", "in_shape"], [26, 1, 1, "c.dp_depthtospace_s", "input"], [26, 1, 1, "c.dp_depthtospace_s", "output"]], "dp_eltwise_p": [[28, 1, 1, "c.dp_eltwise_p", "Input0"], [28, 1, 1, "c.dp_eltwise_p", "Input1"], [28, 1, 1, "c.dp_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.dp_eltwise_p", "length"], [28, 1, 1, "c.dp_eltwise_p", "output"]], "dp_eltwise_s": [[28, 1, 1, "c.dp_eltwise_s", "Input0"], [28, 1, 1, "c.dp_eltwise_s", "Input1"], [28, 1, 1, "c.dp_eltwise_s", "core_mask"], [28, 1, 1, "c.dp_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.dp_eltwise_s", "length"], [28, 1, 1, "c.dp_eltwise_s", "output"]], "dp_equal_p": [[30, 1, 1, "c.dp_equal_p", "Input0"], [30, 1, 1, "c.dp_equal_p", "Input1"], [30, 1, 1, "c.dp_equal_p", "length"], [30, 1, 1, "c.dp_equal_p", "output"]], "dp_equal_s": [[30, 1, 1, "c.dp_equal_s", "Input0"], [30, 1, 1, "c.dp_equal_s", "Input1"], [30, 1, 1, "c.dp_equal_s", "core_mask"], [30, 1, 1, "c.dp_equal_s", "length"], [30, 1, 1, "c.dp_equal_s", "output"]], "dp_expfusion_p": [[32, 1, 1, "c.dp_expfusion_p", "dst_data"], [32, 1, 1, "c.dp_expfusion_p", "in_scale"], [32, 1, 1, "c.dp_expfusion_p", "length"], [32, 1, 1, "c.dp_expfusion_p", "out_scale"], [32, 1, 1, "c.dp_expfusion_p", "scale"], [32, 1, 1, "c.dp_expfusion_p", "src_data"]], "dp_expfusion_s": [[32, 1, 1, "c.dp_expfusion_s", "core_mask"], [32, 1, 1, "c.dp_expfusion_s", "dst_data"], [32, 1, 1, "c.dp_expfusion_s", "in_scale"], [32, 1, 1, "c.dp_expfusion_s", "length"], [32, 1, 1, "c.dp_expfusion_s", "out_scale"], [32, 1, 1, "c.dp_expfusion_s", "scale"], [32, 1, 1, "c.dp_expfusion_s", "src_data"]], "dp_floor_p": [[34, 1, 1, "c.dp_floor_p", "dst_data"], [34, 1, 1, "c.dp_floor_p", "length"], [34, 1, 1, "c.dp_floor_p", "src_data"]], "dp_floor_s": [[34, 1, 1, "c.dp_floor_s", "core_mask"], [34, 1, 1, "c.dp_floor_s", "dst_data"], [34, 1, 1, "c.dp_floor_s", "length"], [34, 1, 1, "c.dp_floor_s", "src_data"]], "dp_floordiv_p": [[35, 1, 1, "c.dp_floordiv_p", "dst_data"], [35, 1, 1, "c.dp_floordiv_p", "length"], [35, 1, 1, "c.dp_floordiv_p", "src_data0"], [35, 1, 1, "c.dp_floordiv_p", "src_data1"]], "dp_floordiv_s": [[35, 1, 1, "c.dp_floordiv_s", "core_mask"], [35, 1, 1, "c.dp_floordiv_s", "dst_data"], [35, 1, 1, "c.dp_floordiv_s", "length"], [35, 1, 1, "c.dp_floordiv_s", "src_data0"], [35, 1, 1, "c.dp_floordiv_s", "src_data1"]], "dp_matmulfusion_p": [[42, 1, 1, "c.dp_matmulfusion_p", "A"], [42, 1, 1, "c.dp_matmulfusion_p", "B"], [42, 1, 1, "c.dp_matmulfusion_p", "C"], [42, 1, 1, "c.dp_matmulfusion_p", "K"], [42, 1, 1, "c.dp_matmulfusion_p", "M"], [42, 1, 1, "c.dp_matmulfusion_p", "N"], [42, 1, 1, "c.dp_matmulfusion_p", "activation_type"], [42, 1, 1, "c.dp_matmulfusion_p", "bias"]], "dp_matmulfusion_s": [[42, 1, 1, "c.dp_matmulfusion_s", "A"], [42, 1, 1, "c.dp_matmulfusion_s", "B"], [42, 1, 1, "c.dp_matmulfusion_s", "C"], [42, 1, 1, "c.dp_matmulfusion_s", "K"], [42, 1, 1, "c.dp_matmulfusion_s", "M"], [42, 1, 1, "c.dp_matmulfusion_s", "N"], [42, 1, 1, "c.dp_matmulfusion_s", "activation_type"], [42, 1, 1, "c.dp_matmulfusion_s", "bias"], [42, 1, 1, "c.dp_matmulfusion_s", "core_mask"]], "dp_raggedrange_p": [[43, 1, 1, "c.dp_raggedrange_p", "deltas"], [43, 1, 1, "c.dp_raggedrange_p", "limits"], [43, 1, 1, "c.dp_raggedrange_p", "range_count"], [43, 1, 1, "c.dp_raggedrange_p", "splits"], [43, 1, 1, "c.dp_raggedrange_p", "starts"], [43, 1, 1, "c.dp_raggedrange_p", "values"]], "dp_raggedrange_s": [[43, 1, 1, "c.dp_raggedrange_s", "core_mask"], [43, 1, 1, "c.dp_raggedrange_s", "deltas"], [43, 1, 1, "c.dp_raggedrange_s", "limits"], [43, 1, 1, "c.dp_raggedrange_s", "range_count"], [43, 1, 1, "c.dp_raggedrange_s", "splits"], [43, 1, 1, "c.dp_raggedrange_s", "starts"], [43, 1, 1, "c.dp_raggedrange_s", "values"]], "dp_range_p": [[44, 1, 1, "c.dp_range_p", "delta"], [44, 1, 1, "c.dp_range_p", "length"], [44, 1, 1, "c.dp_range_p", "output"], [44, 1, 1, "c.dp_range_p", "start"]], "dp_range_s": [[44, 1, 1, "c.dp_range_s", "core_mask"], [44, 1, 1, "c.dp_range_s", "delta"], [44, 1, 1, "c.dp_range_s", "length"], [44, 1, 1, "c.dp_range_s", "output"], [44, 1, 1, "c.dp_range_s", "start"]], "dp_reduce_p": [[45, 1, 1, "c.dp_reduce_p", "core_mask"], [45, 1, 1, "c.dp_reduce_p", "dst_data"], [45, 1, 1, "c.dp_reduce_p", "param"], [45, 1, 1, "c.dp_reduce_p", "src_data"], [45, 1, 1, "c.dp_reduce_p", "tmp_dst_data"], [45, 1, 1, "c.dp_reduce_p", "tmp_src_data"]], "dp_reduce_s": [[45, 1, 1, "c.dp_reduce_s", "core_mask"], [45, 1, 1, "c.dp_reduce_s", "dst_data"], [45, 1, 1, "c.dp_reduce_s", "param"], [45, 1, 1, "c.dp_reduce_s", "src_data"]], "dp_scalefusion_p": [[49, 1, 1, "c.dp_scalefusion_p", "bias"], [49, 1, 1, "c.dp_scalefusion_p", "dst_data"], [49, 1, 1, "c.dp_scalefusion_p", "length"], [49, 1, 1, "c.dp_scalefusion_p", "scale"], [49, 1, 1, "c.dp_scalefusion_p", "src_data"]], "dp_scalefusion_s": [[49, 1, 1, "c.dp_scalefusion_s", "bias"], [49, 1, 1, "c.dp_scalefusion_s", "core_mask"], [49, 1, 1, "c.dp_scalefusion_s", "dst_data"], [49, 1, 1, "c.dp_scalefusion_s", "length"], [49, 1, 1, "c.dp_scalefusion_s", "scale"], [49, 1, 1, "c.dp_scalefusion_s", "src_data"]], "dp_scatter_elements_p": [[50, 1, 1, "c.dp_scatter_elements_p", "core_mask"], [50, 1, 1, "c.dp_scatter_elements_p", "indices"], [50, 1, 1, "c.dp_scatter_elements_p", "input"], [50, 1, 1, "c.dp_scatter_elements_p", "output"], [50, 1, 1, "c.dp_scatter_elements_p", "param"], [50, 1, 1, "c.dp_scatter_elements_p", "updates"]], "dp_scatter_elements_s": [[50, 1, 1, "c.dp_scatter_elements_s", "core_mask"], [50, 1, 1, "c.dp_scatter_elements_s", "indices"], [50, 1, 1, "c.dp_scatter_elements_s", "input"], [50, 1, 1, "c.dp_scatter_elements_s", "output"], [50, 1, 1, "c.dp_scatter_elements_s", "param"], [50, 1, 1, "c.dp_scatter_elements_s", "updates"]], "dp_spacetobatch_p": [[52, 1, 1, "c.dp_spacetobatch_p", "block_size"], [52, 1, 1, "c.dp_spacetobatch_p", "data_size"], [52, 1, 1, "c.dp_spacetobatch_p", "input"], [52, 1, 1, "c.dp_spacetobatch_p", "input_shape"], [52, 1, 1, "c.dp_spacetobatch_p", "output"], [52, 1, 1, "c.dp_spacetobatch_p", "paddings"]], "dp_spacetobatch_s": [[52, 1, 1, "c.dp_spacetobatch_s", "block_size"], [52, 1, 1, "c.dp_spacetobatch_s", "core_mask"], [52, 1, 1, "c.dp_spacetobatch_s", "data_size"], [52, 1, 1, "c.dp_spacetobatch_s", "input"], [52, 1, 1, "c.dp_spacetobatch_s", "input_shape"], [52, 1, 1, "c.dp_spacetobatch_s", "output"], [52, 1, 1, "c.dp_spacetobatch_s", "paddings"]], "dp_spacetobatchnd_p": [[53, 1, 1, "c.dp_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.dp_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.dp_spacetobatchnd_p", "input"], [53, 1, 1, "c.dp_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.dp_spacetobatchnd_p", "output"], [53, 1, 1, "c.dp_spacetobatchnd_p", "paddings"]], "dp_spacetobatchnd_s": [[53, 1, 1, "c.dp_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.dp_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.dp_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.dp_spacetobatchnd_s", "input"], [53, 1, 1, "c.dp_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.dp_spacetobatchnd_s", "output"], [53, 1, 1, "c.dp_spacetobatchnd_s", "paddings"]], "dp_spacetodepth_p": [[54, 1, 1, "c.dp_spacetodepth_p", "block"], [54, 1, 1, "c.dp_spacetodepth_p", "data_size"], [54, 1, 1, "c.dp_spacetodepth_p", "in_shape"], [54, 1, 1, "c.dp_spacetodepth_p", "input"], [54, 1, 1, "c.dp_spacetodepth_p", "output"]], "dp_spacetodepth_s": [[54, 1, 1, "c.dp_spacetodepth_s", "block"], [54, 1, 1, "c.dp_spacetodepth_s", "core_mask"], [54, 1, 1, "c.dp_spacetodepth_s", "data_size"], [54, 1, 1, "c.dp_spacetodepth_s", "in_shape"], [54, 1, 1, "c.dp_spacetodepth_s", "input"], [54, 1, 1, "c.dp_spacetodepth_s", "output"]], "fp_Gru_p": [[38, 1, 1, "c.fp_Gru_p", "buffer"], [38, 1, 1, "c.fp_Gru_p", "core_mask"], [38, 1, 1, "c.fp_Gru_p", "gru_param"], [38, 1, 1, "c.fp_Gru_p", "hidden_state"], [38, 1, 1, "c.fp_Gru_p", "input"], [38, 1, 1, "c.fp_Gru_p", "input_bias"], [38, 1, 1, "c.fp_Gru_p", "output"], [38, 1, 1, "c.fp_Gru_p", "state_bias"], [38, 1, 1, "c.fp_Gru_p", "weight_g"], [38, 1, 1, "c.fp_Gru_p", "weight_r"]], "fp_Gru_s": [[38, 1, 1, "c.fp_Gru_s", "buffer"], [38, 1, 1, "c.fp_Gru_s", "core_mask"], [38, 1, 1, "c.fp_Gru_s", "gru_param"], [38, 1, 1, "c.fp_Gru_s", "hidden_state"], [38, 1, 1, "c.fp_Gru_s", "input"], [38, 1, 1, "c.fp_Gru_s", "input_bias"], [38, 1, 1, "c.fp_Gru_s", "output"], [38, 1, 1, "c.fp_Gru_s", "state_bias"], [38, 1, 1, "c.fp_Gru_s", "weight_g"], [38, 1, 1, "c.fp_Gru_s", "weight_r"]], "fp_Lstm_p": [[41, 1, 1, "c.fp_Lstm_p", "buffer"], [41, 1, 1, "c.fp_Lstm_p", "cell_state"], [41, 1, 1, "c.fp_Lstm_p", "hidden_state"], [41, 1, 1, "c.fp_Lstm_p", "input"], [41, 1, 1, "c.fp_Lstm_p", "input_bias"], [41, 1, 1, "c.fp_Lstm_p", "lstm_param"], [41, 1, 1, "c.fp_Lstm_p", "output"], [41, 1, 1, "c.fp_Lstm_p", "state_bias"], [41, 1, 1, "c.fp_Lstm_p", "weight_h"], [41, 1, 1, "c.fp_Lstm_p", "weight_i"]], "fp_Lstm_s": [[41, 1, 1, "c.fp_Lstm_s", "buffer"], [41, 1, 1, "c.fp_Lstm_s", "cell_state"], [41, 1, 1, "c.fp_Lstm_s", "core_mask"], [41, 1, 1, "c.fp_Lstm_s", "hidden_state"], [41, 1, 1, "c.fp_Lstm_s", "input"], [41, 1, 1, "c.fp_Lstm_s", "input_bias"], [41, 1, 1, "c.fp_Lstm_s", "lstm_param"], [41, 1, 1, "c.fp_Lstm_s", "output"], [41, 1, 1, "c.fp_Lstm_s", "state_bias"], [41, 1, 1, "c.fp_Lstm_s", "weight_h"], [41, 1, 1, "c.fp_Lstm_s", "weight_i"]], "fp_adamweightdecay_p": [[11, 1, 1, "c.fp_adamweightdecay_p", "beta1"], [11, 1, 1, "c.fp_adamweightdecay_p", "beta2"], [11, 1, 1, "c.fp_adamweightdecay_p", "decay"], [11, 1, 1, "c.fp_adamweightdecay_p", "epsilon"], [11, 1, 1, "c.fp_adamweightdecay_p", "gradient"], [11, 1, 1, "c.fp_adamweightdecay_p", "length"], [11, 1, 1, "c.fp_adamweightdecay_p", "lr"], [11, 1, 1, "c.fp_adamweightdecay_p", "m"], [11, 1, 1, "c.fp_adamweightdecay_p", "v"], [11, 1, 1, "c.fp_adamweightdecay_p", "var"]], "fp_adamweightdecay_s": [[11, 1, 1, "c.fp_adamweightdecay_s", "beta1"], [11, 1, 1, "c.fp_adamweightdecay_s", "beta2"], [11, 1, 1, "c.fp_adamweightdecay_s", "core_mask"], [11, 1, 1, "c.fp_adamweightdecay_s", "decay"], [11, 1, 1, "c.fp_adamweightdecay_s", "end"], [11, 1, 1, "c.fp_adamweightdecay_s", "epsilon"], [11, 1, 1, "c.fp_adamweightdecay_s", "gradient"], [11, 1, 1, "c.fp_adamweightdecay_s", "lr"], [11, 1, 1, "c.fp_adamweightdecay_s", "m"], [11, 1, 1, "c.fp_adamweightdecay_s", "start"], [11, 1, 1, "c.fp_adamweightdecay_s", "v"], [11, 1, 1, "c.fp_adamweightdecay_s", "var"]], "fp_adder_p": [[12, 1, 1, "c.fp_adder_p", "bias"], [12, 1, 1, "c.fp_adder_p", "conv_param"], [12, 1, 1, "c.fp_adder_p", "core_mask"], [12, 1, 1, "c.fp_adder_p", "input_w"], [12, 1, 1, "c.fp_adder_p", "input_x"], [12, 1, 1, "c.fp_adder_p", "out_y"]], "fp_adder_s": [[12, 1, 1, "c.fp_adder_s", "bias"], [12, 1, 1, "c.fp_adder_s", "core_mask"], [12, 1, 1, "c.fp_adder_s", "input_w"], [12, 1, 1, "c.fp_adder_s", "input_x"], [12, 1, 1, "c.fp_adder_s", "out_y"], [12, 1, 1, "c.fp_adder_s", "param"]], "fp_applymomentum_p": [[13, 1, 1, "c.fp_applymomentum_p", "accumulate"], [13, 1, 1, "c.fp_applymomentum_p", "gradient"], [13, 1, 1, "c.fp_applymomentum_p", "learning_rate"], [13, 1, 1, "c.fp_applymomentum_p", "length"], [13, 1, 1, "c.fp_applymomentum_p", "moment"], [13, 1, 1, "c.fp_applymomentum_p", "nesterov"], [13, 1, 1, "c.fp_applymomentum_p", "weight"]], "fp_applymomentum_s": [[13, 1, 1, "c.fp_applymomentum_s", "accumulate"], [13, 1, 1, "c.fp_applymomentum_s", "core_mask"], [13, 1, 1, "c.fp_applymomentum_s", "end"], [13, 1, 1, "c.fp_applymomentum_s", "gradient"], [13, 1, 1, "c.fp_applymomentum_s", "learning_rate"], [13, 1, 1, "c.fp_applymomentum_s", "moment"], [13, 1, 1, "c.fp_applymomentum_s", "nesterov"], [13, 1, 1, "c.fp_applymomentum_s", "start"], [13, 1, 1, "c.fp_applymomentum_s", "weight"]], "fp_attention_p": [[15, 1, 1, "c.fp_attention_p", "K"], [15, 1, 1, "c.fp_attention_p", "Q"], [15, 1, 1, "c.fp_attention_p", "QK"], [15, 1, 1, "c.fp_attention_p", "V"], [15, 1, 1, "c.fp_attention_p", "batch_size"], [15, 1, 1, "c.fp_attention_p", "head_dim"], [15, 1, 1, "c.fp_attention_p", "head_num"], [15, 1, 1, "c.fp_attention_p", "output"], [15, 1, 1, "c.fp_attention_p", "seq_len"], [15, 1, 1, "c.fp_attention_p", "softmax_out"]], "fp_attention_s": [[15, 1, 1, "c.fp_attention_s", "K"], [15, 1, 1, "c.fp_attention_s", "Q"], [15, 1, 1, "c.fp_attention_s", "QK"], [15, 1, 1, "c.fp_attention_s", "V"], [15, 1, 1, "c.fp_attention_s", "batch_size"], [15, 1, 1, "c.fp_attention_s", "core_mask"], [15, 1, 1, "c.fp_attention_s", "head_dim"], [15, 1, 1, "c.fp_attention_s", "head_num"], [15, 1, 1, "c.fp_attention_s", "output"], [15, 1, 1, "c.fp_attention_s", "seq_len"], [15, 1, 1, "c.fp_attention_s", "softmax_out"]], "fp_avgpoolinggrad_p": [[16, 1, 1, "c.fp_avgpoolinggrad_p", "batch"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "channel"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "end_idx"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "input"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "input_h"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "input_w"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "output"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "output_h"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "output_w"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "pad_l"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "pad_u"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "start_idx"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "stride_h"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "stride_w"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "window_h"], [16, 1, 1, "c.fp_avgpoolinggrad_p", "window_w"]], "fp_avgpoolinggrad_s": [[16, 1, 1, "c.fp_avgpoolinggrad_s", "batch"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "channel"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "core_mask"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "input"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "input_h"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "input_w"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "output"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "output_h"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "output_w"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "pad_l"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "pad_u"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "stride_h"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "stride_w"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "window_h"], [16, 1, 1, "c.fp_avgpoolinggrad_s", "window_w"]], "fp_batchtospace_p": [[17, 1, 1, "c.fp_batchtospace_p", "block_size"], [17, 1, 1, "c.fp_batchtospace_p", "crops"], [17, 1, 1, "c.fp_batchtospace_p", "data_size"], [17, 1, 1, "c.fp_batchtospace_p", "input"], [17, 1, 1, "c.fp_batchtospace_p", "input_shape"], [17, 1, 1, "c.fp_batchtospace_p", "output"]], "fp_batchtospace_s": [[17, 1, 1, "c.fp_batchtospace_s", "block_size"], [17, 1, 1, "c.fp_batchtospace_s", "core_mask"], [17, 1, 1, "c.fp_batchtospace_s", "crops"], [17, 1, 1, "c.fp_batchtospace_s", "data_size"], [17, 1, 1, "c.fp_batchtospace_s", "input"], [17, 1, 1, "c.fp_batchtospace_s", "input_shape"], [17, 1, 1, "c.fp_batchtospace_s", "output"]], "fp_batchtospacend_p": [[18, 1, 1, "c.fp_batchtospacend_p", "block_size"], [18, 1, 1, "c.fp_batchtospacend_p", "crops"], [18, 1, 1, "c.fp_batchtospacend_p", "data_size"], [18, 1, 1, "c.fp_batchtospacend_p", "input"], [18, 1, 1, "c.fp_batchtospacend_p", "input_shape"], [18, 1, 1, "c.fp_batchtospacend_p", "output"]], "fp_batchtospacend_s": [[18, 1, 1, "c.fp_batchtospacend_s", "block_size"], [18, 1, 1, "c.fp_batchtospacend_s", "core_mask"], [18, 1, 1, "c.fp_batchtospacend_s", "crops"], [18, 1, 1, "c.fp_batchtospacend_s", "data_size"], [18, 1, 1, "c.fp_batchtospacend_s", "input"], [18, 1, 1, "c.fp_batchtospacend_s", "input_shape"], [18, 1, 1, "c.fp_batchtospacend_s", "output"]], "fp_broadcastto_p": [[19, 1, 1, "c.fp_broadcastto_p", "data_size"], [19, 1, 1, "c.fp_broadcastto_p", "input"], [19, 1, 1, "c.fp_broadcastto_p", "input_shape"], [19, 1, 1, "c.fp_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.fp_broadcastto_p", "output"], [19, 1, 1, "c.fp_broadcastto_p", "output_shape"], [19, 1, 1, "c.fp_broadcastto_p", "output_shape_size"]], "fp_broadcastto_s": [[19, 1, 1, "c.fp_broadcastto_s", "core_mask"], [19, 1, 1, "c.fp_broadcastto_s", "data_size"], [19, 1, 1, "c.fp_broadcastto_s", "input"], [19, 1, 1, "c.fp_broadcastto_s", "input_shape"], [19, 1, 1, "c.fp_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.fp_broadcastto_s", "output"], [19, 1, 1, "c.fp_broadcastto_s", "output_shape"], [19, 1, 1, "c.fp_broadcastto_s", "output_shape_size"]], "fp_celu_p": [[10, 1, 1, "c.fp_celu_p", "Input0"], [10, 1, 1, "c.fp_celu_p", "alpha"], [10, 1, 1, "c.fp_celu_p", "length"], [10, 1, 1, "c.fp_celu_p", "output"]], "fp_celu_s": [[10, 1, 1, "c.fp_celu_s", "Input0"], [10, 1, 1, "c.fp_celu_s", "alpha"], [10, 1, 1, "c.fp_celu_s", "core_mask"], [10, 1, 1, "c.fp_celu_s", "length"], [10, 1, 1, "c.fp_celu_s", "output"]], "fp_clip_p": [[10, 1, 1, "c.fp_clip_p", "Input0"], [10, 1, 1, "c.fp_clip_p", "length"], [10, 1, 1, "c.fp_clip_p", "max_val"], [10, 1, 1, "c.fp_clip_p", "min_val"], [10, 1, 1, "c.fp_clip_p", "output"]], "fp_clip_s": [[10, 1, 1, "c.fp_clip_s", "Input0"], [10, 1, 1, "c.fp_clip_s", "core_mask"], [10, 1, 1, "c.fp_clip_s", "length"], [10, 1, 1, "c.fp_clip_s", "max_val"], [10, 1, 1, "c.fp_clip_s", "min_val"], [10, 1, 1, "c.fp_clip_s", "output"]], "fp_conv2d_p": [[20, 1, 1, "c.fp_conv2d_p", "bias"], [20, 1, 1, "c.fp_conv2d_p", "conv_param"], [20, 1, 1, "c.fp_conv2d_p", "core_mask"], [20, 1, 1, "c.fp_conv2d_p", "input_w"], [20, 1, 1, "c.fp_conv2d_p", "input_x"], [20, 1, 1, "c.fp_conv2d_p", "out_y"]], "fp_conv2d_s": [[20, 1, 1, "c.fp_conv2d_s", "bias"], [20, 1, 1, "c.fp_conv2d_s", "conv_param"], [20, 1, 1, "c.fp_conv2d_s", "core_mask"], [20, 1, 1, "c.fp_conv2d_s", "input_w"], [20, 1, 1, "c.fp_conv2d_s", "input_x"], [20, 1, 1, "c.fp_conv2d_s", "out_y"]], "fp_conv2dbackpropfilterfusion_p": [[22, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "conv_param"], [22, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "dw"], [22, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "dy"], [22, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "x"]], "fp_conv2dbackpropfilterfusion_s": [[22, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "conv_param"], [22, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "core_mask"], [22, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "dw"], [22, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "dy"], [22, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "x"]], "fp_conv2dbackpropinputfusion_p": [[23, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "conv_param"], [23, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "dx"], [23, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "dy"], [23, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "w"]], "fp_conv2dbackpropinputfusion_s": [[23, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "conv_param"], [23, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "core_mask"], [23, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "dx"], [23, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "dy"], [23, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "w"]], "fp_convtranspose_p": [[21, 1, 1, "c.fp_convtranspose_p", "bias"], [21, 1, 1, "c.fp_convtranspose_p", "conv_param"], [21, 1, 1, "c.fp_convtranspose_p", "core_mask"], [21, 1, 1, "c.fp_convtranspose_p", "input_w"], [21, 1, 1, "c.fp_convtranspose_p", "input_x"], [21, 1, 1, "c.fp_convtranspose_p", "out_y"]], "fp_convtranspose_s": [[21, 1, 1, "c.fp_convtranspose_s", "bias"], [21, 1, 1, "c.fp_convtranspose_s", "conv_param"], [21, 1, 1, "c.fp_convtranspose_s", "core_mask"], [21, 1, 1, "c.fp_convtranspose_s", "input_w"], [21, 1, 1, "c.fp_convtranspose_s", "input_x"], [21, 1, 1, "c.fp_convtranspose_s", "out_y"]], "fp_crop_and_resize_anycore": [[25, 1, 1, "c.fp_crop_and_resize_anycore", "box_idx"], [25, 1, 1, "c.fp_crop_and_resize_anycore", "boxes"], [25, 1, 1, "c.fp_crop_and_resize_anycore", "core_mask"], [25, 1, 1, "c.fp_crop_and_resize_anycore", "dst"], [25, 1, 1, "c.fp_crop_and_resize_anycore", "extrapolation_value"], [25, 1, 1, "c.fp_crop_and_resize_anycore", "param"], [25, 1, 1, "c.fp_crop_and_resize_anycore", "src"]], "fp_depthtospace_p": [[26, 1, 1, "c.fp_depthtospace_p", "block_size"], [26, 1, 1, "c.fp_depthtospace_p", "data_size"], [26, 1, 1, "c.fp_depthtospace_p", "in_shape"], [26, 1, 1, "c.fp_depthtospace_p", "input"], [26, 1, 1, "c.fp_depthtospace_p", "output"]], "fp_depthtospace_s": [[26, 1, 1, "c.fp_depthtospace_s", "block_size"], [26, 1, 1, "c.fp_depthtospace_s", "core_mask"], [26, 1, 1, "c.fp_depthtospace_s", "data_size"], [26, 1, 1, "c.fp_depthtospace_s", "in_shape"], [26, 1, 1, "c.fp_depthtospace_s", "input"], [26, 1, 1, "c.fp_depthtospace_s", "output"]], "fp_eltwise_p": [[28, 1, 1, "c.fp_eltwise_p", "Input0"], [28, 1, 1, "c.fp_eltwise_p", "Input1"], [28, 1, 1, "c.fp_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.fp_eltwise_p", "length"], [28, 1, 1, "c.fp_eltwise_p", "output"]], "fp_eltwise_s": [[28, 1, 1, "c.fp_eltwise_s", "Input0"], [28, 1, 1, "c.fp_eltwise_s", "Input1"], [28, 1, 1, "c.fp_eltwise_s", "core_mask"], [28, 1, 1, "c.fp_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.fp_eltwise_s", "length"], [28, 1, 1, "c.fp_eltwise_s", "output"]], "fp_elu_p": [[10, 1, 1, "c.fp_elu_p", "Input0"], [10, 1, 1, "c.fp_elu_p", "alpha"], [10, 1, 1, "c.fp_elu_p", "length"], [10, 1, 1, "c.fp_elu_p", "output"]], "fp_elu_s": [[10, 1, 1, "c.fp_elu_s", "Input0"], [10, 1, 1, "c.fp_elu_s", "alpha"], [10, 1, 1, "c.fp_elu_s", "core_mask"], [10, 1, 1, "c.fp_elu_s", "length"], [10, 1, 1, "c.fp_elu_s", "output"]], "fp_embeddinglookup_p": [[29, 1, 1, "c.fp_embeddinglookup_p", "Input0"], [29, 1, 1, "c.fp_embeddinglookup_p", "Input1"], [29, 1, 1, "c.fp_embeddinglookup_p", "length"], [29, 1, 1, "c.fp_embeddinglookup_p", "output"]], "fp_embeddinglookup_s": [[29, 1, 1, "c.fp_embeddinglookup_s", "core_mask"], [29, 1, 1, "c.fp_embeddinglookup_s", "ids"], [29, 1, 1, "c.fp_embeddinglookup_s", "ids_size_"], [29, 1, 1, "c.fp_embeddinglookup_s", "input_data"], [29, 1, 1, "c.fp_embeddinglookup_s", "is_regulated"], [29, 1, 1, "c.fp_embeddinglookup_s", "layer_num_"], [29, 1, 1, "c.fp_embeddinglookup_s", "layer_size_"], [29, 1, 1, "c.fp_embeddinglookup_s", "max_norm_"], [29, 1, 1, "c.fp_embeddinglookup_s", "output"]], "fp_equal_p": [[30, 1, 1, "c.fp_equal_p", "Input0"], [30, 1, 1, "c.fp_equal_p", "Input1"], [30, 1, 1, "c.fp_equal_p", "length"], [30, 1, 1, "c.fp_equal_p", "output"]], "fp_equal_s": [[30, 1, 1, "c.fp_equal_s", "Input0"], [30, 1, 1, "c.fp_equal_s", "Input1"], [30, 1, 1, "c.fp_equal_s", "core_mask"], [30, 1, 1, "c.fp_equal_s", "length"], [30, 1, 1, "c.fp_equal_s", "output"]], "fp_expfusion_p": [[32, 1, 1, "c.fp_expfusion_p", "dst_data"], [32, 1, 1, "c.fp_expfusion_p", "in_scale"], [32, 1, 1, "c.fp_expfusion_p", "length"], [32, 1, 1, "c.fp_expfusion_p", "out_scale"], [32, 1, 1, "c.fp_expfusion_p", "scale"], [32, 1, 1, "c.fp_expfusion_p", "src_data"]], "fp_expfusion_s": [[32, 1, 1, "c.fp_expfusion_s", "core_mask"], [32, 1, 1, "c.fp_expfusion_s", "dst_data"], [32, 1, 1, "c.fp_expfusion_s", "in_scale"], [32, 1, 1, "c.fp_expfusion_s", "length"], [32, 1, 1, "c.fp_expfusion_s", "out_scale"], [32, 1, 1, "c.fp_expfusion_s", "scale"], [32, 1, 1, "c.fp_expfusion_s", "src_data"]], "fp_floor_p": [[34, 1, 1, "c.fp_floor_p", "dst_data"], [34, 1, 1, "c.fp_floor_p", "length"], [34, 1, 1, "c.fp_floor_p", "src_data"]], "fp_floor_s": [[34, 1, 1, "c.fp_floor_s", "core_mask"], [34, 1, 1, "c.fp_floor_s", "dst_data"], [34, 1, 1, "c.fp_floor_s", "length"], [34, 1, 1, "c.fp_floor_s", "src_data"]], "fp_floordiv_p": [[35, 1, 1, "c.fp_floordiv_p", "dst_data"], [35, 1, 1, "c.fp_floordiv_p", "length"], [35, 1, 1, "c.fp_floordiv_p", "src_data0"], [35, 1, 1, "c.fp_floordiv_p", "src_data1"]], "fp_floordiv_s": [[35, 1, 1, "c.fp_floordiv_s", "core_mask"], [35, 1, 1, "c.fp_floordiv_s", "dst_data"], [35, 1, 1, "c.fp_floordiv_s", "length"], [35, 1, 1, "c.fp_floordiv_s", "src_data0"], [35, 1, 1, "c.fp_floordiv_s", "src_data1"]], "fp_fusedbatchnorm_p": [[36, 1, 1, "c.fp_fusedbatchnorm_p", "channel"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "epsilon"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "input"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "mean"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "offset"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "output"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "scale"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "unit"], [36, 1, 1, "c.fp_fusedbatchnorm_p", "variance"]], "fp_fusedbatchnorm_s": [[36, 1, 1, "c.fp_fusedbatchnorm_s", "channel"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "core_mask"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "epsilon"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "input"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "mean"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "offset"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "output"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "scale"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "unit"], [36, 1, 1, "c.fp_fusedbatchnorm_s", "variance"]], "fp_gelu_p": [[10, 1, 1, "c.fp_gelu_p", "Input0"], [10, 1, 1, "c.fp_gelu_p", "approximate"], [10, 1, 1, "c.fp_gelu_p", "length"], [10, 1, 1, "c.fp_gelu_p", "output"]], "fp_gelu_s": [[10, 1, 1, "c.fp_gelu_s", "Input0"], [10, 1, 1, "c.fp_gelu_s", "approximate"], [10, 1, 1, "c.fp_gelu_s", "core_mask"], [10, 1, 1, "c.fp_gelu_s", "length"], [10, 1, 1, "c.fp_gelu_s", "output"]], "fp_groupnormfusion_p": [[37, 1, 1, "c.fp_groupnormfusion_p", "batch"], [37, 1, 1, "c.fp_groupnormfusion_p", "channel"], [37, 1, 1, "c.fp_groupnormfusion_p", "epsilon"], [37, 1, 1, "c.fp_groupnormfusion_p", "input"], [37, 1, 1, "c.fp_groupnormfusion_p", "mean"], [37, 1, 1, "c.fp_groupnormfusion_p", "num_groups"], [37, 1, 1, "c.fp_groupnormfusion_p", "offset"], [37, 1, 1, "c.fp_groupnormfusion_p", "output"], [37, 1, 1, "c.fp_groupnormfusion_p", "scale"], [37, 1, 1, "c.fp_groupnormfusion_p", "unit"], [37, 1, 1, "c.fp_groupnormfusion_p", "variance"]], "fp_groupnormfusion_s": [[37, 1, 1, "c.fp_groupnormfusion_s", "batch"], [37, 1, 1, "c.fp_groupnormfusion_s", "channel"], [37, 1, 1, "c.fp_groupnormfusion_s", "core_mask"], [37, 1, 1, "c.fp_groupnormfusion_s", "epsilon"], [37, 1, 1, "c.fp_groupnormfusion_s", "input"], [37, 1, 1, "c.fp_groupnormfusion_s", "mean"], [37, 1, 1, "c.fp_groupnormfusion_s", "num_groups"], [37, 1, 1, "c.fp_groupnormfusion_s", "offset"], [37, 1, 1, "c.fp_groupnormfusion_s", "output"], [37, 1, 1, "c.fp_groupnormfusion_s", "scale"], [37, 1, 1, "c.fp_groupnormfusion_s", "unit"], [37, 1, 1, "c.fp_groupnormfusion_s", "variance"]], "fp_hardshrink_p": [[10, 1, 1, "c.fp_hardshrink_p", "Input0"], [10, 1, 1, "c.fp_hardshrink_p", "lambd"], [10, 1, 1, "c.fp_hardshrink_p", "length"], [10, 1, 1, "c.fp_hardshrink_p", "output"]], "fp_hardshrink_s": [[10, 1, 1, "c.fp_hardshrink_s", "Input0"], [10, 1, 1, "c.fp_hardshrink_s", "core_mask"], [10, 1, 1, "c.fp_hardshrink_s", "lambd"], [10, 1, 1, "c.fp_hardshrink_s", "length"], [10, 1, 1, "c.fp_hardshrink_s", "output"]], "fp_hardtanh_p": [[10, 1, 1, "c.fp_hardtanh_p", "Input0"], [10, 1, 1, "c.fp_hardtanh_p", "length"], [10, 1, 1, "c.fp_hardtanh_p", "max_val"], [10, 1, 1, "c.fp_hardtanh_p", "min_val"], [10, 1, 1, "c.fp_hardtanh_p", "output"]], "fp_hardtanh_s": [[10, 1, 1, "c.fp_hardtanh_s", "Input0"], [10, 1, 1, "c.fp_hardtanh_s", "core_mask"], [10, 1, 1, "c.fp_hardtanh_s", "length"], [10, 1, 1, "c.fp_hardtanh_s", "max_val"], [10, 1, 1, "c.fp_hardtanh_s", "min_val"], [10, 1, 1, "c.fp_hardtanh_s", "output"]], "fp_hsigmoid_p": [[10, 1, 1, "c.fp_hsigmoid_p", "Input0"], [10, 1, 1, "c.fp_hsigmoid_p", "length"], [10, 1, 1, "c.fp_hsigmoid_p", "output"]], "fp_hsigmoid_s": [[10, 1, 1, "c.fp_hsigmoid_s", "Input0"], [10, 1, 1, "c.fp_hsigmoid_s", "core_mask"], [10, 1, 1, "c.fp_hsigmoid_s", "length"], [10, 1, 1, "c.fp_hsigmoid_s", "output"]], "fp_hswish_p": [[10, 1, 1, "c.fp_hswish_p", "Input0"], [10, 1, 1, "c.fp_hswish_p", "length"], [10, 1, 1, "c.fp_hswish_p", "output"]], "fp_hswish_s": [[10, 1, 1, "c.fp_hswish_s", "Input0"], [10, 1, 1, "c.fp_hswish_s", "core_mask"], [10, 1, 1, "c.fp_hswish_s", "length"], [10, 1, 1, "c.fp_hswish_s", "output"]], "fp_leaky_relu_p": [[39, 1, 1, "c.fp_leaky_relu_p", "alpha"], [39, 1, 1, "c.fp_leaky_relu_p", "core_mask"], [39, 1, 1, "c.fp_leaky_relu_p", "elem_cnt"], [39, 1, 1, "c.fp_leaky_relu_p", "input"], [39, 1, 1, "c.fp_leaky_relu_p", "output"]], "fp_leaky_relu_s": [[39, 1, 1, "c.fp_leaky_relu_s", "alpha"], [39, 1, 1, "c.fp_leaky_relu_s", "core_mask"], [39, 1, 1, "c.fp_leaky_relu_s", "elem_cnt"], [39, 1, 1, "c.fp_leaky_relu_s", "input"], [39, 1, 1, "c.fp_leaky_relu_s", "output"]], "fp_linspace_p": [[40, 1, 1, "c.fp_linspace_p", "num"], [40, 1, 1, "c.fp_linspace_p", "output"], [40, 1, 1, "c.fp_linspace_p", "start"], [40, 1, 1, "c.fp_linspace_p", "step"]], "fp_linspace_s": [[40, 1, 1, "c.fp_linspace_s", "core_mask"], [40, 1, 1, "c.fp_linspace_s", "end"], [40, 1, 1, "c.fp_linspace_s", "length"], [40, 1, 1, "c.fp_linspace_s", "output"], [40, 1, 1, "c.fp_linspace_s", "start"]], "fp_lrelu_p": [[10, 1, 1, "c.fp_lrelu_p", "Input0"], [10, 1, 1, "c.fp_lrelu_p", "alpha"], [10, 1, 1, "c.fp_lrelu_p", "length"], [10, 1, 1, "c.fp_lrelu_p", "output"]], "fp_lrelu_s": [[10, 1, 1, "c.fp_lrelu_s", "Input0"], [10, 1, 1, "c.fp_lrelu_s", "alpha"], [10, 1, 1, "c.fp_lrelu_s", "core_mask"], [10, 1, 1, "c.fp_lrelu_s", "length"], [10, 1, 1, "c.fp_lrelu_s", "output"]], "fp_matmulfusion_p": [[42, 1, 1, "c.fp_matmulfusion_p", "A"], [42, 1, 1, "c.fp_matmulfusion_p", "B"], [42, 1, 1, "c.fp_matmulfusion_p", "C"], [42, 1, 1, "c.fp_matmulfusion_p", "K"], [42, 1, 1, "c.fp_matmulfusion_p", "M"], [42, 1, 1, "c.fp_matmulfusion_p", "N"], [42, 1, 1, "c.fp_matmulfusion_p", "activation_type"], [42, 1, 1, "c.fp_matmulfusion_p", "bias"]], "fp_matmulfusion_s": [[42, 1, 1, "c.fp_matmulfusion_s", "A"], [42, 1, 1, "c.fp_matmulfusion_s", "B"], [42, 1, 1, "c.fp_matmulfusion_s", "C"], [42, 1, 1, "c.fp_matmulfusion_s", "K"], [42, 1, 1, "c.fp_matmulfusion_s", "M"], [42, 1, 1, "c.fp_matmulfusion_s", "N"], [42, 1, 1, "c.fp_matmulfusion_s", "activation_type"], [42, 1, 1, "c.fp_matmulfusion_s", "bias"], [42, 1, 1, "c.fp_matmulfusion_s", "core_mask"]], "fp_raggedrange_p": [[43, 1, 1, "c.fp_raggedrange_p", "deltas"], [43, 1, 1, "c.fp_raggedrange_p", "limits"], [43, 1, 1, "c.fp_raggedrange_p", "range_count"], [43, 1, 1, "c.fp_raggedrange_p", "splits"], [43, 1, 1, "c.fp_raggedrange_p", "starts"], [43, 1, 1, "c.fp_raggedrange_p", "values"]], "fp_raggedrange_s": [[43, 1, 1, "c.fp_raggedrange_s", "core_mask"], [43, 1, 1, "c.fp_raggedrange_s", "deltas"], [43, 1, 1, "c.fp_raggedrange_s", "limits"], [43, 1, 1, "c.fp_raggedrange_s", "range_count"], [43, 1, 1, "c.fp_raggedrange_s", "splits"], [43, 1, 1, "c.fp_raggedrange_s", "starts"], [43, 1, 1, "c.fp_raggedrange_s", "values"]], "fp_range_p": [[44, 1, 1, "c.fp_range_p", "delta"], [44, 1, 1, "c.fp_range_p", "length"], [44, 1, 1, "c.fp_range_p", "output"], [44, 1, 1, "c.fp_range_p", "start"]], "fp_range_s": [[44, 1, 1, "c.fp_range_s", "core_mask"], [44, 1, 1, "c.fp_range_s", "delta"], [44, 1, 1, "c.fp_range_s", "length"], [44, 1, 1, "c.fp_range_s", "output"], [44, 1, 1, "c.fp_range_s", "start"]], "fp_reduce_p": [[45, 1, 1, "c.fp_reduce_p", "core_mask"], [45, 1, 1, "c.fp_reduce_p", "dst_data"], [45, 1, 1, "c.fp_reduce_p", "param"], [45, 1, 1, "c.fp_reduce_p", "src_data"], [45, 1, 1, "c.fp_reduce_p", "tmp_dst_data"], [45, 1, 1, "c.fp_reduce_p", "tmp_src_data"]], "fp_reduce_s": [[45, 1, 1, "c.fp_reduce_s", "core_mask"], [45, 1, 1, "c.fp_reduce_s", "dst_data"], [45, 1, 1, "c.fp_reduce_s", "param"], [45, 1, 1, "c.fp_reduce_s", "src_data"]], "fp_relu6_p": [[10, 1, 1, "c.fp_relu6_p", "Input0"], [10, 1, 1, "c.fp_relu6_p", "length"], [10, 1, 1, "c.fp_relu6_p", "output"]], "fp_relu6_s": [[10, 1, 1, "c.fp_relu6_s", "Input0"], [10, 1, 1, "c.fp_relu6_s", "core_mask"], [10, 1, 1, "c.fp_relu6_s", "length"], [10, 1, 1, "c.fp_relu6_s", "output"]], "fp_relu_p": [[10, 1, 1, "c.fp_relu_p", "Input0"], [10, 1, 1, "c.fp_relu_p", "length"], [10, 1, 1, "c.fp_relu_p", "output"]], "fp_relu_s": [[10, 1, 1, "c.fp_relu_s", "Input0"], [10, 1, 1, "c.fp_relu_s", "core_mask"], [10, 1, 1, "c.fp_relu_s", "length"], [10, 1, 1, "c.fp_relu_s", "output"]], "fp_resize_anycore": [[46, 1, 1, "c.fp_resize_anycore", "core_mask"], [46, 1, 1, "c.fp_resize_anycore", "input"], [46, 1, 1, "c.fp_resize_anycore", "output"], [46, 1, 1, "c.fp_resize_anycore", "param"]], "fp_scalefusion_p": [[49, 1, 1, "c.fp_scalefusion_p", "bias"], [49, 1, 1, "c.fp_scalefusion_p", "dst_data"], [49, 1, 1, "c.fp_scalefusion_p", "length"], [49, 1, 1, "c.fp_scalefusion_p", "scale"], [49, 1, 1, "c.fp_scalefusion_p", "src_data"]], "fp_scalefusion_s": [[49, 1, 1, "c.fp_scalefusion_s", "bias"], [49, 1, 1, "c.fp_scalefusion_s", "core_mask"], [49, 1, 1, "c.fp_scalefusion_s", "dst_data"], [49, 1, 1, "c.fp_scalefusion_s", "length"], [49, 1, 1, "c.fp_scalefusion_s", "scale"], [49, 1, 1, "c.fp_scalefusion_s", "src_data"]], "fp_scatter_elements_p": [[50, 1, 1, "c.fp_scatter_elements_p", "core_mask"], [50, 1, 1, "c.fp_scatter_elements_p", "indices"], [50, 1, 1, "c.fp_scatter_elements_p", "input"], [50, 1, 1, "c.fp_scatter_elements_p", "output"], [50, 1, 1, "c.fp_scatter_elements_p", "param"], [50, 1, 1, "c.fp_scatter_elements_p", "updates"]], "fp_scatter_elements_s": [[50, 1, 1, "c.fp_scatter_elements_s", "core_mask"], [50, 1, 1, "c.fp_scatter_elements_s", "indices"], [50, 1, 1, "c.fp_scatter_elements_s", "input"], [50, 1, 1, "c.fp_scatter_elements_s", "output"], [50, 1, 1, "c.fp_scatter_elements_s", "param"], [50, 1, 1, "c.fp_scatter_elements_s", "updates"]], "fp_sgd_p": [[51, 1, 1, "c.fp_sgd_p", "accumulate"], [51, 1, 1, "c.fp_sgd_p", "dampening"], [51, 1, 1, "c.fp_sgd_p", "gradient"], [51, 1, 1, "c.fp_sgd_p", "learning_rate"], [51, 1, 1, "c.fp_sgd_p", "length"], [51, 1, 1, "c.fp_sgd_p", "moment"], [51, 1, 1, "c.fp_sgd_p", "nesterov"], [51, 1, 1, "c.fp_sgd_p", "weight"], [51, 1, 1, "c.fp_sgd_p", "weight_decay"]], "fp_sgd_s": [[51, 1, 1, "c.fp_sgd_s", "accumulate"], [51, 1, 1, "c.fp_sgd_s", "core_mask"], [51, 1, 1, "c.fp_sgd_s", "dampening"], [51, 1, 1, "c.fp_sgd_s", "end"], [51, 1, 1, "c.fp_sgd_s", "gradient"], [51, 1, 1, "c.fp_sgd_s", "learning_rate"], [51, 1, 1, "c.fp_sgd_s", "moment"], [51, 1, 1, "c.fp_sgd_s", "nesterov"], [51, 1, 1, "c.fp_sgd_s", "start"], [51, 1, 1, "c.fp_sgd_s", "weight"], [51, 1, 1, "c.fp_sgd_s", "weight_decay"]], "fp_sigmoid_p": [[10, 1, 1, "c.fp_sigmoid_p", "Input0"], [10, 1, 1, "c.fp_sigmoid_p", "length"], [10, 1, 1, "c.fp_sigmoid_p", "output"]], "fp_sigmoid_s": [[10, 1, 1, "c.fp_sigmoid_s", "Input0"], [10, 1, 1, "c.fp_sigmoid_s", "core_mask"], [10, 1, 1, "c.fp_sigmoid_s", "length"], [10, 1, 1, "c.fp_sigmoid_s", "output"]], "fp_softplus_p": [[10, 1, 1, "c.fp_softplus_p", "Input0"], [10, 1, 1, "c.fp_softplus_p", "length"], [10, 1, 1, "c.fp_softplus_p", "output"]], "fp_softplus_s": [[10, 1, 1, "c.fp_softplus_s", "Input0"], [10, 1, 1, "c.fp_softplus_s", "core_mask"], [10, 1, 1, "c.fp_softplus_s", "length"], [10, 1, 1, "c.fp_softplus_s", "output"]], "fp_softshrink_p": [[10, 1, 1, "c.fp_softshrink_p", "Input0"], [10, 1, 1, "c.fp_softshrink_p", "lambd"], [10, 1, 1, "c.fp_softshrink_p", "length"], [10, 1, 1, "c.fp_softshrink_p", "output"]], "fp_softshrink_s": [[10, 1, 1, "c.fp_softshrink_s", "Input0"], [10, 1, 1, "c.fp_softshrink_s", "core_mask"], [10, 1, 1, "c.fp_softshrink_s", "lambd"], [10, 1, 1, "c.fp_softshrink_s", "length"], [10, 1, 1, "c.fp_softshrink_s", "output"]], "fp_softsignopt_p": [[10, 1, 1, "c.fp_softsignopt_p", "Input0"], [10, 1, 1, "c.fp_softsignopt_p", "length"], [10, 1, 1, "c.fp_softsignopt_p", "output"]], "fp_softsignopt_s": [[10, 1, 1, "c.fp_softsignopt_s", "Input0"], [10, 1, 1, "c.fp_softsignopt_s", "core_mask"], [10, 1, 1, "c.fp_softsignopt_s", "length"], [10, 1, 1, "c.fp_softsignopt_s", "output"]], "fp_spacetobatch_p": [[52, 1, 1, "c.fp_spacetobatch_p", "block_size"], [52, 1, 1, "c.fp_spacetobatch_p", "data_size"], [52, 1, 1, "c.fp_spacetobatch_p", "input"], [52, 1, 1, "c.fp_spacetobatch_p", "input_shape"], [52, 1, 1, "c.fp_spacetobatch_p", "output"], [52, 1, 1, "c.fp_spacetobatch_p", "paddings"]], "fp_spacetobatch_s": [[52, 1, 1, "c.fp_spacetobatch_s", "block_size"], [52, 1, 1, "c.fp_spacetobatch_s", "core_mask"], [52, 1, 1, "c.fp_spacetobatch_s", "data_size"], [52, 1, 1, "c.fp_spacetobatch_s", "input"], [52, 1, 1, "c.fp_spacetobatch_s", "input_shape"], [52, 1, 1, "c.fp_spacetobatch_s", "output"], [52, 1, 1, "c.fp_spacetobatch_s", "paddings"]], "fp_spacetobatchnd_p": [[53, 1, 1, "c.fp_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.fp_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.fp_spacetobatchnd_p", "input"], [53, 1, 1, "c.fp_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.fp_spacetobatchnd_p", "output"], [53, 1, 1, "c.fp_spacetobatchnd_p", "paddings"]], "fp_spacetobatchnd_s": [[53, 1, 1, "c.fp_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.fp_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.fp_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.fp_spacetobatchnd_s", "input"], [53, 1, 1, "c.fp_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.fp_spacetobatchnd_s", "output"], [53, 1, 1, "c.fp_spacetobatchnd_s", "paddings"]], "fp_spacetodepth_p": [[54, 1, 1, "c.fp_spacetodepth_p", "block"], [54, 1, 1, "c.fp_spacetodepth_p", "data_size"], [54, 1, 1, "c.fp_spacetodepth_p", "in_shape"], [54, 1, 1, "c.fp_spacetodepth_p", "input"], [54, 1, 1, "c.fp_spacetodepth_p", "output"]], "fp_spacetodepth_s": [[54, 1, 1, "c.fp_spacetodepth_s", "block"], [54, 1, 1, "c.fp_spacetodepth_s", "core_mask"], [54, 1, 1, "c.fp_spacetodepth_s", "data_size"], [54, 1, 1, "c.fp_spacetodepth_s", "in_shape"], [54, 1, 1, "c.fp_spacetodepth_s", "input"], [54, 1, 1, "c.fp_spacetodepth_s", "output"]], "fp_swish_p": [[10, 1, 1, "c.fp_swish_p", "Input0"], [10, 1, 1, "c.fp_swish_p", "length"], [10, 1, 1, "c.fp_swish_p", "output"]], "fp_swish_s": [[10, 1, 1, "c.fp_swish_s", "Input0"], [10, 1, 1, "c.fp_swish_s", "core_mask"], [10, 1, 1, "c.fp_swish_s", "length"], [10, 1, 1, "c.fp_swish_s", "output"]], "fp_tanh_p": [[10, 1, 1, "c.fp_tanh_p", "Input0"], [10, 1, 1, "c.fp_tanh_p", "length"], [10, 1, 1, "c.fp_tanh_p", "output"]], "fp_tanh_s": [[10, 1, 1, "c.fp_tanh_s", "Input0"], [10, 1, 1, "c.fp_tanh_s", "core_mask"], [10, 1, 1, "c.fp_tanh_s", "length"], [10, 1, 1, "c.fp_tanh_s", "output"]], "hp_Gru_p": [[38, 1, 1, "c.hp_Gru_p", "buffer"], [38, 1, 1, "c.hp_Gru_p", "core_mask"], [38, 1, 1, "c.hp_Gru_p", "gru_param"], [38, 1, 1, "c.hp_Gru_p", "hidden_state"], [38, 1, 1, "c.hp_Gru_p", "input"], [38, 1, 1, "c.hp_Gru_p", "input_bias"], [38, 1, 1, "c.hp_Gru_p", "output"], [38, 1, 1, "c.hp_Gru_p", "state_bias"], [38, 1, 1, "c.hp_Gru_p", "weight_g"], [38, 1, 1, "c.hp_Gru_p", "weight_r"]], "hp_Gru_s": [[38, 1, 1, "c.hp_Gru_s", "buffer"], [38, 1, 1, "c.hp_Gru_s", "core_mask"], [38, 1, 1, "c.hp_Gru_s", "gru_param"], [38, 1, 1, "c.hp_Gru_s", "hidden_state"], [38, 1, 1, "c.hp_Gru_s", "input"], [38, 1, 1, "c.hp_Gru_s", "input_bias"], [38, 1, 1, "c.hp_Gru_s", "output"], [38, 1, 1, "c.hp_Gru_s", "state_bias"], [38, 1, 1, "c.hp_Gru_s", "weight_g"], [38, 1, 1, "c.hp_Gru_s", "weight_r"]], "hp_adamweightdecay_p": [[11, 1, 1, "c.hp_adamweightdecay_p", "beta1"], [11, 1, 1, "c.hp_adamweightdecay_p", "beta2"], [11, 1, 1, "c.hp_adamweightdecay_p", "decay"], [11, 1, 1, "c.hp_adamweightdecay_p", "epsilon"], [11, 1, 1, "c.hp_adamweightdecay_p", "gradient"], [11, 1, 1, "c.hp_adamweightdecay_p", "length"], [11, 1, 1, "c.hp_adamweightdecay_p", "lr"], [11, 1, 1, "c.hp_adamweightdecay_p", "m"], [11, 1, 1, "c.hp_adamweightdecay_p", "v"], [11, 1, 1, "c.hp_adamweightdecay_p", "var"]], "hp_adamweightdecay_s": [[11, 1, 1, "c.hp_adamweightdecay_s", "beta1"], [11, 1, 1, "c.hp_adamweightdecay_s", "beta2"], [11, 1, 1, "c.hp_adamweightdecay_s", "core_mask"], [11, 1, 1, "c.hp_adamweightdecay_s", "decay"], [11, 1, 1, "c.hp_adamweightdecay_s", "end"], [11, 1, 1, "c.hp_adamweightdecay_s", "epsilon"], [11, 1, 1, "c.hp_adamweightdecay_s", "gradient"], [11, 1, 1, "c.hp_adamweightdecay_s", "lr"], [11, 1, 1, "c.hp_adamweightdecay_s", "m"], [11, 1, 1, "c.hp_adamweightdecay_s", "start"], [11, 1, 1, "c.hp_adamweightdecay_s", "v"], [11, 1, 1, "c.hp_adamweightdecay_s", "var"]], "hp_adder_p": [[12, 1, 1, "c.hp_adder_p", "bias"], [12, 1, 1, "c.hp_adder_p", "conv_param"], [12, 1, 1, "c.hp_adder_p", "core_mask"], [12, 1, 1, "c.hp_adder_p", "input_w"], [12, 1, 1, "c.hp_adder_p", "input_x"], [12, 1, 1, "c.hp_adder_p", "out_y"]], "hp_adder_s": [[12, 1, 1, "c.hp_adder_s", "bias"], [12, 1, 1, "c.hp_adder_s", "core_mask"], [12, 1, 1, "c.hp_adder_s", "input_w"], [12, 1, 1, "c.hp_adder_s", "input_x"], [12, 1, 1, "c.hp_adder_s", "out_y"], [12, 1, 1, "c.hp_adder_s", "param"]], "hp_applymomentum_p": [[13, 1, 1, "c.hp_applymomentum_p", "accumulate"], [13, 1, 1, "c.hp_applymomentum_p", "gradient"], [13, 1, 1, "c.hp_applymomentum_p", "learning_rate"], [13, 1, 1, "c.hp_applymomentum_p", "length"], [13, 1, 1, "c.hp_applymomentum_p", "moment"], [13, 1, 1, "c.hp_applymomentum_p", "nesterov"], [13, 1, 1, "c.hp_applymomentum_p", "weight"]], "hp_applymomentum_s": [[13, 1, 1, "c.hp_applymomentum_s", "accumulate"], [13, 1, 1, "c.hp_applymomentum_s", "core_mask"], [13, 1, 1, "c.hp_applymomentum_s", "end"], [13, 1, 1, "c.hp_applymomentum_s", "gradient"], [13, 1, 1, "c.hp_applymomentum_s", "learning_rate"], [13, 1, 1, "c.hp_applymomentum_s", "moment"], [13, 1, 1, "c.hp_applymomentum_s", "nesterov"], [13, 1, 1, "c.hp_applymomentum_s", "start"], [13, 1, 1, "c.hp_applymomentum_s", "weight"]], "hp_avgpoolinggrad_p": [[16, 1, 1, "c.hp_avgpoolinggrad_p", "batch"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "channel"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "end_idx"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "input"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "input_h"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "input_w"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "output"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "output_h"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "output_w"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "pad_l"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "pad_u"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "start_idx"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "stride_h"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "stride_w"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "window_h"], [16, 1, 1, "c.hp_avgpoolinggrad_p", "window_w"]], "hp_avgpoolinggrad_s": [[16, 1, 1, "c.hp_avgpoolinggrad_s", "batch"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "channel"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "core_mask"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "input"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "input_h"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "input_w"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "output"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "output_h"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "output_w"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "pad_l"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "pad_u"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "stride_h"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "stride_w"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "window_h"], [16, 1, 1, "c.hp_avgpoolinggrad_s", "window_w"]], "hp_batchtospace_p": [[17, 1, 1, "c.hp_batchtospace_p", "block_size"], [17, 1, 1, "c.hp_batchtospace_p", "crops"], [17, 1, 1, "c.hp_batchtospace_p", "data_size"], [17, 1, 1, "c.hp_batchtospace_p", "input"], [17, 1, 1, "c.hp_batchtospace_p", "input_shape"], [17, 1, 1, "c.hp_batchtospace_p", "output"]], "hp_batchtospace_s": [[17, 1, 1, "c.hp_batchtospace_s", "block_size"], [17, 1, 1, "c.hp_batchtospace_s", "core_mask"], [17, 1, 1, "c.hp_batchtospace_s", "crops"], [17, 1, 1, "c.hp_batchtospace_s", "data_size"], [17, 1, 1, "c.hp_batchtospace_s", "input"], [17, 1, 1, "c.hp_batchtospace_s", "input_shape"], [17, 1, 1, "c.hp_batchtospace_s", "output"]], "hp_batchtospacend_p": [[18, 1, 1, "c.hp_batchtospacend_p", "block_size"], [18, 1, 1, "c.hp_batchtospacend_p", "crops"], [18, 1, 1, "c.hp_batchtospacend_p", "data_size"], [18, 1, 1, "c.hp_batchtospacend_p", "input"], [18, 1, 1, "c.hp_batchtospacend_p", "input_shape"], [18, 1, 1, "c.hp_batchtospacend_p", "output"]], "hp_batchtospacend_s": [[18, 1, 1, "c.hp_batchtospacend_s", "block_size"], [18, 1, 1, "c.hp_batchtospacend_s", "core_mask"], [18, 1, 1, "c.hp_batchtospacend_s", "crops"], [18, 1, 1, "c.hp_batchtospacend_s", "data_size"], [18, 1, 1, "c.hp_batchtospacend_s", "input"], [18, 1, 1, "c.hp_batchtospacend_s", "input_shape"], [18, 1, 1, "c.hp_batchtospacend_s", "output"]], "hp_broadcastto_p": [[19, 1, 1, "c.hp_broadcastto_p", "data_size"], [19, 1, 1, "c.hp_broadcastto_p", "input"], [19, 1, 1, "c.hp_broadcastto_p", "input_shape"], [19, 1, 1, "c.hp_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.hp_broadcastto_p", "output"], [19, 1, 1, "c.hp_broadcastto_p", "output_shape"], [19, 1, 1, "c.hp_broadcastto_p", "output_shape_size"]], "hp_broadcastto_s": [[19, 1, 1, "c.hp_broadcastto_s", "core_mask"], [19, 1, 1, "c.hp_broadcastto_s", "data_size"], [19, 1, 1, "c.hp_broadcastto_s", "input"], [19, 1, 1, "c.hp_broadcastto_s", "input_shape"], [19, 1, 1, "c.hp_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.hp_broadcastto_s", "output"], [19, 1, 1, "c.hp_broadcastto_s", "output_shape"], [19, 1, 1, "c.hp_broadcastto_s", "output_shape_size"]], "hp_celu_p": [[10, 1, 1, "c.hp_celu_p", "Input0"], [10, 1, 1, "c.hp_celu_p", "alpha"], [10, 1, 1, "c.hp_celu_p", "length"], [10, 1, 1, "c.hp_celu_p", "output"]], "hp_celu_s": [[10, 1, 1, "c.hp_celu_s", "Input0"], [10, 1, 1, "c.hp_celu_s", "alpha"], [10, 1, 1, "c.hp_celu_s", "core_mask"], [10, 1, 1, "c.hp_celu_s", "length"], [10, 1, 1, "c.hp_celu_s", "output"]], "hp_clip_p": [[10, 1, 1, "c.hp_clip_p", "Input0"], [10, 1, 1, "c.hp_clip_p", "length"], [10, 1, 1, "c.hp_clip_p", "max_val"], [10, 1, 1, "c.hp_clip_p", "min_val"], [10, 1, 1, "c.hp_clip_p", "output"]], "hp_clip_s": [[10, 1, 1, "c.hp_clip_s", "Input0"], [10, 1, 1, "c.hp_clip_s", "core_mask"], [10, 1, 1, "c.hp_clip_s", "length"], [10, 1, 1, "c.hp_clip_s", "max_val"], [10, 1, 1, "c.hp_clip_s", "min_val"], [10, 1, 1, "c.hp_clip_s", "output"]], "hp_conv2d_p": [[20, 1, 1, "c.hp_conv2d_p", "bias"], [20, 1, 1, "c.hp_conv2d_p", "conv_param"], [20, 1, 1, "c.hp_conv2d_p", "core_mask"], [20, 1, 1, "c.hp_conv2d_p", "input_w"], [20, 1, 1, "c.hp_conv2d_p", "input_x"], [20, 1, 1, "c.hp_conv2d_p", "out_y"]], "hp_conv2d_s": [[20, 1, 1, "c.hp_conv2d_s", "bias"], [20, 1, 1, "c.hp_conv2d_s", "conv_param"], [20, 1, 1, "c.hp_conv2d_s", "core_mask"], [20, 1, 1, "c.hp_conv2d_s", "input_w"], [20, 1, 1, "c.hp_conv2d_s", "input_x"], [20, 1, 1, "c.hp_conv2d_s", "out_y"]], "hp_conv2dbackpropfilterfusion_p": [[22, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "conv_param"], [22, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "dw"], [22, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "dy"], [22, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "x"]], "hp_conv2dbackpropfilterfusion_s": [[22, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "conv_param"], [22, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "core_mask"], [22, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "dw"], [22, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "dy"], [22, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "x"]], "hp_conv2dbackpropinputfusion_p": [[23, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "conv_param"], [23, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "dx"], [23, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "dy"], [23, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "w"]], "hp_conv2dbackpropinputfusion_s": [[23, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "conv_param"], [23, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "core_mask"], [23, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "dx"], [23, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "dy"], [23, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "w"]], "hp_convtranspose_p": [[21, 1, 1, "c.hp_convtranspose_p", "bias"], [21, 1, 1, "c.hp_convtranspose_p", "conv_param"], [21, 1, 1, "c.hp_convtranspose_p", "core_mask"], [21, 1, 1, "c.hp_convtranspose_p", "input_w"], [21, 1, 1, "c.hp_convtranspose_p", "input_x"], [21, 1, 1, "c.hp_convtranspose_p", "out_y"]], "hp_convtranspose_s": [[21, 1, 1, "c.hp_convtranspose_s", "bias"], [21, 1, 1, "c.hp_convtranspose_s", "conv_param"], [21, 1, 1, "c.hp_convtranspose_s", "core_mask"], [21, 1, 1, "c.hp_convtranspose_s", "input_w"], [21, 1, 1, "c.hp_convtranspose_s", "input_x"], [21, 1, 1, "c.hp_convtranspose_s", "out_y"]], "hp_crop_and_resize_anycore": [[25, 1, 1, "c.hp_crop_and_resize_anycore", "box_idx"], [25, 1, 1, "c.hp_crop_and_resize_anycore", "boxes"], [25, 1, 1, "c.hp_crop_and_resize_anycore", "core_mask"], [25, 1, 1, "c.hp_crop_and_resize_anycore", "dst"], [25, 1, 1, "c.hp_crop_and_resize_anycore", "extrapolation_value"], [25, 1, 1, "c.hp_crop_and_resize_anycore", "param"], [25, 1, 1, "c.hp_crop_and_resize_anycore", "src"]], "hp_depthtospace_p": [[26, 1, 1, "c.hp_depthtospace_p", "block_size"], [26, 1, 1, "c.hp_depthtospace_p", "data_size"], [26, 1, 1, "c.hp_depthtospace_p", "in_shape"], [26, 1, 1, "c.hp_depthtospace_p", "input"], [26, 1, 1, "c.hp_depthtospace_p", "output"]], "hp_depthtospace_s": [[26, 1, 1, "c.hp_depthtospace_s", "block_size"], [26, 1, 1, "c.hp_depthtospace_s", "core_mask"], [26, 1, 1, "c.hp_depthtospace_s", "data_size"], [26, 1, 1, "c.hp_depthtospace_s", "in_shape"], [26, 1, 1, "c.hp_depthtospace_s", "input"], [26, 1, 1, "c.hp_depthtospace_s", "output"]], "hp_eltwise_p": [[28, 1, 1, "c.hp_eltwise_p", "Input0"], [28, 1, 1, "c.hp_eltwise_p", "Input1"], [28, 1, 1, "c.hp_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.hp_eltwise_p", "length"], [28, 1, 1, "c.hp_eltwise_p", "output"]], "hp_eltwise_s": [[28, 1, 1, "c.hp_eltwise_s", "Input0"], [28, 1, 1, "c.hp_eltwise_s", "Input1"], [28, 1, 1, "c.hp_eltwise_s", "core_mask"], [28, 1, 1, "c.hp_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.hp_eltwise_s", "length"], [28, 1, 1, "c.hp_eltwise_s", "output"]], "hp_elu_p": [[10, 1, 1, "c.hp_elu_p", "Input0"], [10, 1, 1, "c.hp_elu_p", "alpha"], [10, 1, 1, "c.hp_elu_p", "length"], [10, 1, 1, "c.hp_elu_p", "output"]], "hp_elu_s": [[10, 1, 1, "c.hp_elu_s", "Input0"], [10, 1, 1, "c.hp_elu_s", "alpha"], [10, 1, 1, "c.hp_elu_s", "core_mask"], [10, 1, 1, "c.hp_elu_s", "length"], [10, 1, 1, "c.hp_elu_s", "output"]], "hp_embeddinglookup_p": [[29, 1, 1, "c.hp_embeddinglookup_p", "Input0"], [29, 1, 1, "c.hp_embeddinglookup_p", "Input1"], [29, 1, 1, "c.hp_embeddinglookup_p", "length"], [29, 1, 1, "c.hp_embeddinglookup_p", "output"]], "hp_embeddinglookup_s": [[29, 1, 1, "c.hp_embeddinglookup_s", "core_mask"], [29, 1, 1, "c.hp_embeddinglookup_s", "ids"], [29, 1, 1, "c.hp_embeddinglookup_s", "ids_size_"], [29, 1, 1, "c.hp_embeddinglookup_s", "input_data"], [29, 1, 1, "c.hp_embeddinglookup_s", "is_regulated"], [29, 1, 1, "c.hp_embeddinglookup_s", "layer_num_"], [29, 1, 1, "c.hp_embeddinglookup_s", "layer_size_"], [29, 1, 1, "c.hp_embeddinglookup_s", "max_norm_"], [29, 1, 1, "c.hp_embeddinglookup_s", "output"]], "hp_equal_p": [[30, 1, 1, "c.hp_equal_p", "Input0"], [30, 1, 1, "c.hp_equal_p", "Input1"], [30, 1, 1, "c.hp_equal_p", "length"], [30, 1, 1, "c.hp_equal_p", "output"]], "hp_equal_s": [[30, 1, 1, "c.hp_equal_s", "Input0"], [30, 1, 1, "c.hp_equal_s", "Input1"], [30, 1, 1, "c.hp_equal_s", "core_mask"], [30, 1, 1, "c.hp_equal_s", "length"], [30, 1, 1, "c.hp_equal_s", "output"]], "hp_expfusion_p": [[32, 1, 1, "c.hp_expfusion_p", "dst_data"], [32, 1, 1, "c.hp_expfusion_p", "in_scale"], [32, 1, 1, "c.hp_expfusion_p", "length"], [32, 1, 1, "c.hp_expfusion_p", "out_scale"], [32, 1, 1, "c.hp_expfusion_p", "scale"], [32, 1, 1, "c.hp_expfusion_p", "src_data"]], "hp_expfusion_s": [[32, 1, 1, "c.hp_expfusion_s", "core_mask"], [32, 1, 1, "c.hp_expfusion_s", "dst_data"], [32, 1, 1, "c.hp_expfusion_s", "in_scale"], [32, 1, 1, "c.hp_expfusion_s", "length"], [32, 1, 1, "c.hp_expfusion_s", "out_scale"], [32, 1, 1, "c.hp_expfusion_s", "scale"], [32, 1, 1, "c.hp_expfusion_s", "src_data"]], "hp_floor_p": [[34, 1, 1, "c.hp_floor_p", "dst_data"], [34, 1, 1, "c.hp_floor_p", "length"], [34, 1, 1, "c.hp_floor_p", "src_data"]], "hp_floor_s": [[34, 1, 1, "c.hp_floor_s", "core_mask"], [34, 1, 1, "c.hp_floor_s", "dst_data"], [34, 1, 1, "c.hp_floor_s", "length"], [34, 1, 1, "c.hp_floor_s", "src_data"]], "hp_floordiv_p": [[35, 1, 1, "c.hp_floordiv_p", "dst_data"], [35, 1, 1, "c.hp_floordiv_p", "length"], [35, 1, 1, "c.hp_floordiv_p", "src_data0"], [35, 1, 1, "c.hp_floordiv_p", "src_data1"]], "hp_floordiv_s": [[35, 1, 1, "c.hp_floordiv_s", "core_mask"], [35, 1, 1, "c.hp_floordiv_s", "dst_data"], [35, 1, 1, "c.hp_floordiv_s", "length"], [35, 1, 1, "c.hp_floordiv_s", "src_data0"], [35, 1, 1, "c.hp_floordiv_s", "src_data1"]], "hp_fusedbatchnorm_p": [[36, 1, 1, "c.hp_fusedbatchnorm_p", "channel"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "epsilon"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "input"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "mean"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "offset"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "output"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "scale"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "unit"], [36, 1, 1, "c.hp_fusedbatchnorm_p", "variance"]], "hp_fusedbatchnorm_s": [[36, 1, 1, "c.hp_fusedbatchnorm_s", "channel"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "core_mask"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "epsilon"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "input"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "mean"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "offset"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "output"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "scale"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "unit"], [36, 1, 1, "c.hp_fusedbatchnorm_s", "variance"]], "hp_gelu_p": [[10, 1, 1, "c.hp_gelu_p", "Input0"], [10, 1, 1, "c.hp_gelu_p", "approximate"], [10, 1, 1, "c.hp_gelu_p", "length"], [10, 1, 1, "c.hp_gelu_p", "output"]], "hp_gelu_s": [[10, 1, 1, "c.hp_gelu_s", "Input0"], [10, 1, 1, "c.hp_gelu_s", "approximate"], [10, 1, 1, "c.hp_gelu_s", "core_mask"], [10, 1, 1, "c.hp_gelu_s", "length"], [10, 1, 1, "c.hp_gelu_s", "output"]], "hp_groupnormfusion_p": [[37, 1, 1, "c.hp_groupnormfusion_p", "batch"], [37, 1, 1, "c.hp_groupnormfusion_p", "channel"], [37, 1, 1, "c.hp_groupnormfusion_p", "epsilon"], [37, 1, 1, "c.hp_groupnormfusion_p", "input"], [37, 1, 1, "c.hp_groupnormfusion_p", "mean"], [37, 1, 1, "c.hp_groupnormfusion_p", "num_groups"], [37, 1, 1, "c.hp_groupnormfusion_p", "offset"], [37, 1, 1, "c.hp_groupnormfusion_p", "output"], [37, 1, 1, "c.hp_groupnormfusion_p", "scale"], [37, 1, 1, "c.hp_groupnormfusion_p", "unit"], [37, 1, 1, "c.hp_groupnormfusion_p", "variance"]], "hp_groupnormfusion_s": [[37, 1, 1, "c.hp_groupnormfusion_s", "batch"], [37, 1, 1, "c.hp_groupnormfusion_s", "channel"], [37, 1, 1, "c.hp_groupnormfusion_s", "core_mask"], [37, 1, 1, "c.hp_groupnormfusion_s", "epsilon"], [37, 1, 1, "c.hp_groupnormfusion_s", "input"], [37, 1, 1, "c.hp_groupnormfusion_s", "mean"], [37, 1, 1, "c.hp_groupnormfusion_s", "num_groups"], [37, 1, 1, "c.hp_groupnormfusion_s", "offset"], [37, 1, 1, "c.hp_groupnormfusion_s", "output"], [37, 1, 1, "c.hp_groupnormfusion_s", "scale"], [37, 1, 1, "c.hp_groupnormfusion_s", "unit"], [37, 1, 1, "c.hp_groupnormfusion_s", "variance"]], "hp_hardshrink_p": [[10, 1, 1, "c.hp_hardshrink_p", "Input0"], [10, 1, 1, "c.hp_hardshrink_p", "lambd"], [10, 1, 1, "c.hp_hardshrink_p", "length"], [10, 1, 1, "c.hp_hardshrink_p", "output"]], "hp_hardshrink_s": [[10, 1, 1, "c.hp_hardshrink_s", "Input0"], [10, 1, 1, "c.hp_hardshrink_s", "core_mask"], [10, 1, 1, "c.hp_hardshrink_s", "lambd"], [10, 1, 1, "c.hp_hardshrink_s", "length"], [10, 1, 1, "c.hp_hardshrink_s", "output"]], "hp_hardtanh_p": [[10, 1, 1, "c.hp_hardtanh_p", "Input0"], [10, 1, 1, "c.hp_hardtanh_p", "length"], [10, 1, 1, "c.hp_hardtanh_p", "max_val"], [10, 1, 1, "c.hp_hardtanh_p", "min_val"], [10, 1, 1, "c.hp_hardtanh_p", "output"]], "hp_hardtanh_s": [[10, 1, 1, "c.hp_hardtanh_s", "Input0"], [10, 1, 1, "c.hp_hardtanh_s", "core_mask"], [10, 1, 1, "c.hp_hardtanh_s", "length"], [10, 1, 1, "c.hp_hardtanh_s", "max_val"], [10, 1, 1, "c.hp_hardtanh_s", "min_val"], [10, 1, 1, "c.hp_hardtanh_s", "output"]], "hp_hsigmoid_p": [[10, 1, 1, "c.hp_hsigmoid_p", "Input0"], [10, 1, 1, "c.hp_hsigmoid_p", "length"], [10, 1, 1, "c.hp_hsigmoid_p", "output"]], "hp_hsigmoid_s": [[10, 1, 1, "c.hp_hsigmoid_s", "Input0"], [10, 1, 1, "c.hp_hsigmoid_s", "core_mask"], [10, 1, 1, "c.hp_hsigmoid_s", "length"], [10, 1, 1, "c.hp_hsigmoid_s", "output"]], "hp_hswish_p": [[10, 1, 1, "c.hp_hswish_p", "Input0"], [10, 1, 1, "c.hp_hswish_p", "length"], [10, 1, 1, "c.hp_hswish_p", "output"]], "hp_hswish_s": [[10, 1, 1, "c.hp_hswish_s", "Input0"], [10, 1, 1, "c.hp_hswish_s", "core_mask"], [10, 1, 1, "c.hp_hswish_s", "length"], [10, 1, 1, "c.hp_hswish_s", "output"]], "hp_leaky_relu_p": [[39, 1, 1, "c.hp_leaky_relu_p", "alpha"], [39, 1, 1, "c.hp_leaky_relu_p", "core_mask"], [39, 1, 1, "c.hp_leaky_relu_p", "elem_cnt"], [39, 1, 1, "c.hp_leaky_relu_p", "input"], [39, 1, 1, "c.hp_leaky_relu_p", "output"]], "hp_leaky_relu_s": [[39, 1, 1, "c.hp_leaky_relu_s", "alpha"], [39, 1, 1, "c.hp_leaky_relu_s", "core_mask"], [39, 1, 1, "c.hp_leaky_relu_s", "elem_cnt"], [39, 1, 1, "c.hp_leaky_relu_s", "input"], [39, 1, 1, "c.hp_leaky_relu_s", "output"]], "hp_lrelu_p": [[10, 1, 1, "c.hp_lrelu_p", "Input0"], [10, 1, 1, "c.hp_lrelu_p", "alpha"], [10, 1, 1, "c.hp_lrelu_p", "length"], [10, 1, 1, "c.hp_lrelu_p", "output"]], "hp_lrelu_s": [[10, 1, 1, "c.hp_lrelu_s", "Input0"], [10, 1, 1, "c.hp_lrelu_s", "alpha"], [10, 1, 1, "c.hp_lrelu_s", "core_mask"], [10, 1, 1, "c.hp_lrelu_s", "length"], [10, 1, 1, "c.hp_lrelu_s", "output"]], "hp_reduce_p": [[45, 1, 1, "c.hp_reduce_p", "core_mask"], [45, 1, 1, "c.hp_reduce_p", "dst_data"], [45, 1, 1, "c.hp_reduce_p", "param"], [45, 1, 1, "c.hp_reduce_p", "src_data"], [45, 1, 1, "c.hp_reduce_p", "tmp_dst_data"], [45, 1, 1, "c.hp_reduce_p", "tmp_src_data"]], "hp_reduce_s": [[45, 1, 1, "c.hp_reduce_s", "core_mask"], [45, 1, 1, "c.hp_reduce_s", "dst_data"], [45, 1, 1, "c.hp_reduce_s", "param"], [45, 1, 1, "c.hp_reduce_s", "src_data"]], "hp_relu6_p": [[10, 1, 1, "c.hp_relu6_p", "Input0"], [10, 1, 1, "c.hp_relu6_p", "length"], [10, 1, 1, "c.hp_relu6_p", "output"]], "hp_relu6_s": [[10, 1, 1, "c.hp_relu6_s", "Input0"], [10, 1, 1, "c.hp_relu6_s", "core_mask"], [10, 1, 1, "c.hp_relu6_s", "length"], [10, 1, 1, "c.hp_relu6_s", "output"]], "hp_relu_p": [[10, 1, 1, "c.hp_relu_p", "Input0"], [10, 1, 1, "c.hp_relu_p", "length"], [10, 1, 1, "c.hp_relu_p", "output"]], "hp_relu_s": [[10, 1, 1, "c.hp_relu_s", "Input0"], [10, 1, 1, "c.hp_relu_s", "core_mask"], [10, 1, 1, "c.hp_relu_s", "length"], [10, 1, 1, "c.hp_relu_s", "output"]], "hp_resize_anycore": [[46, 1, 1, "c.hp_resize_anycore", "core_mask"], [46, 1, 1, "c.hp_resize_anycore", "input"], [46, 1, 1, "c.hp_resize_anycore", "output"], [46, 1, 1, "c.hp_resize_anycore", "param"]], "hp_scalefusion_p": [[49, 1, 1, "c.hp_scalefusion_p", "bias"], [49, 1, 1, "c.hp_scalefusion_p", "dst_data"], [49, 1, 1, "c.hp_scalefusion_p", "length"], [49, 1, 1, "c.hp_scalefusion_p", "scale"], [49, 1, 1, "c.hp_scalefusion_p", "src_data"]], "hp_scalefusion_s": [[49, 1, 1, "c.hp_scalefusion_s", "bias"], [49, 1, 1, "c.hp_scalefusion_s", "core_mask"], [49, 1, 1, "c.hp_scalefusion_s", "dst_data"], [49, 1, 1, "c.hp_scalefusion_s", "length"], [49, 1, 1, "c.hp_scalefusion_s", "scale"], [49, 1, 1, "c.hp_scalefusion_s", "src_data"]], "hp_scatter_elements_p": [[50, 1, 1, "c.hp_scatter_elements_p", "core_mask"], [50, 1, 1, "c.hp_scatter_elements_p", "indices"], [50, 1, 1, "c.hp_scatter_elements_p", "input"], [50, 1, 1, "c.hp_scatter_elements_p", "output"], [50, 1, 1, "c.hp_scatter_elements_p", "param"], [50, 1, 1, "c.hp_scatter_elements_p", "updates"]], "hp_scatter_elements_s": [[50, 1, 1, "c.hp_scatter_elements_s", "core_mask"], [50, 1, 1, "c.hp_scatter_elements_s", "indices"], [50, 1, 1, "c.hp_scatter_elements_s", "input"], [50, 1, 1, "c.hp_scatter_elements_s", "output"], [50, 1, 1, "c.hp_scatter_elements_s", "param"], [50, 1, 1, "c.hp_scatter_elements_s", "updates"]], "hp_sgd_p": [[51, 1, 1, "c.hp_sgd_p", "accumulate"], [51, 1, 1, "c.hp_sgd_p", "dampening"], [51, 1, 1, "c.hp_sgd_p", "gradient"], [51, 1, 1, "c.hp_sgd_p", "learning_rate"], [51, 1, 1, "c.hp_sgd_p", "length"], [51, 1, 1, "c.hp_sgd_p", "moment"], [51, 1, 1, "c.hp_sgd_p", "nesterov"], [51, 1, 1, "c.hp_sgd_p", "weight"], [51, 1, 1, "c.hp_sgd_p", "weight_decay"]], "hp_sgd_s": [[51, 1, 1, "c.hp_sgd_s", "accumulate"], [51, 1, 1, "c.hp_sgd_s", "core_mask"], [51, 1, 1, "c.hp_sgd_s", "dampening"], [51, 1, 1, "c.hp_sgd_s", "end"], [51, 1, 1, "c.hp_sgd_s", "gradient"], [51, 1, 1, "c.hp_sgd_s", "learning_rate"], [51, 1, 1, "c.hp_sgd_s", "moment"], [51, 1, 1, "c.hp_sgd_s", "nesterov"], [51, 1, 1, "c.hp_sgd_s", "start"], [51, 1, 1, "c.hp_sgd_s", "weight"], [51, 1, 1, "c.hp_sgd_s", "weight_decay"]], "hp_sigmoid_p": [[10, 1, 1, "c.hp_sigmoid_p", "Input0"], [10, 1, 1, "c.hp_sigmoid_p", "length"], [10, 1, 1, "c.hp_sigmoid_p", "output"]], "hp_sigmoid_s": [[10, 1, 1, "c.hp_sigmoid_s", "Input0"], [10, 1, 1, "c.hp_sigmoid_s", "core_mask"], [10, 1, 1, "c.hp_sigmoid_s", "length"], [10, 1, 1, "c.hp_sigmoid_s", "output"]], "hp_softplus_p": [[10, 1, 1, "c.hp_softplus_p", "Input0"], [10, 1, 1, "c.hp_softplus_p", "length"], [10, 1, 1, "c.hp_softplus_p", "output"]], "hp_softplus_s": [[10, 1, 1, "c.hp_softplus_s", "Input0"], [10, 1, 1, "c.hp_softplus_s", "core_mask"], [10, 1, 1, "c.hp_softplus_s", "length"], [10, 1, 1, "c.hp_softplus_s", "output"]], "hp_softshrink_p": [[10, 1, 1, "c.hp_softshrink_p", "Input0"], [10, 1, 1, "c.hp_softshrink_p", "lambd"], [10, 1, 1, "c.hp_softshrink_p", "length"], [10, 1, 1, "c.hp_softshrink_p", "output"]], "hp_softshrink_s": [[10, 1, 1, "c.hp_softshrink_s", "Input0"], [10, 1, 1, "c.hp_softshrink_s", "core_mask"], [10, 1, 1, "c.hp_softshrink_s", "lambd"], [10, 1, 1, "c.hp_softshrink_s", "length"], [10, 1, 1, "c.hp_softshrink_s", "output"]], "hp_softsignopt_p": [[10, 1, 1, "c.hp_softsignopt_p", "Input0"], [10, 1, 1, "c.hp_softsignopt_p", "length"], [10, 1, 1, "c.hp_softsignopt_p", "output"]], "hp_softsignopt_s": [[10, 1, 1, "c.hp_softsignopt_s", "Input0"], [10, 1, 1, "c.hp_softsignopt_s", "core_mask"], [10, 1, 1, "c.hp_softsignopt_s", "length"], [10, 1, 1, "c.hp_softsignopt_s", "output"]], "hp_spacetobatch_p": [[52, 1, 1, "c.hp_spacetobatch_p", "block_size"], [52, 1, 1, "c.hp_spacetobatch_p", "data_size"], [52, 1, 1, "c.hp_spacetobatch_p", "input"], [52, 1, 1, "c.hp_spacetobatch_p", "input_shape"], [52, 1, 1, "c.hp_spacetobatch_p", "output"], [52, 1, 1, "c.hp_spacetobatch_p", "paddings"]], "hp_spacetobatch_s": [[52, 1, 1, "c.hp_spacetobatch_s", "block_size"], [52, 1, 1, "c.hp_spacetobatch_s", "core_mask"], [52, 1, 1, "c.hp_spacetobatch_s", "data_size"], [52, 1, 1, "c.hp_spacetobatch_s", "input"], [52, 1, 1, "c.hp_spacetobatch_s", "input_shape"], [52, 1, 1, "c.hp_spacetobatch_s", "output"], [52, 1, 1, "c.hp_spacetobatch_s", "paddings"]], "hp_spacetobatchnd_p": [[53, 1, 1, "c.hp_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.hp_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.hp_spacetobatchnd_p", "input"], [53, 1, 1, "c.hp_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.hp_spacetobatchnd_p", "output"], [53, 1, 1, "c.hp_spacetobatchnd_p", "paddings"]], "hp_spacetobatchnd_s": [[53, 1, 1, "c.hp_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.hp_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.hp_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.hp_spacetobatchnd_s", "input"], [53, 1, 1, "c.hp_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.hp_spacetobatchnd_s", "output"], [53, 1, 1, "c.hp_spacetobatchnd_s", "paddings"]], "hp_spacetodepth_p": [[54, 1, 1, "c.hp_spacetodepth_p", "block"], [54, 1, 1, "c.hp_spacetodepth_p", "data_size"], [54, 1, 1, "c.hp_spacetodepth_p", "in_shape"], [54, 1, 1, "c.hp_spacetodepth_p", "input"], [54, 1, 1, "c.hp_spacetodepth_p", "output"]], "hp_spacetodepth_s": [[54, 1, 1, "c.hp_spacetodepth_s", "block"], [54, 1, 1, "c.hp_spacetodepth_s", "core_mask"], [54, 1, 1, "c.hp_spacetodepth_s", "data_size"], [54, 1, 1, "c.hp_spacetodepth_s", "in_shape"], [54, 1, 1, "c.hp_spacetodepth_s", "input"], [54, 1, 1, "c.hp_spacetodepth_s", "output"]], "hp_swish_p": [[10, 1, 1, "c.hp_swish_p", "Input0"], [10, 1, 1, "c.hp_swish_p", "length"], [10, 1, 1, "c.hp_swish_p", "output"]], "hp_swish_s": [[10, 1, 1, "c.hp_swish_s", "Input0"], [10, 1, 1, "c.hp_swish_s", "core_mask"], [10, 1, 1, "c.hp_swish_s", "length"], [10, 1, 1, "c.hp_swish_s", "output"]], "hp_tanh_p": [[10, 1, 1, "c.hp_tanh_p", "Input0"], [10, 1, 1, "c.hp_tanh_p", "length"], [10, 1, 1, "c.hp_tanh_p", "output"]], "hp_tanh_s": [[10, 1, 1, "c.hp_tanh_s", "Input0"], [10, 1, 1, "c.hp_tanh_s", "core_mask"], [10, 1, 1, "c.hp_tanh_s", "length"], [10, 1, 1, "c.hp_tanh_s", "output"]], "i16_batchtospace_p": [[17, 1, 1, "c.i16_batchtospace_p", "block_size"], [17, 1, 1, "c.i16_batchtospace_p", "crops"], [17, 1, 1, "c.i16_batchtospace_p", "data_size"], [17, 1, 1, "c.i16_batchtospace_p", "input"], [17, 1, 1, "c.i16_batchtospace_p", "input_shape"], [17, 1, 1, "c.i16_batchtospace_p", "output"]], "i16_batchtospace_s": [[17, 1, 1, "c.i16_batchtospace_s", "block_size"], [17, 1, 1, "c.i16_batchtospace_s", "core_mask"], [17, 1, 1, "c.i16_batchtospace_s", "crops"], [17, 1, 1, "c.i16_batchtospace_s", "data_size"], [17, 1, 1, "c.i16_batchtospace_s", "input"], [17, 1, 1, "c.i16_batchtospace_s", "input_shape"], [17, 1, 1, "c.i16_batchtospace_s", "output"]], "i16_batchtospacend_p": [[18, 1, 1, "c.i16_batchtospacend_p", "block_size"], [18, 1, 1, "c.i16_batchtospacend_p", "crops"], [18, 1, 1, "c.i16_batchtospacend_p", "data_size"], [18, 1, 1, "c.i16_batchtospacend_p", "input"], [18, 1, 1, "c.i16_batchtospacend_p", "input_shape"], [18, 1, 1, "c.i16_batchtospacend_p", "output"]], "i16_batchtospacend_s": [[18, 1, 1, "c.i16_batchtospacend_s", "block_size"], [18, 1, 1, "c.i16_batchtospacend_s", "core_mask"], [18, 1, 1, "c.i16_batchtospacend_s", "crops"], [18, 1, 1, "c.i16_batchtospacend_s", "data_size"], [18, 1, 1, "c.i16_batchtospacend_s", "input"], [18, 1, 1, "c.i16_batchtospacend_s", "input_shape"], [18, 1, 1, "c.i16_batchtospacend_s", "output"]], "i16_broadcastto_p": [[19, 1, 1, "c.i16_broadcastto_p", "data_size"], [19, 1, 1, "c.i16_broadcastto_p", "input"], [19, 1, 1, "c.i16_broadcastto_p", "input_shape"], [19, 1, 1, "c.i16_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.i16_broadcastto_p", "output"], [19, 1, 1, "c.i16_broadcastto_p", "output_shape"], [19, 1, 1, "c.i16_broadcastto_p", "output_shape_size"]], "i16_broadcastto_s": [[19, 1, 1, "c.i16_broadcastto_s", "core_mask"], [19, 1, 1, "c.i16_broadcastto_s", "data_size"], [19, 1, 1, "c.i16_broadcastto_s", "input"], [19, 1, 1, "c.i16_broadcastto_s", "input_shape"], [19, 1, 1, "c.i16_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.i16_broadcastto_s", "output"], [19, 1, 1, "c.i16_broadcastto_s", "output_shape"], [19, 1, 1, "c.i16_broadcastto_s", "output_shape_size"]], "i16_depthtospace_p": [[26, 1, 1, "c.i16_depthtospace_p", "block_size"], [26, 1, 1, "c.i16_depthtospace_p", "data_size"], [26, 1, 1, "c.i16_depthtospace_p", "in_shape"], [26, 1, 1, "c.i16_depthtospace_p", "input"], [26, 1, 1, "c.i16_depthtospace_p", "output"]], "i16_depthtospace_s": [[26, 1, 1, "c.i16_depthtospace_s", "block_size"], [26, 1, 1, "c.i16_depthtospace_s", "core_mask"], [26, 1, 1, "c.i16_depthtospace_s", "data_size"], [26, 1, 1, "c.i16_depthtospace_s", "in_shape"], [26, 1, 1, "c.i16_depthtospace_s", "input"], [26, 1, 1, "c.i16_depthtospace_s", "output"]], "i16_eltwise_p": [[28, 1, 1, "c.i16_eltwise_p", "Input0"], [28, 1, 1, "c.i16_eltwise_p", "Input1"], [28, 1, 1, "c.i16_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.i16_eltwise_p", "length"], [28, 1, 1, "c.i16_eltwise_p", "output"]], "i16_eltwise_s": [[28, 1, 1, "c.i16_eltwise_s", "Input0"], [28, 1, 1, "c.i16_eltwise_s", "Input1"], [28, 1, 1, "c.i16_eltwise_s", "core_mask"], [28, 1, 1, "c.i16_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.i16_eltwise_s", "length"], [28, 1, 1, "c.i16_eltwise_s", "output"]], "i16_equal_p": [[30, 1, 1, "c.i16_equal_p", "Input0"], [30, 1, 1, "c.i16_equal_p", "Input1"], [30, 1, 1, "c.i16_equal_p", "length"], [30, 1, 1, "c.i16_equal_p", "output"]], "i16_equal_s": [[30, 1, 1, "c.i16_equal_s", "Input0"], [30, 1, 1, "c.i16_equal_s", "Input1"], [30, 1, 1, "c.i16_equal_s", "core_mask"], [30, 1, 1, "c.i16_equal_s", "length"], [30, 1, 1, "c.i16_equal_s", "output"]], "i16_expfusion_p": [[32, 1, 1, "c.i16_expfusion_p", "dst_data"], [32, 1, 1, "c.i16_expfusion_p", "in_scale"], [32, 1, 1, "c.i16_expfusion_p", "length"], [32, 1, 1, "c.i16_expfusion_p", "out_scale"], [32, 1, 1, "c.i16_expfusion_p", "scale"], [32, 1, 1, "c.i16_expfusion_p", "src_data"]], "i16_expfusion_s": [[32, 1, 1, "c.i16_expfusion_s", "core_mask"], [32, 1, 1, "c.i16_expfusion_s", "dst_data"], [32, 1, 1, "c.i16_expfusion_s", "in_scale"], [32, 1, 1, "c.i16_expfusion_s", "length"], [32, 1, 1, "c.i16_expfusion_s", "out_scale"], [32, 1, 1, "c.i16_expfusion_s", "scale"], [32, 1, 1, "c.i16_expfusion_s", "src_data"]], "i16_matmulfusion_p": [[42, 1, 1, "c.i16_matmulfusion_p", "A"], [42, 1, 1, "c.i16_matmulfusion_p", "B"], [42, 1, 1, "c.i16_matmulfusion_p", "C"], [42, 1, 1, "c.i16_matmulfusion_p", "K"], [42, 1, 1, "c.i16_matmulfusion_p", "M"], [42, 1, 1, "c.i16_matmulfusion_p", "N"], [42, 1, 1, "c.i16_matmulfusion_p", "activation_type"], [42, 1, 1, "c.i16_matmulfusion_p", "bias"]], "i16_matmulfusion_s": [[42, 1, 1, "c.i16_matmulfusion_s", "A"], [42, 1, 1, "c.i16_matmulfusion_s", "B"], [42, 1, 1, "c.i16_matmulfusion_s", "C"], [42, 1, 1, "c.i16_matmulfusion_s", "K"], [42, 1, 1, "c.i16_matmulfusion_s", "M"], [42, 1, 1, "c.i16_matmulfusion_s", "N"], [42, 1, 1, "c.i16_matmulfusion_s", "activation_type"], [42, 1, 1, "c.i16_matmulfusion_s", "bias"], [42, 1, 1, "c.i16_matmulfusion_s", "core_mask"]], "i16_raggedrange_p": [[43, 1, 1, "c.i16_raggedrange_p", "deltas"], [43, 1, 1, "c.i16_raggedrange_p", "limits"], [43, 1, 1, "c.i16_raggedrange_p", "range_count"], [43, 1, 1, "c.i16_raggedrange_p", "splits"], [43, 1, 1, "c.i16_raggedrange_p", "starts"], [43, 1, 1, "c.i16_raggedrange_p", "values"]], "i16_raggedrange_s": [[43, 1, 1, "c.i16_raggedrange_s", "core_mask"], [43, 1, 1, "c.i16_raggedrange_s", "deltas"], [43, 1, 1, "c.i16_raggedrange_s", "limits"], [43, 1, 1, "c.i16_raggedrange_s", "range_count"], [43, 1, 1, "c.i16_raggedrange_s", "splits"], [43, 1, 1, "c.i16_raggedrange_s", "starts"], [43, 1, 1, "c.i16_raggedrange_s", "values"]], "i16_range_p": [[44, 1, 1, "c.i16_range_p", "delta"], [44, 1, 1, "c.i16_range_p", "length"], [44, 1, 1, "c.i16_range_p", "output"], [44, 1, 1, "c.i16_range_p", "start"]], "i16_range_s": [[44, 1, 1, "c.i16_range_s", "core_mask"], [44, 1, 1, "c.i16_range_s", "delta"], [44, 1, 1, "c.i16_range_s", "length"], [44, 1, 1, "c.i16_range_s", "output"], [44, 1, 1, "c.i16_range_s", "start"]], "i16_reduce_p": [[45, 1, 1, "c.i16_reduce_p", "core_mask"], [45, 1, 1, "c.i16_reduce_p", "dst_data"], [45, 1, 1, "c.i16_reduce_p", "param"], [45, 1, 1, "c.i16_reduce_p", "src_data"], [45, 1, 1, "c.i16_reduce_p", "tmp_dst_data"], [45, 1, 1, "c.i16_reduce_p", "tmp_src_data"]], "i16_reduce_s": [[45, 1, 1, "c.i16_reduce_s", "core_mask"], [45, 1, 1, "c.i16_reduce_s", "dst_data"], [45, 1, 1, "c.i16_reduce_s", "param"], [45, 1, 1, "c.i16_reduce_s", "src_data"]], "i16_scalefusion_p": [[49, 1, 1, "c.i16_scalefusion_p", "bias"], [49, 1, 1, "c.i16_scalefusion_p", "dst_data"], [49, 1, 1, "c.i16_scalefusion_p", "length"], [49, 1, 1, "c.i16_scalefusion_p", "scale"], [49, 1, 1, "c.i16_scalefusion_p", "src_data"]], "i16_scalefusion_s": [[49, 1, 1, "c.i16_scalefusion_s", "bias"], [49, 1, 1, "c.i16_scalefusion_s", "core_mask"], [49, 1, 1, "c.i16_scalefusion_s", "dst_data"], [49, 1, 1, "c.i16_scalefusion_s", "length"], [49, 1, 1, "c.i16_scalefusion_s", "scale"], [49, 1, 1, "c.i16_scalefusion_s", "src_data"]], "i16_scatter_elements_p": [[50, 1, 1, "c.i16_scatter_elements_p", "core_mask"], [50, 1, 1, "c.i16_scatter_elements_p", "indices"], [50, 1, 1, "c.i16_scatter_elements_p", "input"], [50, 1, 1, "c.i16_scatter_elements_p", "output"], [50, 1, 1, "c.i16_scatter_elements_p", "param"], [50, 1, 1, "c.i16_scatter_elements_p", "updates"]], "i16_scatter_elements_s": [[50, 1, 1, "c.i16_scatter_elements_s", "core_mask"], [50, 1, 1, "c.i16_scatter_elements_s", "indices"], [50, 1, 1, "c.i16_scatter_elements_s", "input"], [50, 1, 1, "c.i16_scatter_elements_s", "output"], [50, 1, 1, "c.i16_scatter_elements_s", "param"], [50, 1, 1, "c.i16_scatter_elements_s", "updates"]], "i16_spacetobatch_p": [[52, 1, 1, "c.i16_spacetobatch_p", "block_size"], [52, 1, 1, "c.i16_spacetobatch_p", "data_size"], [52, 1, 1, "c.i16_spacetobatch_p", "input"], [52, 1, 1, "c.i16_spacetobatch_p", "input_shape"], [52, 1, 1, "c.i16_spacetobatch_p", "output"], [52, 1, 1, "c.i16_spacetobatch_p", "paddings"]], "i16_spacetobatch_s": [[52, 1, 1, "c.i16_spacetobatch_s", "block_size"], [52, 1, 1, "c.i16_spacetobatch_s", "core_mask"], [52, 1, 1, "c.i16_spacetobatch_s", "data_size"], [52, 1, 1, "c.i16_spacetobatch_s", "input"], [52, 1, 1, "c.i16_spacetobatch_s", "input_shape"], [52, 1, 1, "c.i16_spacetobatch_s", "output"], [52, 1, 1, "c.i16_spacetobatch_s", "paddings"]], "i16_spacetobatchnd_p": [[53, 1, 1, "c.i16_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.i16_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.i16_spacetobatchnd_p", "input"], [53, 1, 1, "c.i16_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.i16_spacetobatchnd_p", "output"], [53, 1, 1, "c.i16_spacetobatchnd_p", "paddings"]], "i16_spacetobatchnd_s": [[53, 1, 1, "c.i16_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.i16_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.i16_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.i16_spacetobatchnd_s", "input"], [53, 1, 1, "c.i16_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.i16_spacetobatchnd_s", "output"], [53, 1, 1, "c.i16_spacetobatchnd_s", "paddings"]], "i16_spacetodepth_p": [[54, 1, 1, "c.i16_spacetodepth_p", "block"], [54, 1, 1, "c.i16_spacetodepth_p", "data_size"], [54, 1, 1, "c.i16_spacetodepth_p", "in_shape"], [54, 1, 1, "c.i16_spacetodepth_p", "input"], [54, 1, 1, "c.i16_spacetodepth_p", "output"]], "i16_spacetodepth_s": [[54, 1, 1, "c.i16_spacetodepth_s", "block"], [54, 1, 1, "c.i16_spacetodepth_s", "core_mask"], [54, 1, 1, "c.i16_spacetodepth_s", "data_size"], [54, 1, 1, "c.i16_spacetodepth_s", "in_shape"], [54, 1, 1, "c.i16_spacetodepth_s", "input"], [54, 1, 1, "c.i16_spacetodepth_s", "output"]], "i32_batchtospace_p": [[17, 1, 1, "c.i32_batchtospace_p", "block_size"], [17, 1, 1, "c.i32_batchtospace_p", "crops"], [17, 1, 1, "c.i32_batchtospace_p", "data_size"], [17, 1, 1, "c.i32_batchtospace_p", "input"], [17, 1, 1, "c.i32_batchtospace_p", "input_shape"], [17, 1, 1, "c.i32_batchtospace_p", "output"]], "i32_batchtospace_s": [[17, 1, 1, "c.i32_batchtospace_s", "block_size"], [17, 1, 1, "c.i32_batchtospace_s", "core_mask"], [17, 1, 1, "c.i32_batchtospace_s", "crops"], [17, 1, 1, "c.i32_batchtospace_s", "data_size"], [17, 1, 1, "c.i32_batchtospace_s", "input"], [17, 1, 1, "c.i32_batchtospace_s", "input_shape"], [17, 1, 1, "c.i32_batchtospace_s", "output"]], "i32_batchtospacend_p": [[18, 1, 1, "c.i32_batchtospacend_p", "block_size"], [18, 1, 1, "c.i32_batchtospacend_p", "crops"], [18, 1, 1, "c.i32_batchtospacend_p", "data_size"], [18, 1, 1, "c.i32_batchtospacend_p", "input"], [18, 1, 1, "c.i32_batchtospacend_p", "input_shape"], [18, 1, 1, "c.i32_batchtospacend_p", "output"]], "i32_batchtospacend_s": [[18, 1, 1, "c.i32_batchtospacend_s", "block_size"], [18, 1, 1, "c.i32_batchtospacend_s", "core_mask"], [18, 1, 1, "c.i32_batchtospacend_s", "crops"], [18, 1, 1, "c.i32_batchtospacend_s", "data_size"], [18, 1, 1, "c.i32_batchtospacend_s", "input"], [18, 1, 1, "c.i32_batchtospacend_s", "input_shape"], [18, 1, 1, "c.i32_batchtospacend_s", "output"]], "i32_broadcastto_p": [[19, 1, 1, "c.i32_broadcastto_p", "data_size"], [19, 1, 1, "c.i32_broadcastto_p", "input"], [19, 1, 1, "c.i32_broadcastto_p", "input_shape"], [19, 1, 1, "c.i32_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.i32_broadcastto_p", "output"], [19, 1, 1, "c.i32_broadcastto_p", "output_shape"], [19, 1, 1, "c.i32_broadcastto_p", "output_shape_size"]], "i32_broadcastto_s": [[19, 1, 1, "c.i32_broadcastto_s", "core_mask"], [19, 1, 1, "c.i32_broadcastto_s", "data_size"], [19, 1, 1, "c.i32_broadcastto_s", "input"], [19, 1, 1, "c.i32_broadcastto_s", "input_shape"], [19, 1, 1, "c.i32_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.i32_broadcastto_s", "output"], [19, 1, 1, "c.i32_broadcastto_s", "output_shape"], [19, 1, 1, "c.i32_broadcastto_s", "output_shape_size"]], "i32_depthtospace_p": [[26, 1, 1, "c.i32_depthtospace_p", "block_size"], [26, 1, 1, "c.i32_depthtospace_p", "data_size"], [26, 1, 1, "c.i32_depthtospace_p", "in_shape"], [26, 1, 1, "c.i32_depthtospace_p", "input"], [26, 1, 1, "c.i32_depthtospace_p", "output"]], "i32_depthtospace_s": [[26, 1, 1, "c.i32_depthtospace_s", "block_size"], [26, 1, 1, "c.i32_depthtospace_s", "core_mask"], [26, 1, 1, "c.i32_depthtospace_s", "data_size"], [26, 1, 1, "c.i32_depthtospace_s", "in_shape"], [26, 1, 1, "c.i32_depthtospace_s", "input"], [26, 1, 1, "c.i32_depthtospace_s", "output"]], "i32_eltwise_p": [[28, 1, 1, "c.i32_eltwise_p", "Input0"], [28, 1, 1, "c.i32_eltwise_p", "Input1"], [28, 1, 1, "c.i32_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.i32_eltwise_p", "length"], [28, 1, 1, "c.i32_eltwise_p", "output"]], "i32_eltwise_s": [[28, 1, 1, "c.i32_eltwise_s", "Input0"], [28, 1, 1, "c.i32_eltwise_s", "Input1"], [28, 1, 1, "c.i32_eltwise_s", "core_mask"], [28, 1, 1, "c.i32_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.i32_eltwise_s", "length"], [28, 1, 1, "c.i32_eltwise_s", "output"]], "i32_equal_p": [[30, 1, 1, "c.i32_equal_p", "Input0"], [30, 1, 1, "c.i32_equal_p", "Input1"], [30, 1, 1, "c.i32_equal_p", "length"], [30, 1, 1, "c.i32_equal_p", "output"]], "i32_equal_s": [[30, 1, 1, "c.i32_equal_s", "Input0"], [30, 1, 1, "c.i32_equal_s", "Input1"], [30, 1, 1, "c.i32_equal_s", "core_mask"], [30, 1, 1, "c.i32_equal_s", "length"], [30, 1, 1, "c.i32_equal_s", "output"]], "i32_expfusion_p": [[32, 1, 1, "c.i32_expfusion_p", "dst_data"], [32, 1, 1, "c.i32_expfusion_p", "in_scale"], [32, 1, 1, "c.i32_expfusion_p", "length"], [32, 1, 1, "c.i32_expfusion_p", "out_scale"], [32, 1, 1, "c.i32_expfusion_p", "scale"], [32, 1, 1, "c.i32_expfusion_p", "src_data"]], "i32_expfusion_s": [[32, 1, 1, "c.i32_expfusion_s", "core_mask"], [32, 1, 1, "c.i32_expfusion_s", "dst_data"], [32, 1, 1, "c.i32_expfusion_s", "in_scale"], [32, 1, 1, "c.i32_expfusion_s", "length"], [32, 1, 1, "c.i32_expfusion_s", "out_scale"], [32, 1, 1, "c.i32_expfusion_s", "scale"], [32, 1, 1, "c.i32_expfusion_s", "src_data"]], "i32_matmulfusion_p": [[42, 1, 1, "c.i32_matmulfusion_p", "A"], [42, 1, 1, "c.i32_matmulfusion_p", "B"], [42, 1, 1, "c.i32_matmulfusion_p", "C"], [42, 1, 1, "c.i32_matmulfusion_p", "K"], [42, 1, 1, "c.i32_matmulfusion_p", "M"], [42, 1, 1, "c.i32_matmulfusion_p", "N"], [42, 1, 1, "c.i32_matmulfusion_p", "activation_type"], [42, 1, 1, "c.i32_matmulfusion_p", "bias"]], "i32_matmulfusion_s": [[42, 1, 1, "c.i32_matmulfusion_s", "A"], [42, 1, 1, "c.i32_matmulfusion_s", "B"], [42, 1, 1, "c.i32_matmulfusion_s", "C"], [42, 1, 1, "c.i32_matmulfusion_s", "K"], [42, 1, 1, "c.i32_matmulfusion_s", "M"], [42, 1, 1, "c.i32_matmulfusion_s", "N"], [42, 1, 1, "c.i32_matmulfusion_s", "activation_type"], [42, 1, 1, "c.i32_matmulfusion_s", "bias"], [42, 1, 1, "c.i32_matmulfusion_s", "core_mask"]], "i32_raggedrange_p": [[43, 1, 1, "c.i32_raggedrange_p", "deltas"], [43, 1, 1, "c.i32_raggedrange_p", "limits"], [43, 1, 1, "c.i32_raggedrange_p", "range_count"], [43, 1, 1, "c.i32_raggedrange_p", "splits"], [43, 1, 1, "c.i32_raggedrange_p", "starts"], [43, 1, 1, "c.i32_raggedrange_p", "values"]], "i32_raggedrange_s": [[43, 1, 1, "c.i32_raggedrange_s", "core_mask"], [43, 1, 1, "c.i32_raggedrange_s", "deltas"], [43, 1, 1, "c.i32_raggedrange_s", "limits"], [43, 1, 1, "c.i32_raggedrange_s", "range_count"], [43, 1, 1, "c.i32_raggedrange_s", "splits"], [43, 1, 1, "c.i32_raggedrange_s", "starts"], [43, 1, 1, "c.i32_raggedrange_s", "values"]], "i32_range_p": [[44, 1, 1, "c.i32_range_p", "delta"], [44, 1, 1, "c.i32_range_p", "length"], [44, 1, 1, "c.i32_range_p", "output"], [44, 1, 1, "c.i32_range_p", "start"]], "i32_range_s": [[44, 1, 1, "c.i32_range_s", "core_mask"], [44, 1, 1, "c.i32_range_s", "delta"], [44, 1, 1, "c.i32_range_s", "length"], [44, 1, 1, "c.i32_range_s", "output"], [44, 1, 1, "c.i32_range_s", "start"]], "i32_reduce_p": [[45, 1, 1, "c.i32_reduce_p", "core_mask"], [45, 1, 1, "c.i32_reduce_p", "dst_data"], [45, 1, 1, "c.i32_reduce_p", "param"], [45, 1, 1, "c.i32_reduce_p", "src_data"], [45, 1, 1, "c.i32_reduce_p", "tmp_dst_data"], [45, 1, 1, "c.i32_reduce_p", "tmp_src_data"]], "i32_reduce_s": [[45, 1, 1, "c.i32_reduce_s", "core_mask"], [45, 1, 1, "c.i32_reduce_s", "dst_data"], [45, 1, 1, "c.i32_reduce_s", "param"], [45, 1, 1, "c.i32_reduce_s", "src_data"]], "i32_scalefusion_p": [[49, 1, 1, "c.i32_scalefusion_p", "bias"], [49, 1, 1, "c.i32_scalefusion_p", "dst_data"], [49, 1, 1, "c.i32_scalefusion_p", "length"], [49, 1, 1, "c.i32_scalefusion_p", "scale"], [49, 1, 1, "c.i32_scalefusion_p", "src_data"]], "i32_scalefusion_s": [[49, 1, 1, "c.i32_scalefusion_s", "bias"], [49, 1, 1, "c.i32_scalefusion_s", "core_mask"], [49, 1, 1, "c.i32_scalefusion_s", "dst_data"], [49, 1, 1, "c.i32_scalefusion_s", "length"], [49, 1, 1, "c.i32_scalefusion_s", "scale"], [49, 1, 1, "c.i32_scalefusion_s", "src_data"]], "i32_scatter_elements_p": [[50, 1, 1, "c.i32_scatter_elements_p", "core_mask"], [50, 1, 1, "c.i32_scatter_elements_p", "indices"], [50, 1, 1, "c.i32_scatter_elements_p", "input"], [50, 1, 1, "c.i32_scatter_elements_p", "output"], [50, 1, 1, "c.i32_scatter_elements_p", "param"], [50, 1, 1, "c.i32_scatter_elements_p", "updates"]], "i32_scatter_elements_s": [[50, 1, 1, "c.i32_scatter_elements_s", "core_mask"], [50, 1, 1, "c.i32_scatter_elements_s", "indices"], [50, 1, 1, "c.i32_scatter_elements_s", "input"], [50, 1, 1, "c.i32_scatter_elements_s", "output"], [50, 1, 1, "c.i32_scatter_elements_s", "param"], [50, 1, 1, "c.i32_scatter_elements_s", "updates"]], "i32_spacetobatch_p": [[52, 1, 1, "c.i32_spacetobatch_p", "block_size"], [52, 1, 1, "c.i32_spacetobatch_p", "data_size"], [52, 1, 1, "c.i32_spacetobatch_p", "input"], [52, 1, 1, "c.i32_spacetobatch_p", "input_shape"], [52, 1, 1, "c.i32_spacetobatch_p", "output"], [52, 1, 1, "c.i32_spacetobatch_p", "paddings"]], "i32_spacetobatch_s": [[52, 1, 1, "c.i32_spacetobatch_s", "block_size"], [52, 1, 1, "c.i32_spacetobatch_s", "core_mask"], [52, 1, 1, "c.i32_spacetobatch_s", "data_size"], [52, 1, 1, "c.i32_spacetobatch_s", "input"], [52, 1, 1, "c.i32_spacetobatch_s", "input_shape"], [52, 1, 1, "c.i32_spacetobatch_s", "output"], [52, 1, 1, "c.i32_spacetobatch_s", "paddings"]], "i32_spacetobatchnd_p": [[53, 1, 1, "c.i32_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.i32_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.i32_spacetobatchnd_p", "input"], [53, 1, 1, "c.i32_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.i32_spacetobatchnd_p", "output"], [53, 1, 1, "c.i32_spacetobatchnd_p", "paddings"]], "i32_spacetobatchnd_s": [[53, 1, 1, "c.i32_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.i32_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.i32_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.i32_spacetobatchnd_s", "input"], [53, 1, 1, "c.i32_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.i32_spacetobatchnd_s", "output"], [53, 1, 1, "c.i32_spacetobatchnd_s", "paddings"]], "i32_spacetodepth_p": [[54, 1, 1, "c.i32_spacetodepth_p", "block"], [54, 1, 1, "c.i32_spacetodepth_p", "data_size"], [54, 1, 1, "c.i32_spacetodepth_p", "in_shape"], [54, 1, 1, "c.i32_spacetodepth_p", "input"], [54, 1, 1, "c.i32_spacetodepth_p", "output"]], "i32_spacetodepth_s": [[54, 1, 1, "c.i32_spacetodepth_s", "block"], [54, 1, 1, "c.i32_spacetodepth_s", "core_mask"], [54, 1, 1, "c.i32_spacetodepth_s", "data_size"], [54, 1, 1, "c.i32_spacetodepth_s", "in_shape"], [54, 1, 1, "c.i32_spacetodepth_s", "input"], [54, 1, 1, "c.i32_spacetodepth_s", "output"]], "i8_Gru_p": [[38, 1, 1, "c.i8_Gru_p", "buffer"], [38, 1, 1, "c.i8_Gru_p", "core_mask"], [38, 1, 1, "c.i8_Gru_p", "gru_param"], [38, 1, 1, "c.i8_Gru_p", "hidden_state"], [38, 1, 1, "c.i8_Gru_p", "input"], [38, 1, 1, "c.i8_Gru_p", "input_bias"], [38, 1, 1, "c.i8_Gru_p", "output"], [38, 1, 1, "c.i8_Gru_p", "state_bias"], [38, 1, 1, "c.i8_Gru_p", "weight_g"], [38, 1, 1, "c.i8_Gru_p", "weight_r"]], "i8_Gru_s": [[38, 1, 1, "c.i8_Gru_s", "buffer"], [38, 1, 1, "c.i8_Gru_s", "core_mask"], [38, 1, 1, "c.i8_Gru_s", "gru_param"], [38, 1, 1, "c.i8_Gru_s", "hidden_state"], [38, 1, 1, "c.i8_Gru_s", "input"], [38, 1, 1, "c.i8_Gru_s", "input_bias"], [38, 1, 1, "c.i8_Gru_s", "output"], [38, 1, 1, "c.i8_Gru_s", "state_bias"], [38, 1, 1, "c.i8_Gru_s", "weight_g"], [38, 1, 1, "c.i8_Gru_s", "weight_r"]], "i8_adder_p": [[12, 1, 1, "c.i8_adder_p", "bias"], [12, 1, 1, "c.i8_adder_p", "conv_param"], [12, 1, 1, "c.i8_adder_p", "core_mask"], [12, 1, 1, "c.i8_adder_p", "input_w"], [12, 1, 1, "c.i8_adder_p", "input_x"], [12, 1, 1, "c.i8_adder_p", "out_y"], [12, 1, 1, "c.i8_adder_p", "quant_param"]], "i8_adder_s": [[12, 1, 1, "c.i8_adder_s", "bias"], [12, 1, 1, "c.i8_adder_s", "core_mask"], [12, 1, 1, "c.i8_adder_s", "input_w"], [12, 1, 1, "c.i8_adder_s", "input_x"], [12, 1, 1, "c.i8_adder_s", "out_y"], [12, 1, 1, "c.i8_adder_s", "param"]], "i8_batchtospace_p": [[17, 1, 1, "c.i8_batchtospace_p", "block_size"], [17, 1, 1, "c.i8_batchtospace_p", "crops"], [17, 1, 1, "c.i8_batchtospace_p", "data_size"], [17, 1, 1, "c.i8_batchtospace_p", "input"], [17, 1, 1, "c.i8_batchtospace_p", "input_shape"], [17, 1, 1, "c.i8_batchtospace_p", "output"]], "i8_batchtospace_s": [[17, 1, 1, "c.i8_batchtospace_s", "block_size"], [17, 1, 1, "c.i8_batchtospace_s", "core_mask"], [17, 1, 1, "c.i8_batchtospace_s", "crops"], [17, 1, 1, "c.i8_batchtospace_s", "data_size"], [17, 1, 1, "c.i8_batchtospace_s", "input"], [17, 1, 1, "c.i8_batchtospace_s", "input_shape"], [17, 1, 1, "c.i8_batchtospace_s", "output"]], "i8_batchtospacend_p": [[18, 1, 1, "c.i8_batchtospacend_p", "block_size"], [18, 1, 1, "c.i8_batchtospacend_p", "crops"], [18, 1, 1, "c.i8_batchtospacend_p", "data_size"], [18, 1, 1, "c.i8_batchtospacend_p", "input"], [18, 1, 1, "c.i8_batchtospacend_p", "input_shape"], [18, 1, 1, "c.i8_batchtospacend_p", "output"]], "i8_batchtospacend_s": [[18, 1, 1, "c.i8_batchtospacend_s", "block_size"], [18, 1, 1, "c.i8_batchtospacend_s", "core_mask"], [18, 1, 1, "c.i8_batchtospacend_s", "crops"], [18, 1, 1, "c.i8_batchtospacend_s", "data_size"], [18, 1, 1, "c.i8_batchtospacend_s", "input"], [18, 1, 1, "c.i8_batchtospacend_s", "input_shape"], [18, 1, 1, "c.i8_batchtospacend_s", "output"]], "i8_broadcastto_p": [[19, 1, 1, "c.i8_broadcastto_p", "data_size"], [19, 1, 1, "c.i8_broadcastto_p", "input"], [19, 1, 1, "c.i8_broadcastto_p", "input_shape"], [19, 1, 1, "c.i8_broadcastto_p", "input_shape_size"], [19, 1, 1, "c.i8_broadcastto_p", "output"], [19, 1, 1, "c.i8_broadcastto_p", "output_shape"], [19, 1, 1, "c.i8_broadcastto_p", "output_shape_size"]], "i8_broadcastto_s": [[19, 1, 1, "c.i8_broadcastto_s", "core_mask"], [19, 1, 1, "c.i8_broadcastto_s", "data_size"], [19, 1, 1, "c.i8_broadcastto_s", "input"], [19, 1, 1, "c.i8_broadcastto_s", "input_shape"], [19, 1, 1, "c.i8_broadcastto_s", "input_shape_size"], [19, 1, 1, "c.i8_broadcastto_s", "output"], [19, 1, 1, "c.i8_broadcastto_s", "output_shape"], [19, 1, 1, "c.i8_broadcastto_s", "output_shape_size"]], "i8_celu_p": [[10, 1, 1, "c.i8_celu_p", "Input0"], [10, 1, 1, "c.i8_celu_p", "alpha"], [10, 1, 1, "c.i8_celu_p", "length"], [10, 1, 1, "c.i8_celu_p", "output"]], "i8_celu_s": [[10, 1, 1, "c.i8_celu_s", "Input0"], [10, 1, 1, "c.i8_celu_s", "alpha"], [10, 1, 1, "c.i8_celu_s", "core_mask"], [10, 1, 1, "c.i8_celu_s", "length"], [10, 1, 1, "c.i8_celu_s", "output"]], "i8_clip_p": [[10, 1, 1, "c.i8_clip_p", "Input0"], [10, 1, 1, "c.i8_clip_p", "length"], [10, 1, 1, "c.i8_clip_p", "max_val"], [10, 1, 1, "c.i8_clip_p", "min_val"], [10, 1, 1, "c.i8_clip_p", "output"]], "i8_clip_s": [[10, 1, 1, "c.i8_clip_s", "Input0"], [10, 1, 1, "c.i8_clip_s", "core_mask"], [10, 1, 1, "c.i8_clip_s", "length"], [10, 1, 1, "c.i8_clip_s", "max_val"], [10, 1, 1, "c.i8_clip_s", "min_val"], [10, 1, 1, "c.i8_clip_s", "output"]], "i8_conv2d_p": [[20, 1, 1, "c.i8_conv2d_p", "bias"], [20, 1, 1, "c.i8_conv2d_p", "conv_param"], [20, 1, 1, "c.i8_conv2d_p", "core_mask"], [20, 1, 1, "c.i8_conv2d_p", "input_w"], [20, 1, 1, "c.i8_conv2d_p", "input_x"], [20, 1, 1, "c.i8_conv2d_p", "out_y"], [20, 1, 1, "c.i8_conv2d_p", "quant_param"]], "i8_conv2d_s": [[20, 1, 1, "c.i8_conv2d_s", "bias"], [20, 1, 1, "c.i8_conv2d_s", "conv_param"], [20, 1, 1, "c.i8_conv2d_s", "core_mask"], [20, 1, 1, "c.i8_conv2d_s", "input_w"], [20, 1, 1, "c.i8_conv2d_s", "input_x"], [20, 1, 1, "c.i8_conv2d_s", "out_y"], [20, 1, 1, "c.i8_conv2d_s", "quant_param"]], "i8_convtranspose_p": [[21, 1, 1, "c.i8_convtranspose_p", "bias"], [21, 1, 1, "c.i8_convtranspose_p", "conv_param"], [21, 1, 1, "c.i8_convtranspose_p", "core_mask"], [21, 1, 1, "c.i8_convtranspose_p", "input_w"], [21, 1, 1, "c.i8_convtranspose_p", "input_x"], [21, 1, 1, "c.i8_convtranspose_p", "out_y"]], "i8_convtranspose_s": [[21, 1, 1, "c.i8_convtranspose_s", "bias"], [21, 1, 1, "c.i8_convtranspose_s", "conv_param"], [21, 1, 1, "c.i8_convtranspose_s", "core_mask"], [21, 1, 1, "c.i8_convtranspose_s", "input_w"], [21, 1, 1, "c.i8_convtranspose_s", "input_x"], [21, 1, 1, "c.i8_convtranspose_s", "out_y"]], "i8_crop_and_resize_anycore": [[25, 1, 1, "c.i8_crop_and_resize_anycore", "box_idx"], [25, 1, 1, "c.i8_crop_and_resize_anycore", "boxes"], [25, 1, 1, "c.i8_crop_and_resize_anycore", "core_mask"], [25, 1, 1, "c.i8_crop_and_resize_anycore", "dst"], [25, 1, 1, "c.i8_crop_and_resize_anycore", "extrapolation_value"], [25, 1, 1, "c.i8_crop_and_resize_anycore", "param"], [25, 1, 1, "c.i8_crop_and_resize_anycore", "src"]], "i8_depthtospace_p": [[26, 1, 1, "c.i8_depthtospace_p", "block_size"], [26, 1, 1, "c.i8_depthtospace_p", "data_size"], [26, 1, 1, "c.i8_depthtospace_p", "in_shape"], [26, 1, 1, "c.i8_depthtospace_p", "input"], [26, 1, 1, "c.i8_depthtospace_p", "output"]], "i8_depthtospace_s": [[26, 1, 1, "c.i8_depthtospace_s", "block_size"], [26, 1, 1, "c.i8_depthtospace_s", "core_mask"], [26, 1, 1, "c.i8_depthtospace_s", "data_size"], [26, 1, 1, "c.i8_depthtospace_s", "in_shape"], [26, 1, 1, "c.i8_depthtospace_s", "input"], [26, 1, 1, "c.i8_depthtospace_s", "output"]], "i8_eltwise_p": [[28, 1, 1, "c.i8_eltwise_p", "Input0"], [28, 1, 1, "c.i8_eltwise_p", "Input1"], [28, 1, 1, "c.i8_eltwise_p", "eltwise_mode_"], [28, 1, 1, "c.i8_eltwise_p", "length"], [28, 1, 1, "c.i8_eltwise_p", "output"]], "i8_eltwise_s": [[28, 1, 1, "c.i8_eltwise_s", "Input0"], [28, 1, 1, "c.i8_eltwise_s", "Input1"], [28, 1, 1, "c.i8_eltwise_s", "core_mask"], [28, 1, 1, "c.i8_eltwise_s", "eltwise_mode_"], [28, 1, 1, "c.i8_eltwise_s", "length"], [28, 1, 1, "c.i8_eltwise_s", "output"]], "i8_elu_p": [[10, 1, 1, "c.i8_elu_p", "Input0"], [10, 1, 1, "c.i8_elu_p", "alpha"], [10, 1, 1, "c.i8_elu_p", "length"], [10, 1, 1, "c.i8_elu_p", "output"]], "i8_elu_s": [[10, 1, 1, "c.i8_elu_s", "Input0"], [10, 1, 1, "c.i8_elu_s", "alpha"], [10, 1, 1, "c.i8_elu_s", "core_mask"], [10, 1, 1, "c.i8_elu_s", "length"], [10, 1, 1, "c.i8_elu_s", "output"]], "i8_equal_p": [[30, 1, 1, "c.i8_equal_p", "Input0"], [30, 1, 1, "c.i8_equal_p", "Input1"], [30, 1, 1, "c.i8_equal_p", "length"], [30, 1, 1, "c.i8_equal_p", "output"]], "i8_equal_s": [[30, 1, 1, "c.i8_equal_s", "Input0"], [30, 1, 1, "c.i8_equal_s", "Input1"], [30, 1, 1, "c.i8_equal_s", "core_mask"], [30, 1, 1, "c.i8_equal_s", "length"], [30, 1, 1, "c.i8_equal_s", "output"]], "i8_expfusion_p": [[32, 1, 1, "c.i8_expfusion_p", "dst_data"], [32, 1, 1, "c.i8_expfusion_p", "in_scale"], [32, 1, 1, "c.i8_expfusion_p", "length"], [32, 1, 1, "c.i8_expfusion_p", "out_scale"], [32, 1, 1, "c.i8_expfusion_p", "scale"], [32, 1, 1, "c.i8_expfusion_p", "src_data"]], "i8_expfusion_s": [[32, 1, 1, "c.i8_expfusion_s", "core_mask"], [32, 1, 1, "c.i8_expfusion_s", "dst_data"], [32, 1, 1, "c.i8_expfusion_s", "in_scale"], [32, 1, 1, "c.i8_expfusion_s", "length"], [32, 1, 1, "c.i8_expfusion_s", "out_scale"], [32, 1, 1, "c.i8_expfusion_s", "scale"], [32, 1, 1, "c.i8_expfusion_s", "src_data"]], "i8_gelu_p": [[10, 1, 1, "c.i8_gelu_p", "Input0"], [10, 1, 1, "c.i8_gelu_p", "approximate"], [10, 1, 1, "c.i8_gelu_p", "length"], [10, 1, 1, "c.i8_gelu_p", "output"]], "i8_gelu_s": [[10, 1, 1, "c.i8_gelu_s", "Input0"], [10, 1, 1, "c.i8_gelu_s", "approximate"], [10, 1, 1, "c.i8_gelu_s", "core_mask"], [10, 1, 1, "c.i8_gelu_s", "length"], [10, 1, 1, "c.i8_gelu_s", "output"]], "i8_hardshrink_p": [[10, 1, 1, "c.i8_hardshrink_p", "Input0"], [10, 1, 1, "c.i8_hardshrink_p", "lambd"], [10, 1, 1, "c.i8_hardshrink_p", "length"], [10, 1, 1, "c.i8_hardshrink_p", "output"]], "i8_hardshrink_s": [[10, 1, 1, "c.i8_hardshrink_s", "Input0"], [10, 1, 1, "c.i8_hardshrink_s", "core_mask"], [10, 1, 1, "c.i8_hardshrink_s", "lambd"], [10, 1, 1, "c.i8_hardshrink_s", "length"], [10, 1, 1, "c.i8_hardshrink_s", "output"]], "i8_hardtanh_p": [[10, 1, 1, "c.i8_hardtanh_p", "Input0"], [10, 1, 1, "c.i8_hardtanh_p", "length"], [10, 1, 1, "c.i8_hardtanh_p", "max_val"], [10, 1, 1, "c.i8_hardtanh_p", "min_val"], [10, 1, 1, "c.i8_hardtanh_p", "output"]], "i8_hardtanh_s": [[10, 1, 1, "c.i8_hardtanh_s", "Input0"], [10, 1, 1, "c.i8_hardtanh_s", "core_mask"], [10, 1, 1, "c.i8_hardtanh_s", "length"], [10, 1, 1, "c.i8_hardtanh_s", "max_val"], [10, 1, 1, "c.i8_hardtanh_s", "min_val"], [10, 1, 1, "c.i8_hardtanh_s", "output"]], "i8_hsigmoid_p": [[10, 1, 1, "c.i8_hsigmoid_p", "Input0"], [10, 1, 1, "c.i8_hsigmoid_p", "length"], [10, 1, 1, "c.i8_hsigmoid_p", "output"]], "i8_hsigmoid_s": [[10, 1, 1, "c.i8_hsigmoid_s", "Input0"], [10, 1, 1, "c.i8_hsigmoid_s", "core_mask"], [10, 1, 1, "c.i8_hsigmoid_s", "length"], [10, 1, 1, "c.i8_hsigmoid_s", "output"]], "i8_hswish_p": [[10, 1, 1, "c.i8_hswish_p", "Input0"], [10, 1, 1, "c.i8_hswish_p", "length"], [10, 1, 1, "c.i8_hswish_p", "output"]], "i8_hswish_s": [[10, 1, 1, "c.i8_hswish_s", "Input0"], [10, 1, 1, "c.i8_hswish_s", "core_mask"], [10, 1, 1, "c.i8_hswish_s", "length"], [10, 1, 1, "c.i8_hswish_s", "output"]], "i8_leaky_relu_p": [[39, 1, 1, "c.i8_leaky_relu_p", "alpha"], [39, 1, 1, "c.i8_leaky_relu_p", "core_mask"], [39, 1, 1, "c.i8_leaky_relu_p", "elem_cnt"], [39, 1, 1, "c.i8_leaky_relu_p", "input"], [39, 1, 1, "c.i8_leaky_relu_p", "output"]], "i8_leaky_relu_s": [[39, 1, 1, "c.i8_leaky_relu_s", "alpha"], [39, 1, 1, "c.i8_leaky_relu_s", "core_mask"], [39, 1, 1, "c.i8_leaky_relu_s", "elem_cnt"], [39, 1, 1, "c.i8_leaky_relu_s", "input"], [39, 1, 1, "c.i8_leaky_relu_s", "output"]], "i8_lrelu_p": [[10, 1, 1, "c.i8_lrelu_p", "Input0"], [10, 1, 1, "c.i8_lrelu_p", "alpha"], [10, 1, 1, "c.i8_lrelu_p", "length"], [10, 1, 1, "c.i8_lrelu_p", "output"]], "i8_lrelu_s": [[10, 1, 1, "c.i8_lrelu_s", "Input0"], [10, 1, 1, "c.i8_lrelu_s", "alpha"], [10, 1, 1, "c.i8_lrelu_s", "core_mask"], [10, 1, 1, "c.i8_lrelu_s", "length"], [10, 1, 1, "c.i8_lrelu_s", "output"]], "i8_raggedrange_p": [[43, 1, 1, "c.i8_raggedrange_p", "deltas"], [43, 1, 1, "c.i8_raggedrange_p", "limits"], [43, 1, 1, "c.i8_raggedrange_p", "range_count"], [43, 1, 1, "c.i8_raggedrange_p", "splits"], [43, 1, 1, "c.i8_raggedrange_p", "starts"], [43, 1, 1, "c.i8_raggedrange_p", "values"]], "i8_raggedrange_s": [[43, 1, 1, "c.i8_raggedrange_s", "core_mask"], [43, 1, 1, "c.i8_raggedrange_s", "deltas"], [43, 1, 1, "c.i8_raggedrange_s", "limits"], [43, 1, 1, "c.i8_raggedrange_s", "range_count"], [43, 1, 1, "c.i8_raggedrange_s", "splits"], [43, 1, 1, "c.i8_raggedrange_s", "starts"], [43, 1, 1, "c.i8_raggedrange_s", "values"]], "i8_range_p": [[44, 1, 1, "c.i8_range_p", "delta"], [44, 1, 1, "c.i8_range_p", "length"], [44, 1, 1, "c.i8_range_p", "output"], [44, 1, 1, "c.i8_range_p", "start"]], "i8_range_s": [[44, 1, 1, "c.i8_range_s", "core_mask"], [44, 1, 1, "c.i8_range_s", "delta"], [44, 1, 1, "c.i8_range_s", "length"], [44, 1, 1, "c.i8_range_s", "output"], [44, 1, 1, "c.i8_range_s", "start"]], "i8_reduce_p": [[45, 1, 1, "c.i8_reduce_p", "core_mask"], [45, 1, 1, "c.i8_reduce_p", "dst_data"], [45, 1, 1, "c.i8_reduce_p", "param"], [45, 1, 1, "c.i8_reduce_p", "src_data"], [45, 1, 1, "c.i8_reduce_p", "tmp_dst_data"], [45, 1, 1, "c.i8_reduce_p", "tmp_src_data"]], "i8_reduce_s": [[45, 1, 1, "c.i8_reduce_s", "core_mask"], [45, 1, 1, "c.i8_reduce_s", "dst_data"], [45, 1, 1, "c.i8_reduce_s", "param"], [45, 1, 1, "c.i8_reduce_s", "src_data"]], "i8_relu6_p": [[10, 1, 1, "c.i8_relu6_p", "Input0"], [10, 1, 1, "c.i8_relu6_p", "length"], [10, 1, 1, "c.i8_relu6_p", "output"]], "i8_relu6_s": [[10, 1, 1, "c.i8_relu6_s", "Input0"], [10, 1, 1, "c.i8_relu6_s", "core_mask"], [10, 1, 1, "c.i8_relu6_s", "length"], [10, 1, 1, "c.i8_relu6_s", "output"]], "i8_relu_p": [[10, 1, 1, "c.i8_relu_p", "Input0"], [10, 1, 1, "c.i8_relu_p", "length"], [10, 1, 1, "c.i8_relu_p", "output"]], "i8_relu_s": [[10, 1, 1, "c.i8_relu_s", "Input0"], [10, 1, 1, "c.i8_relu_s", "core_mask"], [10, 1, 1, "c.i8_relu_s", "length"], [10, 1, 1, "c.i8_relu_s", "output"]], "i8_resize_anycore": [[46, 1, 1, "c.i8_resize_anycore", "core_mask"], [46, 1, 1, "c.i8_resize_anycore", "input"], [46, 1, 1, "c.i8_resize_anycore", "output"], [46, 1, 1, "c.i8_resize_anycore", "param"]], "i8_scalefusion_p": [[49, 1, 1, "c.i8_scalefusion_p", "bias"], [49, 1, 1, "c.i8_scalefusion_p", "dst_data"], [49, 1, 1, "c.i8_scalefusion_p", "length"], [49, 1, 1, "c.i8_scalefusion_p", "scale"], [49, 1, 1, "c.i8_scalefusion_p", "src_data"]], "i8_scalefusion_s": [[49, 1, 1, "c.i8_scalefusion_s", "bias"], [49, 1, 1, "c.i8_scalefusion_s", "core_mask"], [49, 1, 1, "c.i8_scalefusion_s", "dst_data"], [49, 1, 1, "c.i8_scalefusion_s", "length"], [49, 1, 1, "c.i8_scalefusion_s", "scale"], [49, 1, 1, "c.i8_scalefusion_s", "src_data"]], "i8_scatter_elements_p": [[50, 1, 1, "c.i8_scatter_elements_p", "core_mask"], [50, 1, 1, "c.i8_scatter_elements_p", "indices"], [50, 1, 1, "c.i8_scatter_elements_p", "input"], [50, 1, 1, "c.i8_scatter_elements_p", "output"], [50, 1, 1, "c.i8_scatter_elements_p", "param"], [50, 1, 1, "c.i8_scatter_elements_p", "updates"]], "i8_scatter_elements_s": [[50, 1, 1, "c.i8_scatter_elements_s", "core_mask"], [50, 1, 1, "c.i8_scatter_elements_s", "indices"], [50, 1, 1, "c.i8_scatter_elements_s", "input"], [50, 1, 1, "c.i8_scatter_elements_s", "output"], [50, 1, 1, "c.i8_scatter_elements_s", "param"], [50, 1, 1, "c.i8_scatter_elements_s", "updates"]], "i8_sigmoid_p": [[10, 1, 1, "c.i8_sigmoid_p", "Input0"], [10, 1, 1, "c.i8_sigmoid_p", "length"], [10, 1, 1, "c.i8_sigmoid_p", "output"]], "i8_sigmoid_s": [[10, 1, 1, "c.i8_sigmoid_s", "Input0"], [10, 1, 1, "c.i8_sigmoid_s", "core_mask"], [10, 1, 1, "c.i8_sigmoid_s", "length"], [10, 1, 1, "c.i8_sigmoid_s", "output"]], "i8_softplus_p": [[10, 1, 1, "c.i8_softplus_p", "Input0"], [10, 1, 1, "c.i8_softplus_p", "length"], [10, 1, 1, "c.i8_softplus_p", "output"]], "i8_softplus_s": [[10, 1, 1, "c.i8_softplus_s", "Input0"], [10, 1, 1, "c.i8_softplus_s", "core_mask"], [10, 1, 1, "c.i8_softplus_s", "length"], [10, 1, 1, "c.i8_softplus_s", "output"]], "i8_softshrink_p": [[10, 1, 1, "c.i8_softshrink_p", "Input0"], [10, 1, 1, "c.i8_softshrink_p", "lambd"], [10, 1, 1, "c.i8_softshrink_p", "length"], [10, 1, 1, "c.i8_softshrink_p", "output"]], "i8_softshrink_s": [[10, 1, 1, "c.i8_softshrink_s", "Input0"], [10, 1, 1, "c.i8_softshrink_s", "core_mask"], [10, 1, 1, "c.i8_softshrink_s", "lambd"], [10, 1, 1, "c.i8_softshrink_s", "length"], [10, 1, 1, "c.i8_softshrink_s", "output"]], "i8_softsignopt_p": [[10, 1, 1, "c.i8_softsignopt_p", "Input0"], [10, 1, 1, "c.i8_softsignopt_p", "length"], [10, 1, 1, "c.i8_softsignopt_p", "output"]], "i8_softsignopt_s": [[10, 1, 1, "c.i8_softsignopt_s", "Input0"], [10, 1, 1, "c.i8_softsignopt_s", "core_mask"], [10, 1, 1, "c.i8_softsignopt_s", "length"], [10, 1, 1, "c.i8_softsignopt_s", "output"]], "i8_spacetobatch_p": [[52, 1, 1, "c.i8_spacetobatch_p", "block_size"], [52, 1, 1, "c.i8_spacetobatch_p", "data_size"], [52, 1, 1, "c.i8_spacetobatch_p", "input"], [52, 1, 1, "c.i8_spacetobatch_p", "input_shape"], [52, 1, 1, "c.i8_spacetobatch_p", "output"], [52, 1, 1, "c.i8_spacetobatch_p", "paddings"]], "i8_spacetobatch_s": [[52, 1, 1, "c.i8_spacetobatch_s", "block_size"], [52, 1, 1, "c.i8_spacetobatch_s", "core_mask"], [52, 1, 1, "c.i8_spacetobatch_s", "data_size"], [52, 1, 1, "c.i8_spacetobatch_s", "input"], [52, 1, 1, "c.i8_spacetobatch_s", "input_shape"], [52, 1, 1, "c.i8_spacetobatch_s", "output"], [52, 1, 1, "c.i8_spacetobatch_s", "paddings"]], "i8_spacetobatchnd_p": [[53, 1, 1, "c.i8_spacetobatchnd_p", "block_size"], [53, 1, 1, "c.i8_spacetobatchnd_p", "data_size"], [53, 1, 1, "c.i8_spacetobatchnd_p", "input"], [53, 1, 1, "c.i8_spacetobatchnd_p", "input_shape"], [53, 1, 1, "c.i8_spacetobatchnd_p", "output"], [53, 1, 1, "c.i8_spacetobatchnd_p", "paddings"]], "i8_spacetobatchnd_s": [[53, 1, 1, "c.i8_spacetobatchnd_s", "block_size"], [53, 1, 1, "c.i8_spacetobatchnd_s", "core_mask"], [53, 1, 1, "c.i8_spacetobatchnd_s", "data_size"], [53, 1, 1, "c.i8_spacetobatchnd_s", "input"], [53, 1, 1, "c.i8_spacetobatchnd_s", "input_shape"], [53, 1, 1, "c.i8_spacetobatchnd_s", "output"], [53, 1, 1, "c.i8_spacetobatchnd_s", "paddings"]], "i8_spacetodepth_p": [[54, 1, 1, "c.i8_spacetodepth_p", "block"], [54, 1, 1, "c.i8_spacetodepth_p", "data_size"], [54, 1, 1, "c.i8_spacetodepth_p", "in_shape"], [54, 1, 1, "c.i8_spacetodepth_p", "input"], [54, 1, 1, "c.i8_spacetodepth_p", "output"]], "i8_spacetodepth_s": [[54, 1, 1, "c.i8_spacetodepth_s", "block"], [54, 1, 1, "c.i8_spacetodepth_s", "core_mask"], [54, 1, 1, "c.i8_spacetodepth_s", "data_size"], [54, 1, 1, "c.i8_spacetodepth_s", "in_shape"], [54, 1, 1, "c.i8_spacetodepth_s", "input"], [54, 1, 1, "c.i8_spacetodepth_s", "output"]], "i8_swish_p": [[10, 1, 1, "c.i8_swish_p", "Input0"], [10, 1, 1, "c.i8_swish_p", "length"], [10, 1, 1, "c.i8_swish_p", "output"]], "i8_swish_s": [[10, 1, 1, "c.i8_swish_s", "Input0"], [10, 1, 1, "c.i8_swish_s", "core_mask"], [10, 1, 1, "c.i8_swish_s", "length"], [10, 1, 1, "c.i8_swish_s", "output"]], "i8_tanh_p": [[10, 1, 1, "c.i8_tanh_p", "Input0"], [10, 1, 1, "c.i8_tanh_p", "length"], [10, 1, 1, "c.i8_tanh_p", "output"]], "i8_tanh_s": [[10, 1, 1, "c.i8_tanh_s", "Input0"], [10, 1, 1, "c.i8_tanh_s", "core_mask"], [10, 1, 1, "c.i8_tanh_s", "length"], [10, 1, 1, "c.i8_tanh_s", "output"]], "mindradar": [[6, 2, 1, "", "ComplexAbs"], [7, 2, 1, "", "FFT"], [8, 2, 1, "", "IFFT"]]}, "objnames": {"0": ["c", "function", "C \u51fd\u6570"], "1": ["c", "functionParam", "C \u51fd\u6570\u53c2\u6570"], "2": ["py", "class", "Python \u7c7b"]}, "objtypes": {"0": "c:function", "1": "c:functionParam", "2": "py:class"}, "terms": {"004": 4, "01": 0, "044715": 10, "0b0001": [12, 20, 21, 24, 25, 31, 33, 38, 39, 45, 46, 47, 50, 55, 56], "0b1111": [12, 20, 21, 24, 25, 31, 33, 38, 39, 45, 46, 47, 48, 50, 55, 56], "0f": [40, 51], "0x10000": [22, 23], "0x1000000": 45, "0x10000000": [10, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 26, 33, 34, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 50, 51, 52, 53, 54], "0x10000200": 41, "0x10000400": 41, "0x10000600": 41, "0x10000800": 41, "0x10000a00": 41, "0x10000c00": 41, "0x10000e00": 41, "0x10001000": [21, 41, 43, 50], "0x10001200": 41, "0x10001400": 41, "0x10001600": 41, "0x10001800": 41, "0x10001a00": 41, "0x10001c00": 41, "0x10001f00": 41, "0x10002000": [11, 13, 21, 41, 43, 50, 51], "0x10002200": 41, "0x10003000": [21, 43, 50], "0x10004000": [10, 11, 13, 16, 21, 36, 37, 38, 43, 50, 51], "0x10005000": [21, 50], "0x10006000": [11, 21, 50], "0x10008000": [36, 37, 38], "0x1000c000": [36, 37, 38], "0x1000ll": 48, "0x10010000": [12, 17, 18, 20, 33, 36, 37, 38, 39, 45, 54], "0x10014000": [36, 37, 38], "0x10018000": 38, "0x1001c000": 38, "0x10020000": [12, 19, 20, 22, 23, 38, 42, 45], "0x10021000": 45, "0x10022000": 45, "0x10023000": 45, "0x10024000": [38, 45], "0x10030000": [12, 20, 38], "0x10034000": 38, "0x10038000": 38, "0x1003c000": 38, "0x10040000": [12, 15, 20, 22, 23, 26, 42, 52, 53], "0x10060000": [12, 20, 22, 23], "0x10070000": [12, 20, 22, 23], "0x10080000": 15, "0x100c0000": 15, "0x10100000": 15, "0x10200000": 15, "0x10810000": [28, 29, 30, 32, 35, 49], "0x10820000": [28, 29, 30, 32, 35, 49], "0x10830000": [28, 30, 35, 49], "0x10840000": 49, "0x20000": [22, 23], "0x82000000": 39, "0x84000000": [46, 48], "0x84003000": 48, "0x84004000": 48, "0x84005000": 48, "0x84006000": 48, "0x84007000": 48, "0x85000000": 46, "0x86000000": 46, "0x87000000": 46, "0x87100000": 46, "0x88000000": [12, 20, 21, 24, 25, 31, 33, 38, 39, 45, 46, 47, 48, 50, 55, 56], "0x88100000": 38, "0x88200000": 38, "0x88300000": 38, "0x88400000": 38, "0x88500000": 38, "0x88600000": 38, "0x88700000": 38, "0x88800000": 38, "0x88900000": 38, "0x88a00000": 38, "0x88b00000": 38, "0x88c00000": 38, "0x88d00000": 38, "0x89000000": [12, 20, 21, 25, 46], "0x8a000000": [25, 46], "0x8b000000": [25, 46], "0x8c000000": [25, 46], "0x8d000000": [25, 46], "0x8e000000": [25, 46], "0x8f000000": 25, "0x90000000": [12, 20, 21, 25, 41], "0x91000000": [12, 20, 21, 25], "0x92000000": [12, 20, 21, 25], "0x93000000": 25, "0x94000000": [21, 25], "0x95000000": 25, "0x98000000": [24, 31, 45, 47, 50, 55, 56], "0xa0000000": [10, 11, 13, 14, 15, 16, 17, 18, 19, 21, 22, 23, 26, 28, 29, 30, 32, 33, 34, 35, 36, 37, 40, 41, 42, 43, 44, 49, 51, 52, 53, 54], "0xa0100000": 43, "0xa0200000": 43, "0xa0300000": 43, "0xa0400000": 43, "0xa0872c00": 29, "0xa1000000": [15, 41, 42], "0xa2000000": [15, 42], "0xa3000000": [15, 41, 42], "0xa4000000": 15, "0xa5000000": 15, "0xa8000000": [24, 47, 48, 50], "0xa8010000": 47, "0xa8020000": 47, "0xa8200000": 24, "0xa8410000": 24, "0xa8480000": 45, "0xa8483000": 45, "0xa8484000": 45, "0xa8485000": 45, "0xa8486000": 45, "0xa8490000": 45, "0xb0000000": [11, 13, 16, 17, 18, 19, 22, 23, 26, 28, 30, 35, 36, 37, 41, 49, 51, 52, 53, 54], "0xb0001000": [22, 23, 36, 37], "0xb0002000": [22, 23, 36, 37], "0xb0003000": [36, 37], "0xb0100000": 41, "0xb0200000": 41, "0xb0300000": 41, "0xb0400000": 41, "0xb0500000": 41, "0xb0600000": 41, "0xb0700000": 41, "0xb0800000": 41, "0xb0900000": 41, "0xb0b00000": 41, "0xb1000000": 49, "0xb8000000": [47, 50], "0xc0000000": [10, 11, 13, 14, 22, 23, 28, 29, 30, 32, 34, 35, 36, 37, 41, 49, 51], "0xc0100000": 41, "0xc0200000": 41, "0xc8000000": [47, 48, 50], "0xc8020000": 50, "0xc8040000": 50, "0xd0000000": 11, "0xff": [10, 11, 13, 15, 16, 17, 18, 19, 22, 23, 26, 28, 29, 30, 32, 34, 35, 36, 37, 40, 41, 42, 43, 44, 49, 51, 52, 53, 54], "10": [0, 7, 8, 12, 19, 20, 26, 31, 47, 48, 52, 53, 55, 56, 62], "100": [35, 40, 52, 53], "1000": [10, 28, 30, 32, 33, 34, 35, 39, 40, 44, 48, 49], "1001": 40, "1024": [21, 36], "10pt": 29, "11": [4, 62], "114": 0, "12": 62, "128": [0, 15, 16, 42, 64], "128x128": 0, "12pt": 29, "14": [16, 62], "1536": 4, "16": [16, 26, 37, 64], "1733": 4, "18": 4, "1e": [11, 13, 36, 37, 51], "1i": 4, "1j": 4, "1x1": [22, 23], "20": [4, 19, 41], "2000": 41, "2016": 10, "2019": 62, "2023": 4, "2048": [4, 11, 12, 13, 20, 51], "255": 0, "299": 0, "2_": 37, "2f": [11, 13, 51], "2i": 4, "2j": 4, "2x2": 60, "30": [12, 20, 50, 62], "3072": 4, "32": [0, 15, 36, 37, 62, 64], "32768": 0, "3328": 4, "3414": 4, "36": 37, "3f": [11, 13, 51], "3j": [6, 7, 8], "400": [17, 18], "4096": [4, 11, 13, 51], "47": 4, "49": 37, "4f": [11, 36, 37], "4i": 4, "4j": 4, "50": [0, 4, 19], "512": [36, 42], "587": 0, "5e": [11, 13, 51], "5f": [36, 37, 40, 44], "64": [0, 15, 16, 36, 37, 62, 64], "6678e": 58, "6900": 4, "6f": 11, "6pt": [10, 28, 41], "7004": 58, "7062": 4, "75": 46, "789": 33, "88": 10, "8f": 11, "999f": 11, "99f": 13, "9f": [11, 13, 51], "__init__": [0, 4, 60], "__main__": 0, "__name__": 0, "_count": 43, "_decay": 51, "_h": 16, "_i": 40, "_k": 43, "_len": 41, "_p": 64, "_rate": [13, 51], "_relu": 39, "_s": 64, "_scale": 32, "_shape": 54, "_size": [29, 41], "_t": [11, 51], "_val": 10, "_w": 16, "a_i": 6, "abs": [4, 58], "absgrad": 58, "accu_": 13, "accu_t": 13, "accumul": [0, 13, 51], "accuraci": 0, "act": 42, "activ": [27, 57, 58, 62], "activation_typ": 42, "activationgrad": 58, "adam": [0, 11, 58], "adamweightdecay": [27, 57], "adapthisteq": 4, "add": [50, 62], "add_channel": 0, "adder": [27, 57], "adderfus": 58, "adderparamet": 12, "addfus": 58, "addgrad": 58, "addn": 58, "affin": 58, "after": 0, "ai": [0, 5, 59], "align": [10, 11, 13, 17, 38, 41, 43, 51], "align_corn": 46, "all": [4, 58], "allgath": 58, "alpha": [4, 10, 39], "am": [12, 20, 21, 38, 39, 45, 50], "amd64": 62, "amzipp": 58, "anaconda": 62, "anaconda3": 62, "and": [0, 47], "ani": 62, "anytype_crop_anycor": 24, "anytype_expand_dims_anycor": 31, "anytype_fillv2_": 33, "anytype_fillv2_p": 33, "anytype_reverse_sequence_anycor": 47, "anytype_reversev2_anycor": 48, "anytype_squeeze_anycor": 55, "anytype_unsqueeze_anycor": 56, "api": [4, 57, 59], "app": 0, "append": 4, "applymomentum": [27, 57, 58], "approxim": 10, "arang": [4, 6, 7, 8], "arcsin": 58, "arg": 10, "argc": [10, 14, 15, 17, 18, 19, 26, 28, 29, 30, 32, 34, 35, 40, 41, 42, 43, 44, 49, 52, 53, 54], "argmax": 0, "argmax_indic": 0, "argmaxfus": 58, "argminfus": 58, "argv": [10, 14, 15, 17, 18, 19, 26, 28, 29, 30, 32, 34, 35, 40, 41, 42, 43, 44, 49, 52, 53, 54], "arm": [60, 62], "arm_toolchain": 60, "array": [0, 4, 48], "as": [0, 4, 6, 7, 8, 60], "asnumpi": 0, "assert": [27, 57, 58], "assert_": 14, "assign": 58, "astyp": [0, 6, 7, 8], "at": 47, "attent": [27, 57], "auto": 0, "avgpoolfus": 58, "avgpoolgrad": 58, "avgpoolinggrad": [27, 57], "axe": 45, "axi": [0, 4, 24, 45, 47, 48, 50, 55], "axis_": 50, "axis_flag": 48, "axis_flag_": 48, "axis_ndim": 48, "axis_ndim_": 48, "axis_sizes_": 45, "azimuthfftfftshift": 58, "azimuthifft": 58, "b_": [38, 41], "b_f": 41, "b_g": 41, "b_h": 17, "b_i": [6, 41], "b_o": 41, "b_w": 17, "backprop": [22, 23], "backward": [7, 8], "bat": 60, "batch": [0, 16, 17, 18, 20, 21, 22, 23, 26, 36, 37, 38, 41, 52, 53], "batch_": [38, 41], "batch_dim": 47, "batch_dim_": 47, "batch_siz": [0, 15, 38], "batchnorm": 58, "batchnormgrad": 58, "batchtospac": [27, 57, 58], "batchtospacend": [27, 57, 58], "be": 47, "been": 62, "befor": [0, 47], "begin": [0, 10, 11, 13, 14, 17, 28, 29, 30, 32, 38, 39, 41, 43, 51], "beta1": 11, "beta2": 11, "beta_1": 11, "beta_2": 11, "between": 47, "bias": [12, 20, 21, 41, 42, 49], "bias_data": [12, 20, 21], "bias_i": 49, "biasadd": 58, "biasaddgrad": 58, "bidirect": 38, "bidirectional_": [38, 41], "bigl": 10, "bigr": 10, "bilinear": 46, "binari": 62, "binarycrossentropi": 58, "binarycrossentropygrad": 58, "bitrev": 58, "block": [18, 26, 54], "block_h": [17, 18, 52, 53], "block_siz": [17, 18, 26, 52, 53], "block_w": [17, 18, 52, 53], "bool": [13, 14, 28, 29, 30, 41, 51, 58], "bottom": [17, 18, 52, 53], "box": 25, "box_idx": 25, "box_index": 25, "brdm_2": 0, "break": 45, "broadcastto": [27, 57, 58], "btr_60": 0, "buffer": [38, 41], "buffer_size_": [12, 20, 21, 22, 23], "build": [0, 60], "by": 48, "c128": 64, "c128_add_": 64, "c128_add_p": 64, "c128_batchtospace_": 17, "c128_batchtospace_p": 17, "c128_batchtospacend_": 18, "c128_batchtospacend_p": 18, "c128_broadcastto_": 19, "c128_broadcastto_p": 19, "c128_depthtospace_": 26, "c128_depthtospace_p": 26, "c128_eltwise_": 28, "c128_eltwise_p": 28, "c128_equal_": 30, "c128_equal_p": 30, "c128_expfusion_": 32, "c128_expfusion_p": 32, "c128_matmulfusion_": 42, "c128_matmulfusion_p": 42, "c128_scatter_elements_": 50, "c128_scatter_elements_p": 50, "c128_spacetobatch_": 52, "c128_spacetobatch_p": 52, "c128_spacetobatchnd_": 53, "c128_spacetobatchnd_p": 53, "c128_spacetodepth_": 54, "c128_spacetodepth_p": 54, "c64": 64, "c64_add_": 64, "c64_add_p": 64, "c64_batchtospace_": 17, "c64_batchtospace_p": 17, "c64_batchtospacend_": 18, "c64_batchtospacend_p": 18, "c64_broadcastto_": 19, "c64_broadcastto_p": 19, "c64_depthtospace_": 26, "c64_depthtospace_p": 26, "c64_eltwise_": 28, "c64_eltwise_p": 28, "c64_equal_": 30, "c64_equal_p": 30, "c64_expfusion_": 32, "c64_expfusion_p": 32, "c64_matmulfusion_": 42, "c64_matmulfusion_p": 42, "c64_scatter_elements_": 50, "c64_scatter_elements_p": 50, "c64_spacetobatch_": 52, "c64_spacetobatch_p": 52, "c64_spacetobatchnd_": 53, "c64_spacetobatchnd_p": 53, "c64_spacetodepth_": 54, "c64_spacetodepth_p": 54, "c_": [12, 17, 20, 21, 41], "c_str": 0, "c_t": 41, "calcul": [47, 62], "call": 58, "callback": 0, "case": [10, 13, 14, 28, 29, 30, 32, 39, 51], "cast": [4, 58], "cc": 0, "ccor": 20, "cd_run_param": 4, "cddata1": 4, "cddata2": 4, "cdot": [10, 11, 13, 16, 28, 29, 32, 36, 37, 39, 40, 41, 43, 49, 51], "ceil": 58, "cell": [0, 4, 41, 60], "cell_buff": 41, "cell_stat": 41, "cell_state_": 41, "celu": 10, "cfar": 58, "channel": [0, 16, 17, 18, 26, 36, 37, 52, 53, 54], "char": [10, 14, 15, 17, 18, 19, 26, 28, 29, 30, 32, 34, 35, 40, 41, 42, 43, 44, 49, 52, 53, 54], "check": [21, 45, 46, 47, 48], "check_output_hidden_st": 38, "check_seq_len": 38, "check_seq_len_": 38, "chmod": 60, "cin_channel": 0, "circshift": 4, "ckpt": 0, "class": [0, 4, 6, 7, 8, 60], "class_label": 0, "class_nam": 0, "clip": [0, 10, 58], "clip_by_valu": 0, "cliplimit": 4, "close": 4, "cmake": 60, "cmakelist": [0, 60], "cmd": 62, "cnn": 0, "cnn_model": 0, "col": 0, "color": 0, "color_imag": 0, "color_rgb2gray": 0, "complex": [4, 42], "complex128": [6, 7, 8, 64], "complex64": [4, 6, 7, 8, 64], "complexab": [4, 9, 57, 58, 62], "compon": 62, "compos": 0, "computestrid": 47, "concat": 58, "connect": 0, "const": [0, 11, 13, 16, 17, 18, 19, 22, 23, 26, 36, 37, 39, 41, 51, 52, 53, 54, 60], "constant": 4, "constant_valu": 4, "constantofshap": 58, "construct": [0, 4, 60], "context": [0, 60], "continu": 10, "conv1": 0, "conv2": 0, "conv2d": [0, 21, 22, 23, 27, 57], "conv2dbackpropfilterfus": [27, 57, 58], "conv2dbackpropinputfus": [27, 57, 58], "conv2dfus": 58, "conv2dgradfilt": 22, "conv2dtranspos": [27, 57], "conv2dtransposefus": 58, "conv3": 0, "conv_param": [12, 20, 21, 22, 23], "conv_paramet": [22, 23], "convert": 60, "convert_gray": 0, "convertcolor": 0, "converter_lit": 0, "convertmod": 0, "convertto": 0, "convertto2d": 0, "convparamet": [12, 20, 22, 23], "convquantparamet": [12, 20], "convtransposeparamet": 21, "coordinate_transform_mode_": 46, "coordinatetransformmod": 46, "copi": 47, "copy_elem_num_": 47, "core_id": [12, 20, 21, 24, 25, 33, 38, 45, 46, 47, 48, 50], "core_mask": [10, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56], "core_num": [12, 20, 21, 24, 25, 31, 33, 38, 45, 46, 47, 48, 50, 55, 56], "correct": 62, "correl": 20, "cos": 58, "countelementafterdim": 47, "countelementbeforedim": 47, "cout": 0, "cplx128": [17, 18, 19, 26, 28, 30, 31, 32, 33, 47, 48, 50, 52, 53, 54, 55, 56, 58], "cplx64": [17, 18, 19, 26, 28, 30, 31, 32, 33, 47, 48, 50, 52, 53, 54, 55, 56, 58], "cpu": [0, 6, 7, 8, 58, 60, 62], "cpu_info": 0, "cpudeviceinfo": 0, "creat": 62, "create_dict_iter": 0, "crop": [17, 18, 27, 57, 58], "cropandres": [27, 57, 58], "cropandresizeparamet": 25, "cross": 20, "crossentropyloss": 0, "cs": 4, "cubic": 46, "cubic_coeff": 46, "cubic_coeff_": 46, "cumsum": 58, "cur_coord": 48, "cur_coord_": 48, "current": 62, "custom": 62, "customextractfeatur": 58, "customnorm": 58, "custompredict": 58, "cv": 0, "cv2": 0, "cv_32f": 0, "cvmulcv": 58, "cvmulcvifft": 58, "d_k": 15, "dampen": 51, "data": [0, 4], "data_buffers_": 45, "data_handl": [0, 60], "data_it": 0, "data_path": 0, "data_s": [17, 18, 19, 26, 52, 53, 54, 60], "datas": 0, "dataset": 0, "dataset_dir": 0, "dataset_sink_mod": 0, "ddr": [10, 11, 13, 15, 16, 17, 18, 19, 22, 23, 26, 28, 29, 30, 32, 34, 35, 36, 37, 40, 41, 42, 43, 44, 45, 49, 51, 52, 53, 54], "decay": 11, "decod": 0, "deconv2dgradfilt": 58, "def": [0, 4, 60], "defin": [0, 28], "delegatemod": 0, "delta": [43, 44], "delta_k": 43, "dens": 0, "dense1": 0, "dense2": 0, "depth": 26, "depthtospac": [27, 57, 58], "depthwis": [22, 23], "detectionpostprocess": 58, "device_list": 0, "device_target": [0, 62], "differenti": 10, "dilat": [12, 20, 21, 22, 23], "dilation_h_": [12, 20, 21, 22, 23], "dilation_w_": [12, 20, 21, 22, 23], "dim": [4, 7, 8], "directori": 0, "displaystyl": 29, "distribut": 4, "divfus": 58, "divgrad": 58, "dma": 41, "doppler": 4, "dot": [7, 8, 15, 40, 43, 44], "doubl": [4, 17, 18, 19, 26, 28, 30, 32, 34, 35, 42, 43, 44, 45, 49, 50, 52, 53, 54], "download": 62, "dp": 64, "dp_add_": 64, "dp_add_p": 64, "dp_batchtospace_": 17, "dp_batchtospace_p": 17, "dp_batchtospacend_": 18, "dp_batchtospacend_p": 18, "dp_broadcastto_": 19, "dp_broadcastto_p": 19, "dp_depthtospace_": 26, "dp_depthtospace_p": 26, "dp_eltwise_": 28, "dp_eltwise_p": 28, "dp_equal_": 30, "dp_equal_p": 30, "dp_expfusion_": 32, "dp_expfusion_p": 32, "dp_floor_": 34, "dp_floor_p": 34, "dp_floordiv_": 35, "dp_floordiv_p": 35, "dp_matmulfusion_": 42, "dp_matmulfusion_p": 42, "dp_raggedrange_": 43, "dp_raggedrange_p": 43, "dp_range_": 44, "dp_range_p": 44, "dp_reduce_": 45, "dp_reduce_p": 45, "dp_scalefusion_": 49, "dp_scalefusion_p": 49, "dp_scatter_elements_": 50, "dp_scatter_elements_p": 50, "dp_spacetobatch_": 52, "dp_spacetobatch_p": 52, "dp_spacetobatchnd_": 53, "dp_spacetobatchnd_p": 53, "dp_spacetodepth_": 54, "dp_spacetodepth_p": 54, "dp_xxx_xxx": 64, "dropout": 58, "dropoutgrad": 58, "ds": 0, "dsp": [0, 5, 31, 55, 56, 57, 58, 59, 65], "dsp_info": 0, "dst": [25, 31, 47, 48, 55, 56], "dst_data": [32, 34, 35, 45, 49], "dst_i": [32, 34, 35, 49], "dst_ptr": 0, "dtype": [0, 4, 60], "dw": 22, "dx": 23, "dy": [22, 23], "dynamicqu": 58, "echo": 4, "echo1": 4, "echo2": 4, "echo_d1_mf": 4, "echo_d3_mf": 4, "echo_r": 4, "echo_s1": 4, "echo_s2": 4, "echo_s3": 4, "echo_s4": 4, "echo_s5": 4, "echo_shap": 4, "elem_cnt": 39, "element": [0, 47], "element_count": 0, "elementnum": 0, "els": [0, 46], "eltwis": [27, 57, 58], "eltwise_maximum": 28, "eltwise_mod": 28, "eltwise_mode_": 28, "eltwise_prod": 28, "eltwise_sum": 28, "elu": [10, 58], "embeddinglookup": [27, 57], "embeddinglookupfus": 58, "empti": 0, "end": [0, 10, 11, 13, 14, 17, 28, 29, 30, 32, 38, 39, 40, 41, 43, 51], "end_idx": 16, "endl": 0, "epsilon": [11, 36, 37], "equal": [27, 57, 58], "erf": [10, 58], "error": 10, "exe": 62, "exp": [0, 4, 6, 7, 8, 32], "exp_x": 0, "expand_dim": 0, "expanddim": [27, 57, 58], "expfus": [27, 57, 58], "exponenti": 4, "export": [0, 4, 60], "export_cnn": 0, "export_connect": 0, "export_gray": 0, "ext": 17, "extens": 0, "extrapolation_valu": 25, "f0": 4, "f32_typecast": 0, "f_nc": 4, "f_t": 41, "fa": 4, "fa_axi": 4, "fail": 0, "fals": [0, 4, 10, 13, 14, 30, 41, 51], "fft": [4, 8, 9, 57, 58], "fft1": 4, "fft_nobitrev": 58, "fft_time": 58, "fftshift": [4, 58], "figur": 4, "file_format": [0, 4, 60], "file_nam": [0, 4, 60], "fill": 58, "fillv2": [27, 57], "filter": [12, 22], "filter_zp_ptr_": 20, "finish": 62, "flag": 10, "flatten": [0, 58], "flattengrad": 58, "flip": 4, "flipud": 4, "float": [0, 10, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56], "float16": 64, "float32": [0, 4, 6, 40, 60, 64], "float64": [6, 64], "floatarraytovector": 0, "floor": [27, 57, 58], "floordiv": [27, 57, 58], "floorf": [34, 35], "floormod": 58, "fmk": 0, "for": [0, 45, 47, 48, 50, 62], "foral": 29, "forg": 62, "forget": 41, "forward": [7, 8], "fourpointinterpolatori": 58, "fp": 64, "fp16": [10, 11, 12, 13, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 58], "fp32": [10, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 41, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 58], "fp64": [17, 18, 19, 26, 28, 30, 31, 32, 33, 34, 35, 43, 44, 45, 47, 48, 49, 50, 52, 53, 54, 55, 56, 58], "fp_adamweightdecay_": 11, "fp_adamweightdecay_p": 11, "fp_add_": 64, "fp_add_p": 64, "fp_adder_": 12, "fp_adder_p": 12, "fp_applymomentum_": 13, "fp_applymomentum_p": 13, "fp_attention_": 15, "fp_attention_p": 15, "fp_avgpoolinggrad_": 16, "fp_avgpoolinggrad_p": 16, "fp_batchtospace_": 17, "fp_batchtospace_p": 17, "fp_batchtospacend_": 18, "fp_batchtospacend_p": 18, "fp_broadcastto_": 19, "fp_broadcastto_p": 19, "fp_celu_": 10, "fp_celu_p": 10, "fp_clip_": 10, "fp_clip_p": 10, "fp_conv2d_": 20, "fp_conv2d_p": 20, "fp_conv2dbackpropfilterfusion_": 22, "fp_conv2dbackpropfilterfusion_p": 22, "fp_conv2dbackpropinputfusion_": 23, "fp_conv2dbackpropinputfusion_p": 23, "fp_convtranspose_": 21, "fp_convtranspose_p": 21, "fp_crop_and_resize_anycor": 25, "fp_depthtospace_": 26, "fp_depthtospace_p": 26, "fp_eltwise_": 28, "fp_eltwise_p": 28, "fp_elu_": 10, "fp_elu_p": 10, "fp_embeddinglookup_": 29, "fp_embeddinglookup_p": 29, "fp_equal_": 30, "fp_equal_p": 30, "fp_expfusion_": 32, "fp_expfusion_p": 32, "fp_floor_": 34, "fp_floor_p": 34, "fp_floordiv_": 35, "fp_floordiv_p": 35, "fp_fusedbatchnorm_": 36, "fp_fusedbatchnorm_p": 36, "fp_gelu_": 10, "fp_gelu_p": 10, "fp_groupnormfusion_": 37, "fp_groupnormfusion_p": 37, "fp_gru_": 38, "fp_gru_p": 38, "fp_hardshrink_": 10, "fp_hardshrink_p": 10, "fp_hardtanh_": 10, "fp_hardtanh_p": 10, "fp_hsigmoid_": 10, "fp_hsigmoid_p": 10, "fp_hswish_": 10, "fp_hswish_p": 10, "fp_leaky_relu_": 39, "fp_leaky_relu_p": 39, "fp_linspace_": 40, "fp_linspace_p": 40, "fp_lrelu_": 10, "fp_lrelu_p": 10, "fp_lstm_p": 41, "fp_lstm_s": 41, "fp_matmulfusion_": 42, "fp_matmulfusion_p": 42, "fp_raggedrange_": 43, "fp_raggedrange_p": 43, "fp_range_": 44, "fp_range_p": 44, "fp_reduce_": 45, "fp_reduce_p": 45, "fp_relu6_": 10, "fp_relu6_p": 10, "fp_relu_": 10, "fp_relu_p": 10, "fp_resize_anycor": 46, "fp_scalefusion_": 49, "fp_scalefusion_p": 49, "fp_scatter_elements_": 50, "fp_scatter_elements_p": 50, "fp_sgd_p": 51, "fp_sgd_s": 51, "fp_sigmoid_": 10, "fp_sigmoid_p": 10, "fp_softplus_": 10, "fp_softplus_p": 10, "fp_softshrink_": 10, "fp_softshrink_p": 10, "fp_softsignopt_": 10, "fp_softsignopt_p": 10, "fp_spacetobatch_": 52, "fp_spacetobatch_p": 52, "fp_spacetobatchnd_": 53, "fp_spacetobatchnd_p": 53, "fp_spacetodepth_": 54, "fp_spacetodepth_p": 54, "fp_swish_": 10, "fp_swish_p": 10, "fp_tanh_": 10, "fp_tanh_p": 10, "fp_xxx_xxx": 64, "fprintf": 0, "fr": 4, "fr_axi": 4, "fr_gap": 4, "frac": [7, 8, 10, 11, 15, 16, 17, 29, 35, 36, 37, 40, 43], "from": [0, 4, 60, 62], "ft78ne": [0, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 60], "ft78nedeviceinfo": 0, "ftp": 62, "full": 4, "fullconnect": 58, "fuse": 36, "fusedbatchnorm": [27, 57, 58], "fusion": [22, 23, 37], "g_t": [11, 13, 41, 51], "gate": [10, 41], "gather": [4, 58], "gatherd": 58, "gathernd": 58, "gaussian": 10, "gcc": 62, "ge": 10, "gelu": 10, "geq": 39, "get": 0, "get_core_id": [12, 20, 21, 24, 25, 33, 38, 45, 46, 47, 48, 50], "getcorenum": [12, 20, 21, 24, 25, 31, 33, 38, 45, 46, 47, 48, 50, 55, 56], "getinput": 60, "getinputdata": 60, "getlogiccoreid": [12, 20, 21, 24, 25, 33, 38, 45, 46, 47, 48, 50], "getoutput": 60, "getoutputdata": [0, 60], "gimpel": 10, "glu": 58, "gnueabihf": 62, "googl": 10, "gradient": [11, 13, 51], "graph_mod": 0, "graphcel": 0, "gray": 0, "gray_cnn": 1, "graycnn": 0, "greater": [47, 58], "greater_dim": 47, "greaterequ": 58, "group": [12, 20, 21, 22, 23, 37], "group_": [12, 20, 21, 22, 23], "groupnormfus": [27, 57, 58], "gru": [27, 57, 58], "gru_param": 38, "gruparamet": 38, "gsm": [42, 43], "gt": 10, "h_": [17, 20, 21, 38, 41, 54], "h_t": [38, 41], "hadamard": 38, "half": [10, 11, 12, 13, 16, 17, 18, 19, 20, 21, 22, 23, 25, 26, 28, 29, 30, 32, 34, 35, 36, 37, 38, 39, 45, 46, 49, 50, 51, 52, 53, 54], "hard": 10, "hardshrink": 10, "hardtanh": 10, "has": 62, "has_bia": 0, "has_bias_": 41, "hashtablelookup": 58, "hat": [11, 36, 37], "head_dim": 15, "head_num": 15, "height": [0, 17, 18, 26, 52, 53], "hellodsp": [4, 59, 61], "hendryck": 10, "heterogen": [0, 60], "hf": 41, "hg": 41, "hi": 41, "hidden": 41, "hidden_buff": 41, "hidden_s": [38, 41], "hidden_size_": [38, 41], "hidden_st": [38, 41], "hidden_state_": 41, "hidden_state_batch": 38, "hn": 38, "ho": 41, "hp": 64, "hp_adamweightdecay_": 11, "hp_adamweightdecay_p": 11, "hp_add_": 64, "hp_add_p": 64, "hp_adder_": 12, "hp_adder_p": 12, "hp_applymomentum_": 13, "hp_applymomentum_p": 13, "hp_avgpoolinggrad_": 16, "hp_avgpoolinggrad_p": 16, "hp_batchtospace_": 17, "hp_batchtospace_p": 17, "hp_batchtospacend_": 18, "hp_batchtospacend_p": 18, "hp_broadcastto_": 19, "hp_broadcastto_p": 19, "hp_celu_": 10, "hp_celu_p": 10, "hp_clip_": 10, "hp_clip_p": 10, "hp_conv2d_": 20, "hp_conv2d_p": 20, "hp_conv2dbackpropfilterfusion_": 22, "hp_conv2dbackpropfilterfusion_p": 22, "hp_conv2dbackpropinputfusion_": 23, "hp_conv2dbackpropinputfusion_p": 23, "hp_convtranspose_": 21, "hp_convtranspose_p": 21, "hp_crop_and_resize_anycor": 25, "hp_depthtospace_": 26, "hp_depthtospace_p": 26, "hp_eltwise_": 28, "hp_eltwise_p": 28, "hp_elu_": 10, "hp_elu_p": 10, "hp_embeddinglookup_": 29, "hp_embeddinglookup_p": 29, "hp_equal_": 30, "hp_equal_p": 30, "hp_expfusion_": 32, "hp_expfusion_p": 32, "hp_floor_": 34, "hp_floor_p": 34, "hp_floordiv_": 35, "hp_floordiv_p": 35, "hp_fusedbatchnorm_": 36, "hp_fusedbatchnorm_p": 36, "hp_gelu_": 10, "hp_gelu_p": 10, "hp_groupnormfusion_": 37, "hp_groupnormfusion_p": 37, "hp_gru_": 38, "hp_gru_p": 38, "hp_hardshrink_": 10, "hp_hardshrink_p": 10, "hp_hardtanh_": 10, "hp_hardtanh_p": 10, "hp_hsigmoid_": 10, "hp_hsigmoid_p": 10, "hp_hswish_": 10, "hp_hswish_p": 10, "hp_leaky_relu_": 39, "hp_leaky_relu_p": 39, "hp_lrelu_": 10, "hp_lrelu_p": 10, "hp_reduce_": 45, "hp_reduce_p": 45, "hp_relu6_": 10, "hp_relu6_p": 10, "hp_relu_": 10, "hp_relu_p": 10, "hp_resize_anycor": 46, "hp_scalefusion_": 49, "hp_scalefusion_p": 49, "hp_scatter_elements_": 50, "hp_scatter_elements_p": 50, "hp_sgd_p": 51, "hp_sgd_s": 51, "hp_sigmoid_": 10, "hp_sigmoid_p": 10, "hp_softplus_": 10, "hp_softplus_p": 10, "hp_softshrink_": 10, "hp_softshrink_p": 10, "hp_softsignopt_": 10, "hp_softsignopt_p": 10, "hp_spacetobatch_": 52, "hp_spacetobatch_p": 52, "hp_spacetobatchnd_": 53, "hp_spacetobatchnd_p": 53, "hp_spacetodepth_": 54, "hp_spacetodepth_p": 54, "hp_swish_": 10, "hp_swish_p": 10, "hp_tanh_": 10, "hp_tanh_p": 10, "hpp": 0, "hr": 38, "hsigmoid": 10, "hswish": 10, "https": [0, 62], "hwc2chw": 0, "hyperbol": 10, "hz": 38, "i16": 64, "i16_add_": 64, "i16_add_p": 64, "i16_batchtospace_": 17, "i16_batchtospace_p": 17, "i16_batchtospacend_": 18, "i16_batchtospacend_p": 18, "i16_broadcastto_": 19, "i16_broadcastto_p": 19, "i16_depthtospace_": 26, "i16_depthtospace_p": 26, "i16_eltwise_": 28, "i16_eltwise_p": 28, "i16_equal_": 30, "i16_equal_p": 30, "i16_expfusion_": 32, "i16_expfusion_p": 32, "i16_matmulfusion_": 42, "i16_matmulfusion_p": 42, "i16_raggedrange_": 43, "i16_raggedrange_p": 43, "i16_range_": 44, "i16_range_p": 44, "i16_reduce_": 45, "i16_reduce_p": 45, "i16_scalefusion_": 49, "i16_scalefusion_p": 49, "i16_scatter_elements_": 50, "i16_scatter_elements_p": 50, "i16_spacetobatch_": 52, "i16_spacetobatch_p": 52, "i16_spacetobatchnd_": 53, "i16_spacetobatchnd_p": 53, "i16_spacetodepth_": 54, "i16_spacetodepth_p": 54, "i32": 64, "i32_add_": 64, "i32_add_p": 64, "i32_batchtospace_": 17, "i32_batchtospace_p": 17, "i32_batchtospacend_": 18, "i32_batchtospacend_p": 18, "i32_broadcastto_": 19, "i32_broadcastto_p": 19, "i32_depthtospace_": 26, "i32_depthtospace_p": 26, "i32_eltwise_": 28, "i32_eltwise_p": 28, "i32_equal_": 30, "i32_equal_p": 30, "i32_expfusion_": 32, "i32_expfusion_p": 32, "i32_matmulfusion_": 42, "i32_matmulfusion_p": 42, "i32_raggedrange_": 43, "i32_raggedrange_p": 43, "i32_range_": 44, "i32_range_p": 44, "i32_reduce_": 45, "i32_reduce_p": 45, "i32_scalefusion_": 49, "i32_scalefusion_p": 49, "i32_scatter_elements_": 50, "i32_scatter_elements_p": 50, "i32_spacetobatch_": 52, "i32_spacetobatch_p": 52, "i32_spacetobatchnd_": 53, "i32_spacetobatchnd_p": 53, "i32_spacetodepth_": 54, "i32_spacetodepth_p": 54, "i686": 62, "i8": 64, "i8_add_": 64, "i8_add_p": 64, "i8_adder_": 12, "i8_adder_p": 12, "i8_batchtospace_": 17, "i8_batchtospace_p": 17, "i8_batchtospacend_": 18, "i8_batchtospacend_p": 18, "i8_broadcastto_": 19, "i8_broadcastto_p": 19, "i8_celu_": 10, "i8_celu_p": 10, "i8_clip_": 10, "i8_clip_p": 10, "i8_conv2d_": 20, "i8_conv2d_p": 20, "i8_convtranspose_": 21, "i8_convtranspose_p": 21, "i8_crop_and_resize_anycor": 25, "i8_depthtospace_": 26, "i8_depthtospace_p": 26, "i8_eltwise_": 28, "i8_eltwise_p": 28, "i8_elu_": 10, "i8_elu_p": 10, "i8_equal_": 30, "i8_equal_p": 30, "i8_expfusion_": 32, "i8_expfusion_p": 32, "i8_gelu_": 10, "i8_gelu_p": 10, "i8_gru_": 38, "i8_gru_p": 38, "i8_hardshrink_": 10, "i8_hardshrink_p": 10, "i8_hardtanh_": 10, "i8_hardtanh_p": 10, "i8_hsigmoid_": 10, "i8_hsigmoid_p": 10, "i8_hswish_": 10, "i8_hswish_p": 10, "i8_leaky_relu_": 39, "i8_leaky_relu_p": 39, "i8_lrelu_": 10, "i8_lrelu_p": 10, "i8_raggedrange_": 43, "i8_raggedrange_p": 43, "i8_range_": 44, "i8_range_p": 44, "i8_reduce_": 45, "i8_reduce_p": 45, "i8_relu6_": 10, "i8_relu6_p": 10, "i8_relu_": 10, "i8_relu_p": 10, "i8_resize_anycor": 46, "i8_scalefusion_": 49, "i8_scalefusion_p": 49, "i8_scatter_elements_": 50, "i8_scatter_elements_p": 50, "i8_sigmoid_": 10, "i8_sigmoid_p": 10, "i8_softplus_": 10, "i8_softplus_p": 10, "i8_softshrink_": 10, "i8_softshrink_p": 10, "i8_softsignopt_": 10, "i8_softsignopt_p": 10, "i8_spacetobatch_": 52, "i8_spacetobatch_p": 52, "i8_spacetobatchnd_": 53, "i8_spacetobatchnd_p": 53, "i8_spacetodepth_": 54, "i8_spacetodepth_p": 54, "i8_swish_": 10, "i8_swish_p": 10, "i8_tanh_": 10, "i8_tanh_p": 10, "i_k": 29, "i_t": 41, "ide": [0, 4, 60], "ident": 42, "ids": 29, "ids_siz": 29, "ids_size_": 29, "if": [0, 10, 12, 13, 14, 20, 21, 24, 25, 28, 29, 30, 33, 38, 39, 41, 45, 46, 47, 48, 50, 51], "ifft": [4, 9, 57, 58], "ifft1": 4, "ifft_nobitrev": 58, "ifft_tim": 58, "ig": 41, "ii": 41, "imag": [0, 4], "image_four_channel": 0, "image_height": 25, "image_origin": 0, "imagefolderdataset": 0, "imagepath": 0, "imagesc": 4, "img_siz": 0, "import": [0, 4, 6, 7, 8, 60, 62], "importdata": 4, "imread": 0, "imread_color": 0, "imshow": 4, "in": [12, 20, 21, 29, 32, 38], "in_channel": [12, 20, 21, 22, 23], "in_h": [22, 23], "in_scal": 32, "in_shap": [24, 26, 54], "in_w": [22, 23], "includ": [0, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 23, 26, 28, 29, 30, 32, 34, 35, 36, 37, 40, 41, 42, 43, 44, 49, 51, 52, 53, 54], "index": 0, "indic": 50, "indices_data": 50, "indices_shap": 50, "indices_stride_": 50, "indices_total_num_": 50, "initi": 48, "inner_count": 47, "inner_count_": 47, "inner_s": 45, "inner_sizes_": 45, "inner_stride_": 47, "inp_box": 25, "inp_box_idx": 25, "input": [0, 4, 6, 10, 14, 16, 17, 18, 19, 23, 24, 25, 26, 28, 29, 30, 32, 33, 34, 35, 36, 37, 38, 39, 41, 45, 46, 49, 50, 52, 53, 54, 60], "input0": [10, 28, 29, 30, 32, 34, 35, 49], "input0_i": [28, 30, 35], "input1": [28, 29, 30, 35], "input1_i": [28, 30, 35], "input_": 16, "input_axis_size_": 50, "input_batch_": [12, 20, 21, 22, 23], "input_bia": [38, 41], "input_bias_": 41, "input_channel_": [12, 20, 21, 22, 23], "input_col_align": 38, "input_col_align_": [38, 41], "input_column": 0, "input_data": [12, 20, 21, 29, 47, 48, 50], "input_dims_": 50, "input_h": 16, "input_h_": [12, 20, 21, 22, 23], "input_i": [10, 14], "input_row_align_": [38, 41], "input_s": [38, 41], "input_shap": [12, 17, 18, 19, 20, 21, 24, 25, 45, 46, 50, 52, 53], "input_shape_": [25, 46, 48], "input_shape_s": 19, "input_size_": [38, 41], "input_strides_": 48, "input_tensor": 0, "input_total_num_": 50, "input_w": [12, 16, 20, 21], "input_w_": [12, 20, 21, 22, 23], "input_x": [12, 20, 21], "instal": [4, 62], "instancenorm": 58, "int": [0, 7, 8, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56], "int16": [17, 18, 19, 26, 28, 30, 31, 32, 33, 42, 43, 44, 45, 47, 48, 49, 50, 52, 53, 54, 55, 56, 58, 64], "int16_t": [17, 18, 19, 26, 28, 30, 32, 42, 43, 44, 45, 49, 50, 52, 53, 54], "int32": [17, 18, 19, 26, 28, 30, 31, 32, 33, 42, 43, 44, 45, 47, 48, 49, 50, 52, 53, 54, 55, 56, 58, 64], "int32_t": [17, 18, 19, 20, 25, 26, 28, 30, 42, 43, 44, 52, 53, 54], "int8": [4, 10, 12, 17, 18, 19, 20, 21, 24, 25, 26, 28, 30, 31, 32, 33, 38, 39, 43, 44, 45, 46, 47, 48, 49, 50, 52, 53, 54, 55, 56, 58, 64], "int8_t": [10, 12, 17, 18, 19, 20, 21, 25, 26, 28, 30, 32, 38, 39, 43, 44, 45, 46, 49, 50, 52, 53, 54], "inter_linear": 0, "interpol": 0, "invertpermut": 58, "io": [4, 41], "ip": 60, "ir": [38, 60], "is": [0, 62], "is_regul": 29, "is_regulated_": 29, "isfinit": 58, "iz": 38, "jpeg": 0, "jpg": [0, 4], "ka": 4, "keep_dim": 45, "keepdim": 0, "kernel_h": [22, 23], "kernel_h_": [12, 20, 21, 22, 23], "kernel_s": [0, 20], "kernel_w": [22, 23], "kernel_w_": [12, 20, 21, 22, 23], "kpnna": 0, "kr": 4, "l1": 12, "l2": [11, 13, 15, 16, 17, 18, 19, 22, 23, 26, 28, 30, 32, 34, 35, 36, 37, 40, 42, 43, 44, 49, 51, 52, 53, 54], "l2normalizefus": 58, "l_k": 43, "lambd": 10, "lambda": [0, 10], "lamda": 4, "layer": 29, "layer_num": 29, "layer_num_": 29, "layer_s": 29, "layer_size_": 29, "layernormfus": 58, "layernormgrad": 58, "lceil": 43, "leaki": [10, 39], "leakyrelu": [27, 57, 58], "learn": [13, 51], "learning_r": [0, 13, 51], "left": [10, 15, 17, 18, 43, 52, 53], "left_matrix": 41, "left_shift_": 20, "leftarrow": 29, "len": 0, "length": [10, 11, 13, 28, 29, 30, 32, 33, 34, 35, 39, 40, 44, 45, 49, 51], "less": [47, 58], "less_dim": 47, "lessequ": 58, "lib": 0, "librari": [57, 59], "limit": 43, "limit_k": 43, "linaro": 62, "line_buffers_": [25, 46], "linear": 10, "linspac": [27, 57], "linux": 62, "lite": [0, 4, 58, 66], "ln": 10, "load": 0, "load_checkpoint": 0, "load_param_into_net": 0, "loadmat": 4, "log": 58, "log1p": 58, "loggrad": 58, "logic_core_id": [12, 20, 21, 24, 25, 33, 38, 45, 46, 47, 48, 50], "logicaland": 58, "logicalnot": 58, "logicalor": 58, "logsoftmax": 58, "loss": 0, "loss_fn": 0, "lossmonitor": 0, "lr": 11, "lrelu": 10, "lrn": 58, "lshproject": 58, "lstm": [27, 57, 58], "lstm_param": 41, "lstmgrad": 58, "lstmgraddata": 58, "lstmgradweight": 58, "lstmp": 41, "lstmparamet": 41, "m_": [11, 51], "m_t": [11, 51], "main": [0, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56], "major": 15, "make_shar": 0, "map": 0, "mat": [0, 4], "mathbf": 28, "mathrm": 10, "matio": 4, "matmul": [4, 60], "matmulfus": [27, 57, 58], "matplotlib": 4, "max": [10, 28], "max_norm": 29, "max_norm_": 29, "max_val": 10, "maxi_": 20, "maximum": 58, "maximumgrad": 58, "maxpool2d": 0, "maxpoolfus": 58, "maxpoolgrad": 58, "mean": [0, 36, 37], "mean_c": 36, "memcpi": [0, 12, 20, 21, 24, 25, 38, 46, 47], "memset": 21, "merg": 58, "method": 46, "method_": 46, "metric": 0, "midir": 0, "min": 10, "min_val": 10, "mindir": [0, 4, 60], "mindradar": [4, 6, 7, 8], "mindspor": [0, 6, 7, 8, 38, 58, 65, 66], "mindspore_py38": 62, "mingw32_arm": 62, "mini_": 20, "miniconda3": 62, "minimum": 58, "minimumgrad": 58, "minsdpor": 66, "mobilenetv3": 10, "mod": 58, "mode": [0, 45, 46], "mode_": 45, "model": [0, 4, 60], "model_buf": 60, "model_context": 60, "model_data": 60, "model_typ": 60, "modelfil": 0, "modetyp": 60, "moment": [13, 51], "momentum": 13, "mr": [4, 6, 7, 8], "ms": [0, 4, 6, 7, 8, 60], "mstensor": [0, 60], "mt7004": [10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56], "mu_": 37, "mulfus": 58, "mulgrad": 58, "multipl": 62, "multipli": 4, "multiplier_": 20, "mutabledeviceinfo": [0, 60], "my_model": 0, "n_": 17, "n_1": [7, 8], "n_d": [7, 8], "n_i": [7, 8, 20], "n_t": 38, "na": 4, "name": 0, "ndim": [45, 47, 48, 50], "ndim_": [47, 48], "nearest": 46, "neg": 58, "neggrad": 58, "neq": [30, 32], "nesterov": [13, 51], "net": 6, "netron": 0, "new_shap": 4, "next": 62, "nhwc": [12, 20, 21, 54], "nllloss": 58, "nlllossgrad": 58, "nn": [0, 4, 60], "none": [42, 50, 62], "nonmaxsupppress": 58, "nonzero": 58, "norm": [7, 8], "normal": [36, 37], "notequ": 58, "np": [0, 4, 6, 7, 8, 60], "nr": 4, "null": 42, "num": [0, 40], "num_ax": 45, "num_axes_": 45, "num_class": 0, "num_direct": 38, "num_elem_": 48, "num_epoch": 0, "num_group": 37, "number": 47, "numpi": [0, 4, 6, 7, 8, 60], "numpy_exp": 0, "nweights_": [22, 23], "o_t": 41, "odot": [38, 41], "of": [47, 62], "offset": [24, 36, 37], "offset_": 24, "offset_c": [36, 37], "offset_s": 45, "omega_1": [7, 8], "omega_d": [7, 8], "omega_i": [7, 8], "on": 62, "one": [0, 47, 60], "onehot": 58, "oneslik": 58, "opencv": 0, "opencv2": 0, "oper": 0, "operatornam": [15, 42], "ops": [0, 4, 60], "optim": [0, 10], "org": 62, "orig_seq_length": 47, "ortho": [7, 8], "other": 41, "otherwis": [10, 13, 39, 51], "out": [0, 6, 7, 8, 17, 20, 32, 54, 60], "out_channel": [12, 20, 21, 22, 23], "out_data": 0, "out_h": [22, 23], "out_i": [6, 12, 20, 21], "out_j": 20, "out_scal": 32, "out_shap": 24, "out_w": [22, 23], "outer_count_": 47, "outer_s": 45, "outer_sizes_": 45, "outer_stride_": 47, "output": [0, 10, 14, 15, 16, 17, 18, 19, 24, 25, 26, 28, 29, 30, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 44, 45, 46, 49, 50, 52, 53, 54, 60], "output_": [16, 41], "output_batch_": [12, 20, 21, 22, 23], "output_channel_": [12, 20, 21, 22, 23], "output_column": 0, "output_data": [12, 20, 21, 29, 47, 48, 50], "output_h": 16, "output_h_": [12, 20, 21, 22, 23], "output_hidden_st": 38, "output_i": [10, 14, 28, 30, 44], "output_np": 0, "output_num": 0, "output_num_": 45, "output_prob": 0, "output_shap": [12, 19, 20, 21, 24, 25, 46], "output_shape_": [25, 46], "output_shape_s": 19, "output_size_": 41, "output_step_": [38, 41], "output_stride_": 50, "output_w": 16, "output_w_": [12, 20, 21, 22, 23], "output_zp_": 20, "outputfil": 0, "packed_input_": 41, "packed_output": 41, "packed_ptr": 41, "packed_st": 41, "packparam": [45, 50], "pad": [4, 12, 20, 21, 22, 23, 52, 53], "pad_d_": [22, 23], "pad_l": 16, "pad_l_": [12, 20, 21, 22, 23], "pad_r_": [22, 23], "pad_u": 16, "pad_u_": [12, 20, 21, 22, 23], "padarray": 4, "padfus": 58, "para": 4, "param": [12, 20, 21, 22, 23, 25, 38, 45, 46, 47, 48, 50], "param_dict": 0, "param_not_load": 0, "paramet": [0, 4, 41], "parti": 62, "partialfus": 58, "pass": 0, "path": 62, "pcolor": 4, "per_channel_": 20, "phasemul": 58, "phi": 10, "pi": [4, 6, 7, 8, 10], "pil": 0, "pip": [4, 62], "platform": 62, "plt": 4, "png": 0, "pool1": 0, "pool2": 0, "pool3": 0, "powergrad": 58, "powfus": 58, "predict": [0, 60], "predicted_class": 0, "prelufus": 58, "preparecropandresizebilinear": 25, "prepareresizebicub": 46, "prepareresizebilinear": 46, "prf": 4, "print": [0, 4, 6, 7, 8, 60], "priorbox": 58, "prod_": 8, "product": 15, "proj_col_align_": 41, "project_size_": 41, "promt": 62, "ptr": 0, "push_back": 0, "py": [0, 60], "py3": 62, "py_transform": 0, "pynative_mod": 0, "pyplot": 4, "python": 4, "python3": 62, "pytorch": 38, "qk": 15, "quad": [29, 32, 36, 37, 40, 43, 44], "quant_param": [12, 20], "quantdtypecast": 58, "r0": 4, "r_t": 38, "raggedrang": [27, 57, 58], "randn": 0, "random": 0, "randomnorm": 58, "randomstandardnorm": 58, "rang": [4, 27, 43, 57, 58], "range_count": 43, "rangfft": 58, "rank": 58, "rceil": 43, "rd": 4, "rdsar": 3, "rdsarv3": 4, "read": 0, "readfil": 60, "readimag": 0, "realdiv": 58, "reciproc": 58, "rectifi": 10, "reduc": [27, 57], "reduce_asum": 45, "reduce_axi": 45, "reduce_l2norm": 45, "reduce_max": 45, "reduce_mean": 45, "reduce_min": 45, "reduce_prod": 45, "reduce_sum": 45, "reduce_sumsquar": 45, "reducefus": 58, "reduceparamet": 45, "reducescatt": 58, "reduct": [0, 50], "reduction_typ": 50, "reduction_type_": 50, "refer": [57, 59], "reinterpret_cast": 0, "releas": 62, "reload_cnn": 0, "relu": [0, 10, 39, 42], "relu6": [10, 42], "requir": 62, "rescale_to_0_1": 0, "rescale_transform": 0, "reserv": 0, "reshap": [0, 7, 8, 58], "resiz": [0, 27, 47, 48, 57, 58], "resize_op": 0, "resized_imag": 0, "resized_image_tensor": 0, "resizegrad": 58, "resizemethod": 46, "resizeparamet": 46, "reslut": 0, "result": [0, 4, 62], "return": [0, 4, 10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 23, 26, 28, 29, 30, 32, 34, 35, 36, 37, 38, 40, 41, 42, 43, 44, 49, 51, 52, 53, 54, 60], "revers": 58, "reversesequ": [27, 57, 58], "reversesequenceparamet": 47, "reversev2": [27, 57, 58], "reversev2paramet": 48, "rgb": 0, "right": [10, 15, 17, 18, 43, 52, 53], "right_shift_": 20, "rnn": [38, 41], "roipool": 58, "roll": 4, "round": [4, 58], "row": [0, 15], "rsqrt": 58, "rsqrtgrad": 58, "run_check": 62, "s_": 32, "sar": [0, 3], "sar_imaging_with_rd_cs_wk": 4, "satur": 4, "saturation_tensor": 4, "save_checkpoint": 0, "savefig": 4, "scale": [15, 32, 36, 37, 49], "scale_c": [36, 37], "scale_i": 49, "scalefus": [27, 57, 58], "scatterel": [27, 57], "scatterelementsparamet": 50, "scatternd": 58, "scatterndupd": 58, "scipi": 4, "select": 58, "self": [0, 4, 10, 60], "selu": 58, "seq": 41, "seq_dim": 47, "seq_dim_": 47, "seq_len": [15, 38, 41], "seq_len_": [38, 41], "seq_length": 47, "serial": 0, "set_context": [0, 62], "setbuiltindeleg": 0, "sgd": [27, 57, 58], "shape": [0, 4, 20, 21, 31, 33, 47, 48, 55, 56, 58, 60], "shape_": 47, "shared_ptr": 60, "shift": 4, "should": 47, "show": 4, "shrinkag": 10, "sigma": [10, 37, 38, 41], "sigmoid": [10, 38, 41], "sigmoidcroosentropywithlogit": 58, "sigmoidcroosentropywithlogitsgrad": 58, "signal": [60, 62, 65], "signal_ndim": [7, 8], "simd": 41, "sin": 58, "sio": 4, "size": [0, 4, 20, 21, 47, 58], "size_t": [0, 60], "sizeof": [0, 12, 17, 18, 19, 20, 21, 24, 25, 26, 31, 38, 46, 47, 52, 53, 54, 55, 56], "skipgram": 58, "slice": 24, "slicefus": 58, "slici": 0, "sm": 15, "smc": [42, 43, 45], "smoothl1loss": 58, "smoothl1lossgrad": 58, "soft": 10, "softmax": [15, 58], "softmax_out": 15, "softmaxgrad": 58, "softplus": [10, 58], "softshrink": 10, "softsign": 10, "softsignopt": 10, "space": 26, "spacetobatch": [27, 57, 58], "spacetobatchnd": [27, 57, 58], "spacetodepth": [27, 57, 58], "sparsesoftmaxcrossentropywithlogit": 58, "sparsetodens": 58, "splice": 58, "split": [0, 43, 58], "splitwithoverlap": 58, "sq_fa_axi": 4, "sq_fr_axi": 4, "sqrt": [6, 7, 8, 10, 11, 15, 36, 37, 58], "sqrtgrad": 58, "squar": [4, 58], "squareddiffer": 58, "squeez": [0, 27, 57, 58], "src": [25, 31, 47, 48, 55, 56], "src_data": [32, 34, 45, 49], "src_data0": 35, "src_data1": 35, "src_i": [32, 34, 49], "src_ptr": 0, "ssh": 60, "stack": 58, "start": [11, 13, 40, 43, 44, 51], "start_idx": 16, "start_k": 43, "state_bia": [38, 41], "state_bias_": 41, "state_col_align": 38, "state_col_align_": [38, 41], "state_g": 41, "state_row_align_": [38, 41], "std": [0, 60], "stdbool": [13, 14, 42, 43, 44, 51], "stderr": 0, "stdio": [10, 11, 13, 14, 15, 16, 17, 18, 19, 22, 23, 26, 28, 29, 30, 32, 34, 35, 36, 37, 40, 41, 42, 43, 44, 49, 51, 52, 53, 54], "step": [4, 40], "str": [7, 8], "strict_load": 0, "stride": [0, 12, 20, 21, 22, 23, 47], "stride_h": 16, "stride_h_": [12, 20, 21, 22, 23], "stride_w": 16, "stride_w_": [12, 20, 21, 22, 23], "stridedslic": 58, "stridedslicegrad": 58, "strides_": 47, "string": 0, "struct": [12, 20, 21, 25, 38, 41, 45, 46, 47, 48, 50], "subfus": 58, "subgrad": 58, "success": 62, "sum": 0, "sum_": [7, 8, 12, 20, 29], "sum_row": 0, "super": [0, 4, 60], "swish": 10, "switch": 58, "switchlay": 58, "sys_bar": [12, 20, 21, 24, 25, 33, 38, 45, 46, 47, 48, 50], "system": 62, "t62": 0, "ta_axi": 4, "ta_gap": 4, "tangent": 10, "tanh": [10, 38, 41], "tar": 62, "target": 0, "tdpp": 58, "temp": 4, "temp1": 4, "temp2": 4, "temp3": 4, "tensor": [0, 4, 6, 7, 8, 20, 25, 33, 48, 50, 55, 60], "tensorlistfromtensor": 58, "tensorlistgetitem": 58, "tensorlistreserv": 58, "tensorlistsetitem": 58, "tensorliststack": 58, "tensorscatteradd": 58, "test_matmul": 60, "testadderl2fp32": 12, "testaddersmcfp32": 12, "testconvl2fp32": 20, "testconvsmcfp32": 20, "testconvtransposel2fp32": 21, "testconvtransposesmcfp32": 21, "testcropandresizesmcfp32": 25, "testcropsmcfp32": 24, "testgrul2fp32": 38, "testgrusmcfp32": 38, "testreducel2fp32": 45, "testreducesmcfp32": 45, "testresizefp32smc": 46, "testreversesequencefp32": 47, "testreversev2": 48, "testscatterelementsl2": 50, "testscatterelementssmc": 50, "text": [10, 13, 14, 15, 16, 17, 20, 22, 23, 28, 29, 30, 32, 39, 40, 41, 42, 43, 44, 51, 54], "textbf": 29, "tfrac": 10, "the": [47, 62], "third": 62, "tilefus": 58, "time": [0, 15, 17, 42, 44, 47], "titl": 4, "tmp_dst_data": 45, "tmp_input": 45, "tmp_input_shap": 45, "tmp_output": 45, "tmp_src_data": 45, "to": 62, "toolchain": 62, "top": [15, 17, 18, 23, 52, 53], "topkfus": 58, "total_copy_s": [31, 55, 56], "total_num": 45, "total_num_": 45, "tr": 4, "tr_axi": 4, "train": 0, "train_data": 0, "trainable_param": 0, "transform": 0, "transpos": 58, "true": [0, 4, 10, 13, 14, 30, 41, 51], "twod": 0, "txt": [0, 60, 62], "type_s": [24, 33, 48], "type_size_": [47, 48], "typecast": 0, "typedef": [12, 20, 21, 25, 38, 41, 45, 46, 47, 48, 50], "u_t": 51, "uint8": [0, 58], "uniformr": 58, "uniqu": 58, "unit": [10, 36, 37], "unsortedsegmentsum": 58, "unsqueez": [4, 27, 57, 58], "unstack": 58, "updat": 50, "update_t": 13, "updates_data": 50, "user": 62, "v3": 4, "v_": 11, "v_t": 11, "valu": [33, 43], "var": 11, "var_": 11, "var_t": 11, "varianc": [36, 37], "variance_c": 36, "vec": 0, "vecatan": 58, "vector": [0, 60], "version": 62, "vision": 0, "void": [0, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 60], "vr": 4, "vscode": 60, "w_": [17, 20, 21, 38, 41, 51, 54], "w_t": 51, "weight": [11, 12, 13, 20, 21, 51], "weight_": 13, "weight_decay": 51, "weight_g": 38, "weight_h": 41, "weight_i": 41, "weight_r": 38, "weight_shap": [12, 20, 21], "weight_t": 13, "where": [4, 32, 58], "whl": 62, "width": [0, 17, 18, 26, 52, 53], "win_amd64": 62, "window": [16, 60, 62], "window_h": 16, "window_w": 16, "workspac": [12, 20], "workspace_": [12, 20, 21, 22, 23], "www": 62, "x1": [0, 25, 60], "x1_tensor": 60, "x2": [25, 60], "x2_tensor": 60, "x86": 62, "x_": [29, 36, 37], "x_h": 16, "x_lefts_": [25, 46], "x_rights_": [25, 46], "x_t": [38, 41], "x_tensor": [6, 7, 8], "x_w": 16, "x_weights_": [25, 46], "xxx": [60, 62, 67], "xxx_add_sub_xxx": 64, "xxx_add_xxx": 64, "xxx_mul_xxx": 64, "xxx_xxx_p": 64, "xxx_xxx_s": 64, "xxx_xxx_xxx": 64, "xz": 62, "y1": 25, "y2": 25, "y_": [36, 37], "y_bottoms_": [25, 46], "y_h": 16, "y_k": 29, "y_tops_": [25, 46], "y_w": 16, "y_weights_": [25, 46], "yhft": [0, 4, 60], "z_t": 38, "zeroslik": 58, "zoneout": 41, "zoneout_cell_": 41, "zoneout_hidden_": 41, "zsu_23_4": 0}, "titles": ["GRAY_CNN(\u56fe\u7247\u7070\u5ea6\u5316\u5904\u7406+\u56fe\u7247\u8bc6\u522b)", "AI+DSP\u5e94\u7528\u793a\u4f8b", "AI\u8f85\u52a9\u5f00\u53d1\u793a\u4f8b", "DSP\u5e94\u7528\u793a\u4f8b", "RDSAR\uff08\u8ddd\u79bb-\u591a\u666e\u52d2SAR\u6210\u50cf\u7b97\u6cd5\uff09", "\u5e94\u7528\u5f00\u53d1\u793a\u4f8b", "ComplexAbs", "FFT", "IFFT", "\u81ea\u5b9a\u4e49\u7b97\u5b50\u5217\u8868", "Activation", "AdamWeightDecay", "Adder", "ApplyMomentum", "Assert", "Attention", "AvgPoolingGrad", "BatchToSpace", "BatchToSpaceND", "BroadcastTo", "Conv2d", "Conv2dTranspose", "Conv2DBackpropFilterFusion", "Conv2DBackpropInputFusion", "Crop", "CropAndResize", "DepthToSpace", "DSP Library C API Reference", "Eltwise", "EmbeddingLookup", "Equal", "ExpandDims", "ExpFusion", "FillV2", "Floor", "FloorDiv", "FusedBatchNorm", "GroupNormFusion", "GRU", "LeakyReLu", "LinSpace", "LSTM", "MatMulFusion", "RaggedRange", "Range", "Reduce", "Resize", "ReverseSequence", "ReverseV2", "ScaleFusion", "ScatterElements", "SGD", "SpaceToBatch", "SpaceToBatchND", "SpaceToDepth", "Squeeze", "UnSqueeze", "\u7b97\u5b50\u5e93\u652f\u6301", "\u7b97\u5b50\u5e93\u652f\u6301\u60c5\u51b5", "MindSpore Signal+ \u4f7f\u7528\u624b\u518c", "HelloDSP", "\u5feb\u901f\u5165\u95e8", "\u73af\u5883\u5b89\u88c5", "\u6574\u4f53\u6982\u89c8", "DSP\u7b97\u5b50C\u63a5\u53e3\u547d\u540d\u89c4\u8303", "\u53c2\u8003\u8d44\u6599", "\u5b98\u65b9\u8d44\u6599", "MindSpore Signal+ \u8c03\u5ea6\u65b9\u6848"], "titleterms": {"activ": 10, "adamweightdecay": 11, "adder": 12, "ai": [1, 2], "api": 27, "applymomentum": 13, "assert": 14, "attent": 15, "avgpoolinggrad": 16, "batchtospac": 17, "batchtospacend": 18, "broadcastto": 19, "cc": 60, "cmake": 62, "complexab": 6, "conda": 62, "conv2d": 20, "conv2dbackpropfilterfus": 22, "conv2dbackpropinputfus": 23, "conv2dtranspos": 21, "crop": 24, "cropandres": 25, "depthtospac": 26, "dsp": [1, 3, 27, 64], "eltwis": 28, "embeddinglookup": 29, "equal": 30, "expanddim": 31, "expfus": 32, "fft": 7, "fillv2": 33, "floor": 34, "floordiv": 35, "fusedbatchnorm": 36, "gray_cnn": 0, "groupnormfus": 37, "gru": 38, "hellodsp": 60, "ide": 62, "ifft": 8, "leakyrelu": 39, "librari": 27, "linspac": 40, "lite": 60, "lstm": 41, "main": 60, "matlab": 4, "matmulfus": 42, "mindradar": 62, "mindspor": [4, 59, 60, 62, 67], "mt7004": 60, "netron": 62, "python": [0, 60, 62], "radar": 62, "raggedrang": 43, "rang": 44, "rdsar": 4, "reduc": 45, "refer": 27, "resiz": 46, "reversesequ": 47, "reversev2": 48, "sar": 4, "scalefus": 49, "scatterel": 50, "sgd": 51, "signal": [4, 59, 67], "spacetobatch": 52, "spacetobatchnd": 53, "spacetodepth": 54, "squeez": 55, "unsqueez": 56, "yhft": 62}}) \ No newline at end of file +Search.setIndex({"alltitles": {"1. \u57fa\u672c\u539f\u5219": [[235, "id1"]], "1. \u5b9a\u4e49\u6a21\u578b": [[0, "id3"]], "1. \u6570\u636e\u8bfb\u53d6": [[4, "id2"]], "1. \u65b0\u5efa\u5de5\u7a0b": [[231, "id4"]], "1.1 \u4e0b\u8f7d\u5b89\u88c5\u5305": [[233, "id3"]], "1.1 \u65b0\u5efaPython\u6587\u4ef6": [[231, "id1"]], "1.2 \u5f00\u59cb\u5b89\u88c5": [[233, "id4"]], "1.2 \u7f16\u5199Python\u4ee3\u7801": [[231, "id2"]], "1.3 \u8fd0\u884cPython\u4ee3\u7801": [[231, "id3"]], "1.3 \u9a8c\u8bc1\u5b89\u88c5": [[233, "id5"]], "1.MindSpore Python\u7aef": [[231, "mindspore-python"]], "1.\u5b89\u88c5\u5305\u65b9\u5f0f\u5b89\u88c5": [[233, "id2"]], "2. \u5177\u4f53\u547d\u540d\u793a\u4f8b": [[235, "id2"]], "2. \u5de5\u7a0b\u76ee\u5f55\u7ed3\u6784": [[231, "id5"]], "2. \u6570\u636e\u9884\u5904\u7406": [[4, "id3"]], "2. \u6a21\u578b\u8bad\u7ec3": [[0, "model-training"]], "2.1 \u52a0\u6cd5\u7b97\u5b50": [[235, "id3"]], "2.Conda\u65b9\u5f0f\u5b89\u88c5": [[233, "conda"]], "3. \u5176\u4ed6\u6ce8\u610f\u4e8b\u9879": [[235, "id4"]], "3. \u66f4\u6539\u8f93\u5165\u6570\u636e": [[231, "id6"]], "3. \u6838\u5fc3\u8ba1\u7b97": [[4, "id4"]], "3. \u91cd\u65b0\u7ec4\u7f51": [[0, "id5"]], "4. main.cc \u529f\u80fd\u4ecb\u7ecd": [[231, "main-cc"]], "4. \u6570\u636e\u540e\u5904\u7406": [[4, "id5"]], "4. \u8f6c\u6362\u6a21\u578b": [[0, "id6"]], "4.1 \u8bfb\u53d6\u6a21\u578b\u6587\u4ef6": [[231, "id7"]], "4.2 \u8bbe\u7f6e\u8fd0\u884c\u540e\u7aef": [[231, "id8"]], "4.3 \u7f16\u8bd1\u6a21\u578b\u56fe": [[231, "id9"]], "4.4 \u83b7\u53d6\u6a21\u578b\u8f93\u5165": [[231, "id10"]], "4.5 \u6267\u884c\u6a21\u578b\u63a8\u7406": [[231, "id11"]], "4.6 \u83b7\u53d6\u6a21\u578b\u7ed3\u679c": [[231, "id12"]], "5. \u7f16\u8bd1\u5de5\u7a0b": [[231, "id13"]], "5. \u90e8\u7f72\u548c\u8fd0\u884c\u7a0b\u5e8f": [[0, "id7"]], "6. \u8fd0\u884c\u5de5\u7a0b": [[231, "id14"]], "6.1 \u8fde\u63a5 MT7004 \u677f\u5361": [[231, "mt7004"]], "6.2 \u90e8\u7f72\u548c\u8fd0\u884c\u7a0b\u5e8f": [[231, "id15"]], "AI+DSP\u5e94\u7528\u793a\u4f8b": [[1, null]], "AI\u8f85\u52a9\u5f00\u53d1\u793a\u4f8b": [[2, null]], "Abs": [[10, null]], "Absgrad": [[11, null]], "Activation": [[12, null]], "ActivationGrad": [[13, null]], "Adam": [[14, null]], "AdamWeightDecay": [[15, null]], "AddN": [[19, null]], "Adder": [[16, null]], "Addfusion": [[17, null]], "Addgrad": [[18, null]], "Affine": [[20, null]], "All": [[21, null]], "AllGather": [[22, null]], "ApplyMomentum": [[23, null]], "Argmax": [[24, null]], "Argmin": [[25, null]], "Assert": [[26, null]], "Assign": [[27, null]], "AssignAdd": [[28, null]], "Attention": [[29, null]], "AudioSpectrogram": [[30, null]], "AvgPoolingGrad": [[32, null]], "Avgpooling": [[31, null]], "BatchNorm": [[33, null]], "BatchToSpace": [[35, null]], "BatchToSpaceND": [[36, null]], "Batchnormgrad": [[34, null]], "Biasadd": [[37, null]], "Biasaddgrad": [[38, null]], "Binarycrossentropy": [[39, null]], "Binarycrossentropygrad": [[40, null]], "BroadcastTo": [[41, null]], "Cast": [[42, null]], "Ceil": [[43, null]], "Clip": [[44, null]], "ComplexAbs": [[6, null]], "Concat": [[45, null]], "ConstantOfShape": [[46, null]], "Conv2DBackpropFilterFusion": [[49, null]], "Conv2DBackpropInputFusion": [[50, null]], "Conv2d": [[47, null]], "Conv2dTranspose": [[48, null]], "Cos": [[51, null]], "Crop": [[52, null]], "CropAndResize": [[53, null]], "Cumsum": [[54, null]], "CustomExtractFeatures": [[55, null]], "CustomPredict": [[57, null]], "Customnormalize": [[56, null]], "DSP Library C API Reference": [[65, null]], "DSP\u5e94\u7528\u793a\u4f8b": [[3, null]], "DSP\u7b97\u5b50C\u63a5\u53e3\u547d\u540d\u89c4\u8303": [[235, null]], "DeconvGradFilter": [[58, null]], "DepthToSpace": [[59, null]], "DetectionPostProcess": [[60, null]], "DivFusion": [[61, null]], "Divgrad": [[62, null]], "Dropout": [[63, null]], "Dropoutgrad": [[64, null]], "DynamicQuant": [[66, null]], "Eltwise": [[67, null]], "Elu": [[68, null]], "EmbeddingLookup": [[69, null]], "Equal": [[70, null]], "Erf": [[71, null]], "ExpFusion": [[73, null]], "ExpandDims": [[72, null]], "FFT": [[7, null]], "FFTImag": [[76, null]], "FFTReal": [[77, null]], "FakeQuantWithMinMaxVars": [[74, null]], "FakeQuantWithMinMaxVarsPerChannel": [[75, null]], "Fill": [[78, null]], "FillV2": [[79, null]], "Flatten": [[80, null]], "FlattenGrad": [[81, null]], "Floor": [[82, null]], "FloorDiv": [[83, null]], "Floormod": [[84, null]], "FormatTranspose": [[85, null]], "FullConnection": [[86, null]], "FusedBatchNorm": [[87, null]], "GLU": [[91, null]], "GRAY_CNN(\u56fe\u7247\u7070\u5ea6\u5316\u5904\u7406+\u56fe\u7247\u8bc6\u522b)": [[0, null]], "GRU": [[95, null]], "Gather": [[88, null]], "GatherD": [[90, null]], "GatherNd": [[89, null]], "Greater": [[92, null]], "Greaterequal": [[93, null]], "GroupNormFusion": [[94, null]], "HashtableLookup": [[96, null]], "HelloDSP": [[231, null]], "IFFT": [[8, null]], "InstanceNorm": [[97, null]], "InvertPermutation": [[98, null]], "Isfinite": [[99, null]], "L2norm": [[100, null]], "LSTM": [[117, null]], "LayerNormFusion": [[101, null]], "Layernormgrad": [[102, null]], "LeakyReLu": [[103, null]], "Less": [[104, null]], "Lessequal": [[105, null]], "LinSpace": [[106, null]], "Log": [[107, null]], "Log1p": [[108, null]], "LogGrad": [[109, null]], "LogSoftmax": [[113, null]], "LogicalAnd": [[112, null]], "LogicalNot": [[110, null]], "LogicalOr": [[111, null]], "LpNormalization": [[114, null]], "Lrn": [[115, null]], "LshProjection": [[116, null]], "LstmGrad": [[118, null]], "LstmGradData": [[119, null]], "LstmGradWeight": [[120, null]], "MATLAB \u5b9e\u73b0": [[4, "matlab"]], "MFCC": [[126, null]], "MatMulFusion": [[121, null]], "MaxPoolFusion": [[124, null]], "MaxPoolGrad": [[125, null]], "Maximum": [[122, null]], "Maximumgrad": [[123, null]], "MindSpore Lite\u7aef": [[231, "mindspore-lite"]], "MindSpore Signal+ \u4f7f\u7528\u624b\u518c": [[230, null], [239, null]], "MindSpore Signal+ \u5b9e\u73b0": [[4, "mindspore-signal"]], "MindSpore Signal+ \u8c03\u5ea6\u65b9\u6848": [[238, null]], "Minimum": [[127, null]], "Minimumgrad": [[128, null]], "Mod": [[129, null]], "Mul": [[130, null]], "Mulgrad": [[131, null]], "NLLLoss": [[134, null]], "NLLLossGrad": [[135, null]], "Neg": [[132, null]], "NegGrad": [[133, null]], "NonMaxSuppression": [[136, null]], "NonZero": [[137, null]], "NotEqual": [[138, null]], "OneHot": [[139, null]], "OnesLike": [[140, null]], "PReLUFusion": [[144, null]], "PadFusion": [[141, null]], "PowFusion": [[142, null]], "PowerGrad": [[143, null]], "Priorbox": [[145, null]], "QuantDTypeCast": [[146, null]], "RDSAR\uff08\u8ddd\u79bb-\u591a\u666e\u52d2SAR\u6210\u50cf\u7b97\u6cd5\uff09": [[4, null]], "RFFT": [[161, null]], "ROIPooling": [[162, null]], "RaggedRange": [[147, null]], "RandomNormal": [[148, null]], "RandomStandardNormal": [[149, null]], "Range": [[150, null]], "Rank": [[151, null]], "RealDiv": [[152, null]], "Reciprocal": [[153, null]], "Reduce": [[154, null]], "ReduceScatter": [[155, null]], "Reshape": [[156, null]], "Resize": [[157, null]], "Resizegrad": [[158, null]], "ReverseSequence": [[159, null]], "ReverseV2": [[160, null]], "Round": [[163, null]], "Rsqrt": [[164, null]], "Rsqrtgrad": [[165, null]], "SGD": [[171, null]], "ScaleFusion": [[166, null]], "ScatterElements": [[167, null]], "ScatterNd": [[168, null]], "ScatterNdUpdate": [[169, null]], "Select": [[170, null]], "Shape": [[172, null]], "SigmoidCrossEntropyWithLogits": [[174, null]], "SigmoidCrossEntropyWithLogitsGrad": [[173, null]], "Sin": [[175, null]], "Size": [[176, null]], "Skipgram": [[177, null]], "Slice": [[178, null]], "SmoothL1Loss": [[179, null]], "Smoothl1lossgrad": [[180, null]], "Softmax": [[181, null]], "SoftmaxCrossEntropyWithLogits": [[182, null]], "SpaceToBatch": [[183, null]], "SpaceToBatchND": [[184, null]], "SpaceToDepth": [[185, null]], "SparseFillEmptyRows": [[187, null]], "SparseSoftmaxCrossEntropyWithLogits": [[186, null]], "Sparsereshape": [[188, null]], "Sparsesegmentsum": [[189, null]], "Sparsetodense": [[190, null]], "Splice": [[191, null]], "Split": [[192, null]], "SplitWithOverlap": [[193, null]], "Sqrt": [[194, null]], "Sqrtgrad": [[195, null]], "Square": [[196, null]], "Squaredifference": [[197, null]], "Squeeze": [[198, null]], "Stack": [[199, null]], "Stridedslice": [[200, null]], "Stridedslicegrad": [[201, null]], "Subfusion": [[202, null]], "Subgrad": [[203, null]], "Switch": [[204, null]], "Switchlayer": [[205, null]], "TensorScatterAdd": [[206, null]], "Tensorarray": [[207, null]], "TensorarrayWrite": [[209, null]], "Tensorarrayread": [[208, null]], "Tensorlistfromtensor": [[210, null]], "Tensorlistgetitem": [[211, null]], "Tensorlistreserve": [[212, null]], "Tensorlistsetitem": [[213, null]], "Tensorliststack": [[214, null]], "Tile": [[215, null]], "TopkFusion": [[216, null]], "Transpose": [[217, null]], "Tril": [[218, null]], "Triu": [[219, null]], "UnSqueeze": [[223, null]], "UniformReal": [[220, null]], "Unique": [[221, null]], "UnsortedSegmentSum": [[222, null]], "Unstack": [[224, null]], "Where": [[225, null]], "ZerosLike": [[226, null]], "python\u5b8c\u6574\u4ee3\u7801\u793a\u4f8b": [[0, "python"]], "\u521b\u5efa\u5e76\u8fdb\u5165Conda\u865a\u62df\u73af\u5883": [[233, "id6"]], "\u53c2\u8003\u4e0e\u6e90\u7801": [[4, "id7"]], "\u53c2\u8003\u8d44\u6599": [[236, null]], "\u5b89\u88c5CMake": [[233, "cmake"]], "\u5b89\u88c5MindRadar": [[233, "mindradar"]], "\u5b89\u88c5MindSpore": [[233, "mindspore"]], "\u5b89\u88c5MindSpore Radar \u4e0e\u4f9d\u8d56\u8f6f\u4ef6": [[233, "mindspore-radar"]], "\u5b89\u88c5Netron": [[233, "netron"]], "\u5b89\u88c5Python": [[233, "python"]], "\u5b89\u88c5YHFT-IDE": [[233, "yhft-ide"]], "\u5b98\u65b9\u8d44\u6599": [[237, null]], "\u5e94\u7528\u5f00\u53d1\u793a\u4f8b": [[5, null]], "\u5e94\u7528\u6982\u8ff0": [[0, "id1"]], "\u5f00\u53d1\u6d41\u7a0b": [[0, "id2"]], "\u5feb\u901f\u5165\u95e8": [[232, null]], "\u6574\u4f53\u6982\u89c8": [[234, null]], "\u677f\u5361\u90e8\u7f72": [[4, "id6"]], "\u73af\u5883\u5b89\u88c5": [[233, null]], "\u7b97\u5b50\u5e93\u652f\u6301": [[227, null], [228, null]], "\u7b97\u5b50\u5e93\u652f\u6301\u60c5\u51b5": [[229, null]], "\u7b97\u6cd5\u6982\u8ff0": [[4, "id1"]], "\u81ea\u5b9a\u4e49\u7b97\u5b50\u5217\u8868": [[9, null]], "\u914d\u7f6e\u4ea4\u53c9\u7f16\u8bd1\u5de5\u5177\u94fe": [[233, "toolchain"]]}, "docnames": ["appdevelop/ai_dsp/gray_cnn", "appdevelop/ai_dsp/index", "appdevelop/autocodegen/index", "appdevelop/dsp/index", "appdevelop/dsp/rdsar", "appdevelop/index", "functionlib/custom_op/complex_abs", "functionlib/custom_op/fft", "functionlib/custom_op/ifft", "functionlib/custom_op/index", "functionlib/dsplib/abs", "functionlib/dsplib/absgrad", "functionlib/dsplib/activation", "functionlib/dsplib/activation_grad", "functionlib/dsplib/adam", "functionlib/dsplib/adamweightdecay", "functionlib/dsplib/adder", "functionlib/dsplib/addfusion", "functionlib/dsplib/addgrad", "functionlib/dsplib/addn", "functionlib/dsplib/affine", "functionlib/dsplib/all", "functionlib/dsplib/allgather", "functionlib/dsplib/applymomentum", "functionlib/dsplib/argmax", "functionlib/dsplib/argmin", "functionlib/dsplib/assert", "functionlib/dsplib/assign", "functionlib/dsplib/assignadd", "functionlib/dsplib/attention", "functionlib/dsplib/audio_spectrogram", "functionlib/dsplib/avgpooling", "functionlib/dsplib/avgpoolinggrad", "functionlib/dsplib/batchnorm", "functionlib/dsplib/batchnormgrad", "functionlib/dsplib/batchtospace", "functionlib/dsplib/batchtospacend", "functionlib/dsplib/biasadd", "functionlib/dsplib/biasaddgrad", "functionlib/dsplib/binarycrossentropy", "functionlib/dsplib/binarycrossentropygrad", "functionlib/dsplib/broadcastto", "functionlib/dsplib/cast", "functionlib/dsplib/ceil", "functionlib/dsplib/clip", "functionlib/dsplib/concat", "functionlib/dsplib/constant_of_shape", "functionlib/dsplib/conv2d", "functionlib/dsplib/conv2d_transpose", "functionlib/dsplib/conv2dbackpropfilterfusion", "functionlib/dsplib/conv2dbackpropinputfusion", "functionlib/dsplib/cos", "functionlib/dsplib/crop", "functionlib/dsplib/crop_and_resize", "functionlib/dsplib/cumsum", "functionlib/dsplib/customextractfeatures", "functionlib/dsplib/customnormalize", "functionlib/dsplib/custompredict", "functionlib/dsplib/deconvgradfilter", "functionlib/dsplib/depthtospace", "functionlib/dsplib/detection_post_process", "functionlib/dsplib/div_fusion", "functionlib/dsplib/divgrad", "functionlib/dsplib/dropout", "functionlib/dsplib/dropoutgrad", "functionlib/dsplib/dsplib_index", "functionlib/dsplib/dynamicquant", "functionlib/dsplib/eltwise", "functionlib/dsplib/elu", "functionlib/dsplib/embeddinglookup", "functionlib/dsplib/equal", "functionlib/dsplib/erf", "functionlib/dsplib/expand_dims", "functionlib/dsplib/expfusion", "functionlib/dsplib/fake_quant_with_min_max_vars", "functionlib/dsplib/fake_quant_with_min_max_vars_per_channel", "functionlib/dsplib/fftimag", "functionlib/dsplib/fftreal", "functionlib/dsplib/fill", "functionlib/dsplib/fillv2", "functionlib/dsplib/flatten", "functionlib/dsplib/flattengrad", "functionlib/dsplib/floor", "functionlib/dsplib/floordiv", "functionlib/dsplib/floormod", "functionlib/dsplib/formattranspose", "functionlib/dsplib/fullconnection", "functionlib/dsplib/fusedbatchnorm", "functionlib/dsplib/gather", "functionlib/dsplib/gather_nd", "functionlib/dsplib/gatherd", "functionlib/dsplib/glu", "functionlib/dsplib/greater", "functionlib/dsplib/greaterequal", "functionlib/dsplib/groupnormfusion", "functionlib/dsplib/gru", "functionlib/dsplib/hashtablelookup", "functionlib/dsplib/instancenorm", "functionlib/dsplib/invertpermutation", "functionlib/dsplib/isfinite", "functionlib/dsplib/l2norm", "functionlib/dsplib/layernormfusion", "functionlib/dsplib/layernormgrad", "functionlib/dsplib/leaky_relu", "functionlib/dsplib/less", "functionlib/dsplib/lessequal", "functionlib/dsplib/linspace", "functionlib/dsplib/log", "functionlib/dsplib/log1p", "functionlib/dsplib/loggrad", "functionlib/dsplib/logical_not", "functionlib/dsplib/logical_or", "functionlib/dsplib/logicaland", "functionlib/dsplib/logsoftmax", "functionlib/dsplib/lpnormalization", "functionlib/dsplib/lrn", "functionlib/dsplib/lsh_projection", "functionlib/dsplib/lstm", "functionlib/dsplib/lstmgrad", "functionlib/dsplib/lstmgraddata", "functionlib/dsplib/lstmgradweight", "functionlib/dsplib/matmulfusion", "functionlib/dsplib/maximum", "functionlib/dsplib/maximumgrad", "functionlib/dsplib/maxpoolfusion", "functionlib/dsplib/maxpoolgrad", "functionlib/dsplib/mfcc", "functionlib/dsplib/minimum", "functionlib/dsplib/minimumgrad", "functionlib/dsplib/mod", "functionlib/dsplib/mul", "functionlib/dsplib/mulgrad", "functionlib/dsplib/neg", "functionlib/dsplib/neg_grad", "functionlib/dsplib/nllloss", "functionlib/dsplib/nlllossgrad", "functionlib/dsplib/non_max_suppression", "functionlib/dsplib/nonzero", "functionlib/dsplib/not_equal", "functionlib/dsplib/onehot", "functionlib/dsplib/ones_like", "functionlib/dsplib/padfusion", "functionlib/dsplib/pow_fusion", "functionlib/dsplib/power_grad", "functionlib/dsplib/prelufusion", "functionlib/dsplib/priorbox", "functionlib/dsplib/quantdtypecast", "functionlib/dsplib/raggedrange", "functionlib/dsplib/random_normal", "functionlib/dsplib/random_standard_normal", "functionlib/dsplib/range", "functionlib/dsplib/rank", "functionlib/dsplib/real_div", "functionlib/dsplib/reciprocal", "functionlib/dsplib/reduce", "functionlib/dsplib/reducescatter", "functionlib/dsplib/reshape", "functionlib/dsplib/resize", "functionlib/dsplib/resizegrad", "functionlib/dsplib/reverse_sequence", "functionlib/dsplib/reversev2", "functionlib/dsplib/rfft", "functionlib/dsplib/roipooling", "functionlib/dsplib/round", "functionlib/dsplib/rsqrt", "functionlib/dsplib/rsqrtgrad", "functionlib/dsplib/scalefusion", "functionlib/dsplib/scatter_elements", "functionlib/dsplib/scatter_nd", "functionlib/dsplib/scatter_nd_update", "functionlib/dsplib/select", "functionlib/dsplib/sgd", "functionlib/dsplib/shape", "functionlib/dsplib/sigmoidcrossentropwithlogitsgrad", "functionlib/dsplib/sigmoidcrossentropywithlogits", "functionlib/dsplib/sin", "functionlib/dsplib/size", "functionlib/dsplib/skipgram", "functionlib/dsplib/slice", "functionlib/dsplib/smooth1loss", "functionlib/dsplib/smoothl1lossgrad", "functionlib/dsplib/softmax", "functionlib/dsplib/softmax_cross_entropy_with_logits", "functionlib/dsplib/spacetobatch", "functionlib/dsplib/spacetobatchnd", "functionlib/dsplib/spacetodepth", "functionlib/dsplib/sparse_softmax_cross_entropy_with_logits", "functionlib/dsplib/sparsefillemptyrows", "functionlib/dsplib/sparsereshape", "functionlib/dsplib/sparsesegmentsum", "functionlib/dsplib/sparsetodense", "functionlib/dsplib/splice", "functionlib/dsplib/split", "functionlib/dsplib/split_with_overlap", "functionlib/dsplib/sqrt", "functionlib/dsplib/sqrtgrad", "functionlib/dsplib/square", "functionlib/dsplib/squaredifference", "functionlib/dsplib/squeeze", "functionlib/dsplib/stack", "functionlib/dsplib/stridedslice", "functionlib/dsplib/stridedslicegrad", "functionlib/dsplib/subfusion", "functionlib/dsplib/subgrad", "functionlib/dsplib/switch", "functionlib/dsplib/switchlayer", "functionlib/dsplib/tensor_scatter_add", "functionlib/dsplib/tensorarray", "functionlib/dsplib/tensorarrayread", "functionlib/dsplib/tensorarraywrite", "functionlib/dsplib/tensorlistfromtensor", "functionlib/dsplib/tensorlistgetitem", "functionlib/dsplib/tensorlistreserve", "functionlib/dsplib/tensorlistsetitem", "functionlib/dsplib/tensorliststack", "functionlib/dsplib/tile", "functionlib/dsplib/topkfusion", "functionlib/dsplib/transpose", "functionlib/dsplib/tril", "functionlib/dsplib/triu", "functionlib/dsplib/uniform_real", "functionlib/dsplib/unique", "functionlib/dsplib/unsortedsegmentsum", "functionlib/dsplib/unsqueeze", "functionlib/dsplib/unstack", "functionlib/dsplib/where", "functionlib/dsplib/zeroslike", "functionlib/index", "functionlib/rstfiles", "functionlib/supported_op", "index", "quickstart/hellodsp", "quickstart/index", "quickstart/installation", "quickstart/overview", "refdoc/dsplib_description", "refdoc/index", "refdoc/mindspore", "refdoc/signal_scheduling", "rstfiles"], "envversion": {"sphinx": 64, "sphinx.domains.c": 3, "sphinx.domains.changeset": 1, "sphinx.domains.citation": 1, "sphinx.domains.cpp": 9, "sphinx.domains.index": 1, "sphinx.domains.javascript": 3, "sphinx.domains.math": 2, "sphinx.domains.python": 4, "sphinx.domains.rst": 2, "sphinx.domains.std": 2}, "filenames": ["appdevelop/ai_dsp/gray_cnn.rst", "appdevelop/ai_dsp/index.rst", "appdevelop/autocodegen/index.rst", "appdevelop/dsp/index.rst", "appdevelop/dsp/rdsar.md", "appdevelop/index.rst", "functionlib/custom_op/complex_abs.rst", "functionlib/custom_op/fft.rst", "functionlib/custom_op/ifft.rst", "functionlib/custom_op/index.rst", "functionlib/dsplib/abs.rst", "functionlib/dsplib/absgrad.rst", "functionlib/dsplib/activation.rst", "functionlib/dsplib/activation_grad.rst", "functionlib/dsplib/adam.rst", "functionlib/dsplib/adamweightdecay.rst", "functionlib/dsplib/adder.rst", "functionlib/dsplib/addfusion.rst", "functionlib/dsplib/addgrad.rst", "functionlib/dsplib/addn.rst", "functionlib/dsplib/affine.rst", "functionlib/dsplib/all.rst", "functionlib/dsplib/allgather.rst", "functionlib/dsplib/applymomentum.rst", "functionlib/dsplib/argmax.rst", "functionlib/dsplib/argmin.rst", "functionlib/dsplib/assert.rst", "functionlib/dsplib/assign.rst", "functionlib/dsplib/assignadd.rst", "functionlib/dsplib/attention.rst", "functionlib/dsplib/audio_spectrogram.rst", "functionlib/dsplib/avgpooling.rst", "functionlib/dsplib/avgpoolinggrad.rst", "functionlib/dsplib/batchnorm.rst", "functionlib/dsplib/batchnormgrad.rst", "functionlib/dsplib/batchtospace.rst", "functionlib/dsplib/batchtospacend.rst", "functionlib/dsplib/biasadd.rst", "functionlib/dsplib/biasaddgrad.rst", "functionlib/dsplib/binarycrossentropy.rst", "functionlib/dsplib/binarycrossentropygrad.rst", "functionlib/dsplib/broadcastto.rst", "functionlib/dsplib/cast.rst", "functionlib/dsplib/ceil.rst", "functionlib/dsplib/clip.rst", "functionlib/dsplib/concat.rst", "functionlib/dsplib/constant_of_shape.rst", "functionlib/dsplib/conv2d.rst", "functionlib/dsplib/conv2d_transpose.rst", "functionlib/dsplib/conv2dbackpropfilterfusion.rst", "functionlib/dsplib/conv2dbackpropinputfusion.rst", "functionlib/dsplib/cos.rst", "functionlib/dsplib/crop.rst", "functionlib/dsplib/crop_and_resize.rst", "functionlib/dsplib/cumsum.rst", "functionlib/dsplib/customextractfeatures.rst", "functionlib/dsplib/customnormalize.rst", "functionlib/dsplib/custompredict.rst", "functionlib/dsplib/deconvgradfilter.rst", "functionlib/dsplib/depthtospace.rst", "functionlib/dsplib/detection_post_process.rst", "functionlib/dsplib/div_fusion.rst", "functionlib/dsplib/divgrad.rst", "functionlib/dsplib/dropout.rst", "functionlib/dsplib/dropoutgrad.rst", "functionlib/dsplib/dsplib_index.rst", "functionlib/dsplib/dynamicquant.rst", "functionlib/dsplib/eltwise.rst", "functionlib/dsplib/elu.rst", "functionlib/dsplib/embeddinglookup.rst", "functionlib/dsplib/equal.rst", "functionlib/dsplib/erf.rst", "functionlib/dsplib/expand_dims.rst", "functionlib/dsplib/expfusion.rst", "functionlib/dsplib/fake_quant_with_min_max_vars.rst", "functionlib/dsplib/fake_quant_with_min_max_vars_per_channel.rst", "functionlib/dsplib/fftimag.rst", "functionlib/dsplib/fftreal.rst", "functionlib/dsplib/fill.rst", "functionlib/dsplib/fillv2.rst", "functionlib/dsplib/flatten.rst", "functionlib/dsplib/flattengrad.rst", "functionlib/dsplib/floor.rst", "functionlib/dsplib/floordiv.rst", "functionlib/dsplib/floormod.rst", "functionlib/dsplib/formattranspose.rst", "functionlib/dsplib/fullconnection.rst", "functionlib/dsplib/fusedbatchnorm.rst", "functionlib/dsplib/gather.rst", "functionlib/dsplib/gather_nd.rst", "functionlib/dsplib/gatherd.rst", "functionlib/dsplib/glu.rst", "functionlib/dsplib/greater.rst", "functionlib/dsplib/greaterequal.rst", "functionlib/dsplib/groupnormfusion.rst", "functionlib/dsplib/gru.rst", "functionlib/dsplib/hashtablelookup.rst", "functionlib/dsplib/instancenorm.rst", "functionlib/dsplib/invertpermutation.rst", "functionlib/dsplib/isfinite.rst", "functionlib/dsplib/l2norm.rst", "functionlib/dsplib/layernormfusion.rst", "functionlib/dsplib/layernormgrad.rst", "functionlib/dsplib/leaky_relu.rst", "functionlib/dsplib/less.rst", "functionlib/dsplib/lessequal.rst", "functionlib/dsplib/linspace.rst", "functionlib/dsplib/log.rst", "functionlib/dsplib/log1p.rst", "functionlib/dsplib/loggrad.rst", "functionlib/dsplib/logical_not.rst", "functionlib/dsplib/logical_or.rst", "functionlib/dsplib/logicaland.rst", "functionlib/dsplib/logsoftmax.rst", "functionlib/dsplib/lpnormalization.rst", "functionlib/dsplib/lrn.rst", "functionlib/dsplib/lsh_projection.rst", "functionlib/dsplib/lstm.rst", "functionlib/dsplib/lstmgrad.rst", "functionlib/dsplib/lstmgraddata.rst", "functionlib/dsplib/lstmgradweight.rst", "functionlib/dsplib/matmulfusion.rst", "functionlib/dsplib/maximum.rst", "functionlib/dsplib/maximumgrad.rst", "functionlib/dsplib/maxpoolfusion.rst", "functionlib/dsplib/maxpoolgrad.rst", "functionlib/dsplib/mfcc.rst", "functionlib/dsplib/minimum.rst", "functionlib/dsplib/minimumgrad.rst", "functionlib/dsplib/mod.rst", "functionlib/dsplib/mul.rst", "functionlib/dsplib/mulgrad.rst", "functionlib/dsplib/neg.rst", "functionlib/dsplib/neg_grad.rst", "functionlib/dsplib/nllloss.rst", "functionlib/dsplib/nlllossgrad.rst", "functionlib/dsplib/non_max_suppression.rst", "functionlib/dsplib/nonzero.rst", "functionlib/dsplib/not_equal.rst", "functionlib/dsplib/onehot.rst", "functionlib/dsplib/ones_like.rst", "functionlib/dsplib/padfusion.rst", "functionlib/dsplib/pow_fusion.rst", "functionlib/dsplib/power_grad.rst", "functionlib/dsplib/prelufusion.rst", "functionlib/dsplib/priorbox.rst", "functionlib/dsplib/quantdtypecast.rst", "functionlib/dsplib/raggedrange.rst", "functionlib/dsplib/random_normal.rst", "functionlib/dsplib/random_standard_normal.rst", "functionlib/dsplib/range.rst", "functionlib/dsplib/rank.rst", "functionlib/dsplib/real_div.rst", "functionlib/dsplib/reciprocal.rst", "functionlib/dsplib/reduce.rst", "functionlib/dsplib/reducescatter.rst", "functionlib/dsplib/reshape.rst", "functionlib/dsplib/resize.rst", "functionlib/dsplib/resizegrad.rst", "functionlib/dsplib/reverse_sequence.rst", "functionlib/dsplib/reversev2.rst", "functionlib/dsplib/rfft.rst", "functionlib/dsplib/roipooling.rst", "functionlib/dsplib/round.rst", "functionlib/dsplib/rsqrt.rst", "functionlib/dsplib/rsqrtgrad.rst", "functionlib/dsplib/scalefusion.rst", "functionlib/dsplib/scatter_elements.rst", "functionlib/dsplib/scatter_nd.rst", "functionlib/dsplib/scatter_nd_update.rst", "functionlib/dsplib/select.rst", "functionlib/dsplib/sgd.rst", "functionlib/dsplib/shape.rst", "functionlib/dsplib/sigmoidcrossentropwithlogitsgrad.rst", "functionlib/dsplib/sigmoidcrossentropywithlogits.rst", "functionlib/dsplib/sin.rst", "functionlib/dsplib/size.rst", "functionlib/dsplib/skipgram.rst", "functionlib/dsplib/slice.rst", "functionlib/dsplib/smooth1loss.rst", "functionlib/dsplib/smoothl1lossgrad.rst", "functionlib/dsplib/softmax.rst", "functionlib/dsplib/softmax_cross_entropy_with_logits.rst", "functionlib/dsplib/spacetobatch.rst", "functionlib/dsplib/spacetobatchnd.rst", "functionlib/dsplib/spacetodepth.rst", "functionlib/dsplib/sparse_softmax_cross_entropy_with_logits.rst", "functionlib/dsplib/sparsefillemptyrows.rst", "functionlib/dsplib/sparsereshape.rst", "functionlib/dsplib/sparsesegmentsum.rst", "functionlib/dsplib/sparsetodense.rst", "functionlib/dsplib/splice.rst", "functionlib/dsplib/split.rst", "functionlib/dsplib/split_with_overlap.rst", "functionlib/dsplib/sqrt.rst", "functionlib/dsplib/sqrtgrad.rst", "functionlib/dsplib/square.rst", "functionlib/dsplib/squaredifference.rst", "functionlib/dsplib/squeeze.rst", "functionlib/dsplib/stack.rst", "functionlib/dsplib/stridedslice.rst", "functionlib/dsplib/stridedslicegrad.rst", "functionlib/dsplib/subfusion.rst", "functionlib/dsplib/subgrad.rst", "functionlib/dsplib/switch.rst", "functionlib/dsplib/switchlayer.rst", "functionlib/dsplib/tensor_scatter_add.rst", "functionlib/dsplib/tensorarray.rst", "functionlib/dsplib/tensorarrayread.rst", "functionlib/dsplib/tensorarraywrite.rst", "functionlib/dsplib/tensorlistfromtensor.rst", "functionlib/dsplib/tensorlistgetitem.rst", "functionlib/dsplib/tensorlistreserve.rst", "functionlib/dsplib/tensorlistsetitem.rst", "functionlib/dsplib/tensorliststack.rst", "functionlib/dsplib/tile.rst", "functionlib/dsplib/topkfusion.rst", "functionlib/dsplib/transpose.rst", "functionlib/dsplib/tril.rst", "functionlib/dsplib/triu.rst", "functionlib/dsplib/uniform_real.rst", "functionlib/dsplib/unique.rst", "functionlib/dsplib/unsortedsegmentsum.rst", "functionlib/dsplib/unsqueeze.rst", "functionlib/dsplib/unstack.rst", "functionlib/dsplib/where.rst", "functionlib/dsplib/zeroslike.rst", "functionlib/index.rst", "functionlib/rstfiles.rst", "functionlib/supported_op.md", "index.rst", "quickstart/hellodsp.md", "quickstart/index.rst", "quickstart/installation.md", "quickstart/overview.md", "refdoc/dsplib_description.md", "refdoc/index.rst", "refdoc/mindspore.md", "refdoc/signal_scheduling.md", "rstfiles.rst"], "indexentries": {"anytype_crop_anycore\uff08c function\uff09": [[52, "c.anytype_crop_anycore", false]], "anytype_expand_dims_anycore\uff08c function\uff09": [[72, "c.anytype_expand_dims_anycore", false]], "anytype_fillv2_p\uff08c function\uff09": [[79, "c.anytype_fillv2_p", false]], "anytype_fillv2_s\uff08c function\uff09": [[79, "c.anytype_fillv2_s", false]], "anytype_reverse_sequence_anycore\uff08c function\uff09": [[159, "c.anytype_reverse_sequence_anycore", false]], "anytype_reversev2_anycore\uff08c function\uff09": [[160, "c.anytype_reversev2_anycore", false]], "anytype_squeeze_anycore\uff08c function\uff09": [[198, "c.anytype_squeeze_anycore", false]], "anytype_unsqueeze_anycore\uff08c function\uff09": [[223, "c.anytype_unsqueeze_anycore", false]], "assert\uff08c function\uff09": [[26, "c.assert", false]], "c128_abs_p\uff08c function\uff09": [[10, "c.c128_abs_p", false]], "c128_abs_s\uff08c function\uff09": [[10, "c.c128_abs_s", false]], "c128_addext_p\uff08c function\uff09": [[17, "c.c128_addext_p", false]], "c128_addext_s\uff08c function\uff09": [[17, "c.c128_addext_s", false]], "c128_addn_p\uff08c function\uff09": [[19, "c.c128_addn_p", false]], "c128_addn_s\uff08c function\uff09": [[19, "c.c128_addn_s", false]], "c128_addrelu6_p\uff08c function\uff09": [[17, "c.c128_addrelu6_p", false]], "c128_addrelu6_s\uff08c function\uff09": [[17, "c.c128_addrelu6_s", false]], "c128_addrelu_p\uff08c function\uff09": [[17, "c.c128_addrelu_p", false]], "c128_addrelu_s\uff08c function\uff09": [[17, "c.c128_addrelu_s", false]], "c128_allgather_p\uff08c function\uff09": [[22, "c.c128_allgather_p", false]], "c128_allgather_s\uff08c function\uff09": [[22, "c.c128_allgather_s", false]], "c128_assign_p\uff08c function\uff09": [[27, "c.c128_assign_p", false]], "c128_assign_s\uff08c function\uff09": [[27, "c.c128_assign_s", false]], "c128_assignadd_p\uff08c function\uff09": [[28, "c.c128_assignadd_p", false]], "c128_assignadd_s\uff08c function\uff09": [[28, "c.c128_assignadd_s", false]], "c128_batchtospace_p\uff08c function\uff09": [[35, "c.c128_batchtospace_p", false]], "c128_batchtospace_s\uff08c function\uff09": [[35, "c.c128_batchtospace_s", false]], "c128_batchtospacend_p\uff08c function\uff09": [[36, "c.c128_batchtospacend_p", false]], "c128_batchtospacend_s\uff08c function\uff09": [[36, "c.c128_batchtospacend_s", false]], "c128_biasadd_p\uff08c function\uff09": [[37, "c.c128_biasadd_p", false]], "c128_biasadd_s\uff08c function\uff09": [[37, "c.c128_biasadd_s", false]], "c128_broadcastto_p\uff08c function\uff09": [[41, "c.c128_broadcastto_p", false]], "c128_broadcastto_s\uff08c function\uff09": [[41, "c.c128_broadcastto_s", false]], "c128_concat_p\uff08c function\uff09": [[45, "c.c128_concat_p", false]], "c128_concat_s\uff08c function\uff09": [[45, "c.c128_concat_s", false]], "c128_constant_of_shape_p\uff08c function\uff09": [[46, "c.c128_constant_of_shape_p", false]], "c128_constant_of_shape_s\uff08c function\uff09": [[46, "c.c128_constant_of_shape_s", false]], "c128_cumsum_p\uff08c function\uff09": [[54, "c.c128_cumsum_p", false]], "c128_cumsum_s\uff08c function\uff09": [[54, "c.c128_cumsum_s", false]], "c128_depthtospace_p\uff08c function\uff09": [[59, "c.c128_depthtospace_p", false]], "c128_depthtospace_s\uff08c function\uff09": [[59, "c.c128_depthtospace_s", false]], "c128_div_fusion_p\uff08c function\uff09": [[61, "c.c128_div_fusion_p", false]], "c128_div_fusion_s\uff08c function\uff09": [[61, "c.c128_div_fusion_s", false]], "c128_eltwise_p\uff08c function\uff09": [[67, "c.c128_eltwise_p", false]], "c128_eltwise_s\uff08c function\uff09": [[67, "c.c128_eltwise_s", false]], "c128_equal_p\uff08c function\uff09": [[70, "c.c128_equal_p", false]], "c128_equal_s\uff08c function\uff09": [[70, "c.c128_equal_s", false]], "c128_expfusion_p\uff08c function\uff09": [[73, "c.c128_expfusion_p", false]], "c128_expfusion_s\uff08c function\uff09": [[73, "c.c128_expfusion_s", false]], "c128_extract_features_p\uff08c function\uff09": [[55, "c.c128_extract_features_p", false]], "c128_extract_features_s\uff08c function\uff09": [[55, "c.c128_extract_features_s", false]], "c128_fill_p\uff08c function\uff09": [[78, "c.c128_fill_p", false]], "c128_fill_s\uff08c function\uff09": [[78, "c.c128_fill_s", false]], "c128_formattranspose_p\uff08c function\uff09": [[85, "c.c128_formattranspose_p", false]], "c128_formattranspose_s\uff08c function\uff09": [[85, "c.c128_formattranspose_s", false]], "c128_gather_nd_p\uff08c function\uff09": [[89, "c.c128_gather_nd_p", false]], "c128_gather_nd_s\uff08c function\uff09": [[89, "c.c128_gather_nd_s", false]], "c128_gather_p\uff08c function\uff09": [[88, "c.c128_gather_p", false]], "c128_gather_s\uff08c function\uff09": [[88, "c.c128_gather_s", false]], "c128_gatherd_p\uff08c function\uff09": [[90, "c.c128_gatherd_p", false]], "c128_gatherd_s\uff08c function\uff09": [[90, "c.c128_gatherd_s", false]], "c128_isfinite_p\uff08c function\uff09": [[99, "c.c128_isfinite_p", false]], "c128_isfinite_s\uff08c function\uff09": [[99, "c.c128_isfinite_s", false]], "c128_matmulfusion_p\uff08c function\uff09": [[121, "c.c128_matmulfusion_p", false]], "c128_matmulfusion_s\uff08c function\uff09": [[121, "c.c128_matmulfusion_s", false]], "c128_mul_p\uff08c function\uff09": [[130, "c.c128_mul_p", false]], "c128_mul_s\uff08c function\uff09": [[130, "c.c128_mul_s", false]], "c128_neg_grad_p\uff08c function\uff09": [[133, "c.c128_neg_grad_p", false]], "c128_neg_grad_s\uff08c function\uff09": [[133, "c.c128_neg_grad_s", false]], "c128_neg_p\uff08c function\uff09": [[132, "c.c128_neg_p", false]], "c128_neg_s\uff08c function\uff09": [[132, "c.c128_neg_s", false]], "c128_nonzero_p\uff08c function\uff09": [[137, "c.c128_nonzero_p", false]], "c128_nonzero_s\uff08c function\uff09": [[137, "c.c128_nonzero_s", false]], "c128_not_equal_p\uff08c function\uff09": [[138, "c.c128_not_equal_p", false]], "c128_not_equal_s\uff08c function\uff09": [[138, "c.c128_not_equal_s", false]], "c128_onehot_p\uff08c function\uff09": [[139, "c.c128_onehot_p", false]], "c128_onehot_s\uff08c function\uff09": [[139, "c.c128_onehot_s", false]], "c128_ones_like_p\uff08c function\uff09": [[140, "c.c128_ones_like_p", false]], "c128_ones_like_s\uff08c function\uff09": [[140, "c.c128_ones_like_s", false]], "c128_padfusion_p\uff08c function\uff09": [[141, "c.c128_padfusion_p", false]], "c128_padfusion_s\uff08c function\uff09": [[141, "c.c128_padfusion_s", false]], "c128_real_div_p\uff08c function\uff09": [[152, "c.c128_real_div_p", false]], "c128_real_div_s\uff08c function\uff09": [[152, "c.c128_real_div_s", false]], "c128_reciprocal_p\uff08c function\uff09": [[153, "c.c128_reciprocal_p", false]], "c128_reciprocal_s\uff08c function\uff09": [[153, "c.c128_reciprocal_s", false]], "c128_reduceall_p\uff08c function\uff09": [[21, "c.c128_reduceall_p", false]], "c128_reduceall_s\uff08c function\uff09": [[21, "c.c128_reduceall_s", false]], "c128_reshape_p\uff08c function\uff09": [[156, "c.c128_reshape_p", false]], "c128_reshape_s\uff08c function\uff09": [[156, "c.c128_reshape_s", false]], "c128_rfft_p\uff08c function\uff09": [[161, "c.c128_rfft_p", false]], "c128_rfft_s\uff08c function\uff09": [[161, "c.c128_rfft_s", false]], "c128_rsqrt_p\uff08c function\uff09": [[164, "c.c128_rsqrt_p", false]], "c128_rsqrt_s\uff08c function\uff09": [[164, "c.c128_rsqrt_s", false]], "c128_scatter_elements_p\uff08c function\uff09": [[167, "c.c128_scatter_elements_p", false]], "c128_scatter_elements_s\uff08c function\uff09": [[167, "c.c128_scatter_elements_s", false]], "c128_scatter_nd_p\uff08c function\uff09": [[168, "c.c128_scatter_nd_p", false]], "c128_scatter_nd_s\uff08c function\uff09": [[168, "c.c128_scatter_nd_s", false]], "c128_scatter_nd_update_p\uff08c function\uff09": [[169, "c.c128_scatter_nd_update_p", false]], "c128_scatter_nd_update_s\uff08c function\uff09": [[169, "c.c128_scatter_nd_update_s", false]], "c128_select_p\uff08c function\uff09": [[170, "c.c128_select_p", false]], "c128_select_s\uff08c function\uff09": [[170, "c.c128_select_s", false]], "c128_slice_p\uff08c function\uff09": [[178, "c.c128_slice_p", false]], "c128_slice_s\uff08c function\uff09": [[178, "c.c128_slice_s", false]], "c128_spacetobatch_p\uff08c function\uff09": [[183, "c.c128_spacetobatch_p", false]], "c128_spacetobatch_s\uff08c function\uff09": [[183, "c.c128_spacetobatch_s", false]], "c128_spacetobatchnd_p\uff08c function\uff09": [[184, "c.c128_spacetobatchnd_p", false]], "c128_spacetobatchnd_s\uff08c function\uff09": [[184, "c.c128_spacetobatchnd_s", false]], "c128_spacetodepth_p\uff08c function\uff09": [[185, "c.c128_spacetodepth_p", false]], "c128_spacetodepth_s\uff08c function\uff09": [[185, "c.c128_spacetodepth_s", false]], "c128_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.c128_sparsefillemptyrows_p", false]], "c128_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.c128_sparsefillemptyrows_s", false]], "c128_sparsesegmentsum_p\uff08c function\uff09": [[189, "c.c128_sparsesegmentsum_p", false]], "c128_sparsesegmentsum_s\uff08c function\uff09": [[189, "c.c128_sparsesegmentsum_s", false]], "c128_sparsetodense_p\uff08c function\uff09": [[190, "c.c128_sparsetodense_p", false]], "c128_sparsetodense_s\uff08c function\uff09": [[190, "c.c128_sparsetodense_s", false]], "c128_splice_p\uff08c function\uff09": [[191, "c.c128_splice_p", false]], "c128_splice_s\uff08c function\uff09": [[191, "c.c128_splice_s", false]], "c128_split_p\uff08c function\uff09": [[192, "c.c128_split_p", false]], "c128_split_s\uff08c function\uff09": [[192, "c.c128_split_s", false]], "c128_split_with_overlap_p\uff08c function\uff09": [[193, "c.c128_split_with_overlap_p", false]], "c128_split_with_overlap_s\uff08c function\uff09": [[193, "c.c128_split_with_overlap_s", false]], "c128_sqrt_p\uff08c function\uff09": [[194, "c.c128_sqrt_p", false]], "c128_sqrt_s\uff08c function\uff09": [[194, "c.c128_sqrt_s", false]], "c128_sqrtgrad_p\uff08c function\uff09": [[195, "c.c128_sqrtgrad_p", false]], "c128_sqrtgrad_s\uff08c function\uff09": [[195, "c.c128_sqrtgrad_s", false]], "c128_square_p\uff08c function\uff09": [[196, "c.c128_square_p", false]], "c128_square_s\uff08c function\uff09": [[196, "c.c128_square_s", false]], "c128_squaredifference_p\uff08c function\uff09": [[197, "c.c128_squaredifference_p", false]], "c128_squaredifference_s\uff08c function\uff09": [[197, "c.c128_squaredifference_s", false]], "c128_stack_p\uff08c function\uff09": [[199, "c.c128_stack_p", false]], "c128_stack_s\uff08c function\uff09": [[199, "c.c128_stack_s", false]], "c128_subext_p\uff08c function\uff09": [[202, "c.c128_subext_p", false]], "c128_subext_s\uff08c function\uff09": [[202, "c.c128_subext_s", false]], "c128_subrelu6_p\uff08c function\uff09": [[202, "c.c128_subrelu6_p", false]], "c128_subrelu6_s\uff08c function\uff09": [[202, "c.c128_subrelu6_s", false]], "c128_subrelu_p\uff08c function\uff09": [[202, "c.c128_subrelu_p", false]], "c128_subrelu_s\uff08c function\uff09": [[202, "c.c128_subrelu_s", false]], "c128_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.c128_tensor_scatter_add_p", false]], "c128_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.c128_tensor_scatter_add_s", false]], "c128_tensorarrayread_p\uff08c function\uff09": [[208, "c.c128_tensorarrayread_p", false]], "c128_tensorarrayread_s\uff08c function\uff09": [[208, "c.c128_tensorarrayread_s", false]], "c128_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.c128_tensorlistfromtensor_p", false]], "c128_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.c128_tensorlistfromtensor_s", false]], "c128_tile_p\uff08c function\uff09": [[215, "c.c128_tile_p", false]], "c128_tile_s\uff08c function\uff09": [[215, "c.c128_tile_s", false]], "c128_transpose_p\uff08c function\uff09": [[217, "c.c128_transpose_p", false]], "c128_transpose_s\uff08c function\uff09": [[217, "c.c128_transpose_s", false]], "c128_tril_p\uff08c function\uff09": [[218, "c.c128_tril_p", false]], "c128_tril_s\uff08c function\uff09": [[218, "c.c128_tril_s", false]], "c128_triu_p\uff08c function\uff09": [[219, "c.c128_triu_p", false]], "c128_triu_s\uff08c function\uff09": [[219, "c.c128_triu_s", false]], "c128_unique_p\uff08c function\uff09": [[221, "c.c128_Unique_p", false]], "c128_unique_s\uff08c function\uff09": [[221, "c.c128_Unique_s", false]], "c128_unsorted_segment_sum_p\uff08c function\uff09": [[222, "c.c128_unsorted_segment_sum_p", false]], "c128_unsorted_segment_sum_s\uff08c function\uff09": [[222, "c.c128_unsorted_segment_sum_s", false]], "c128_where_p\uff08c function\uff09": [[225, "c.c128_where_p", false]], "c128_where_s\uff08c function\uff09": [[225, "c.c128_where_s", false]], "c128_zerolike_p\uff08c function\uff09": [[226, "c.c128_zerolike_p", false]], "c128_zerolike_s\uff08c function\uff09": [[226, "c.c128_zerolike_s", false]], "c64_abs_p\uff08c function\uff09": [[10, "c.c64_abs_p", false]], "c64_abs_s\uff08c function\uff09": [[10, "c.c64_abs_s", false]], "c64_addext_p\uff08c function\uff09": [[17, "c.c64_addext_p", false]], "c64_addext_s\uff08c function\uff09": [[17, "c.c64_addext_s", false]], "c64_addn_p\uff08c function\uff09": [[19, "c.c64_addn_p", false]], "c64_addn_s\uff08c function\uff09": [[19, "c.c64_addn_s", false]], "c64_addrelu6_p\uff08c function\uff09": [[17, "c.c64_addrelu6_p", false]], "c64_addrelu6_s\uff08c function\uff09": [[17, "c.c64_addrelu6_s", false]], "c64_addrelu_p\uff08c function\uff09": [[17, "c.c64_addrelu_p", false]], "c64_addrelu_s\uff08c function\uff09": [[17, "c.c64_addrelu_s", false]], "c64_allgather_p\uff08c function\uff09": [[22, "c.c64_allgather_p", false]], "c64_allgather_s\uff08c function\uff09": [[22, "c.c64_allgather_s", false]], "c64_assign_p\uff08c function\uff09": [[27, "c.c64_assign_p", false]], "c64_assign_s\uff08c function\uff09": [[27, "c.c64_assign_s", false]], "c64_assignadd_p\uff08c function\uff09": [[28, "c.c64_assignadd_p", false]], "c64_assignadd_s\uff08c function\uff09": [[28, "c.c64_assignadd_s", false]], "c64_batchtospace_p\uff08c function\uff09": [[35, "c.c64_batchtospace_p", false]], "c64_batchtospace_s\uff08c function\uff09": [[35, "c.c64_batchtospace_s", false]], "c64_batchtospacend_p\uff08c function\uff09": [[36, "c.c64_batchtospacend_p", false]], "c64_batchtospacend_s\uff08c function\uff09": [[36, "c.c64_batchtospacend_s", false]], "c64_biasadd_p\uff08c function\uff09": [[37, "c.c64_biasadd_p", false]], "c64_biasadd_s\uff08c function\uff09": [[37, "c.c64_biasadd_s", false]], "c64_broadcastto_p\uff08c function\uff09": [[41, "c.c64_broadcastto_p", false]], "c64_broadcastto_s\uff08c function\uff09": [[41, "c.c64_broadcastto_s", false]], "c64_concat_p\uff08c function\uff09": [[45, "c.c64_concat_p", false]], "c64_concat_s\uff08c function\uff09": [[45, "c.c64_concat_s", false]], "c64_constant_of_shape_p\uff08c function\uff09": [[46, "c.c64_constant_of_shape_p", false]], "c64_constant_of_shape_s\uff08c function\uff09": [[46, "c.c64_constant_of_shape_s", false]], "c64_cumsum_p\uff08c function\uff09": [[54, "c.c64_cumsum_p", false]], "c64_cumsum_s\uff08c function\uff09": [[54, "c.c64_cumsum_s", false]], "c64_depthtospace_p\uff08c function\uff09": [[59, "c.c64_depthtospace_p", false]], "c64_depthtospace_s\uff08c function\uff09": [[59, "c.c64_depthtospace_s", false]], "c64_div_fusion_p\uff08c function\uff09": [[61, "c.c64_div_fusion_p", false]], "c64_div_fusion_s\uff08c function\uff09": [[61, "c.c64_div_fusion_s", false]], "c64_eltwise_p\uff08c function\uff09": [[67, "c.c64_eltwise_p", false]], "c64_eltwise_s\uff08c function\uff09": [[67, "c.c64_eltwise_s", false]], "c64_equal_p\uff08c function\uff09": [[70, "c.c64_equal_p", false]], "c64_equal_s\uff08c function\uff09": [[70, "c.c64_equal_s", false]], "c64_expfusion_p\uff08c function\uff09": [[73, "c.c64_expfusion_p", false]], "c64_expfusion_s\uff08c function\uff09": [[73, "c.c64_expfusion_s", false]], "c64_extract_features_p\uff08c function\uff09": [[55, "c.c64_extract_features_p", false]], "c64_extract_features_s\uff08c function\uff09": [[55, "c.c64_extract_features_s", false]], "c64_fill_p\uff08c function\uff09": [[78, "c.c64_fill_p", false]], "c64_fill_s\uff08c function\uff09": [[78, "c.c64_fill_s", false]], "c64_formattranspose_p\uff08c function\uff09": [[85, "c.c64_formattranspose_p", false]], "c64_formattranspose_s\uff08c function\uff09": [[85, "c.c64_formattranspose_s", false]], "c64_gather_nd_p\uff08c function\uff09": [[89, "c.c64_gather_nd_p", false]], "c64_gather_nd_s\uff08c function\uff09": [[89, "c.c64_gather_nd_s", false]], "c64_gather_p\uff08c function\uff09": [[88, "c.c64_gather_p", false]], "c64_gather_s\uff08c function\uff09": [[88, "c.c64_gather_s", false]], "c64_gatherd_p\uff08c function\uff09": [[90, "c.c64_gatherd_p", false]], "c64_gatherd_s\uff08c function\uff09": [[90, "c.c64_gatherd_s", false]], "c64_isfinite_p\uff08c function\uff09": [[99, "c.c64_isfinite_p", false]], "c64_isfinite_s\uff08c function\uff09": [[99, "c.c64_isfinite_s", false]], "c64_matmulfusion_p\uff08c function\uff09": [[121, "c.c64_matmulfusion_p", false]], "c64_matmulfusion_s\uff08c function\uff09": [[121, "c.c64_matmulfusion_s", false]], "c64_mul_p\uff08c function\uff09": [[130, "c.c64_mul_p", false]], "c64_mul_s\uff08c function\uff09": [[130, "c.c64_mul_s", false]], "c64_neg_grad_p\uff08c function\uff09": [[133, "c.c64_neg_grad_p", false]], "c64_neg_grad_s\uff08c function\uff09": [[133, "c.c64_neg_grad_s", false]], "c64_neg_p\uff08c function\uff09": [[132, "c.c64_neg_p", false]], "c64_neg_s\uff08c function\uff09": [[132, "c.c64_neg_s", false]], "c64_nonzero_p\uff08c function\uff09": [[137, "c.c64_nonzero_p", false]], "c64_nonzero_s\uff08c function\uff09": [[137, "c.c64_nonzero_s", false]], "c64_not_equal_p\uff08c function\uff09": [[138, "c.c64_not_equal_p", false]], "c64_not_equal_s\uff08c function\uff09": [[138, "c.c64_not_equal_s", false]], "c64_onehot_p\uff08c function\uff09": [[139, "c.c64_onehot_p", false]], "c64_onehot_s\uff08c function\uff09": [[139, "c.c64_onehot_s", false]], "c64_ones_like_p\uff08c function\uff09": [[140, "c.c64_ones_like_p", false]], "c64_ones_like_s\uff08c function\uff09": [[140, "c.c64_ones_like_s", false]], "c64_padfusion_p\uff08c function\uff09": [[141, "c.c64_padfusion_p", false]], "c64_padfusion_s\uff08c function\uff09": [[141, "c.c64_padfusion_s", false]], "c64_real_div_p\uff08c function\uff09": [[152, "c.c64_real_div_p", false]], "c64_real_div_s\uff08c function\uff09": [[152, "c.c64_real_div_s", false]], "c64_reciprocal_p\uff08c function\uff09": [[153, "c.c64_reciprocal_p", false]], "c64_reciprocal_s\uff08c function\uff09": [[153, "c.c64_reciprocal_s", false]], "c64_reduceall_p\uff08c function\uff09": [[21, "c.c64_reduceall_p", false]], "c64_reduceall_s\uff08c function\uff09": [[21, "c.c64_reduceall_s", false]], "c64_reshape_p\uff08c function\uff09": [[156, "c.c64_reshape_p", false]], "c64_reshape_s\uff08c function\uff09": [[156, "c.c64_reshape_s", false]], "c64_rfft_p\uff08c function\uff09": [[161, "c.c64_rfft_p", false]], "c64_rfft_s\uff08c function\uff09": [[161, "c.c64_rfft_s", false]], "c64_rsqrt_p\uff08c function\uff09": [[164, "c.c64_rsqrt_p", false]], "c64_rsqrt_s\uff08c function\uff09": [[164, "c.c64_rsqrt_s", false]], "c64_scatter_elements_p\uff08c function\uff09": [[167, "c.c64_scatter_elements_p", false]], "c64_scatter_elements_s\uff08c function\uff09": [[167, "c.c64_scatter_elements_s", false]], "c64_scatter_nd_p\uff08c function\uff09": [[168, "c.c64_scatter_nd_p", false]], "c64_scatter_nd_s\uff08c function\uff09": [[168, "c.c64_scatter_nd_s", false]], "c64_scatter_nd_update_p\uff08c function\uff09": [[169, "c.c64_scatter_nd_update_p", false]], "c64_scatter_nd_update_s\uff08c function\uff09": [[169, "c.c64_scatter_nd_update_s", false]], "c64_select_p\uff08c function\uff09": [[170, "c.c64_select_p", false]], "c64_select_s\uff08c function\uff09": [[170, "c.c64_select_s", false]], "c64_slice_p\uff08c function\uff09": [[178, "c.c64_slice_p", false]], "c64_slice_s\uff08c function\uff09": [[178, "c.c64_slice_s", false]], "c64_spacetobatch_p\uff08c function\uff09": [[183, "c.c64_spacetobatch_p", false]], "c64_spacetobatch_s\uff08c function\uff09": [[183, "c.c64_spacetobatch_s", false]], "c64_spacetobatchnd_p\uff08c function\uff09": [[184, "c.c64_spacetobatchnd_p", false]], "c64_spacetobatchnd_s\uff08c function\uff09": [[184, "c.c64_spacetobatchnd_s", false]], "c64_spacetodepth_p\uff08c function\uff09": [[185, "c.c64_spacetodepth_p", false]], "c64_spacetodepth_s\uff08c function\uff09": [[185, "c.c64_spacetodepth_s", false]], "c64_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.c64_sparsefillemptyrows_p", false]], "c64_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.c64_sparsefillemptyrows_s", false]], "c64_sparsesegmentsum_p\uff08c function\uff09": [[189, "c.c64_sparsesegmentsum_p", false]], "c64_sparsesegmentsum_s\uff08c function\uff09": [[189, "c.c64_sparsesegmentsum_s", false]], "c64_sparsetodense_p\uff08c function\uff09": [[190, "c.c64_sparsetodense_p", false]], "c64_sparsetodense_s\uff08c function\uff09": [[190, "c.c64_sparsetodense_s", false]], "c64_splice_p\uff08c function\uff09": [[191, "c.c64_splice_p", false]], "c64_splice_s\uff08c function\uff09": [[191, "c.c64_splice_s", false]], "c64_split_p\uff08c function\uff09": [[192, "c.c64_split_p", false]], "c64_split_s\uff08c function\uff09": [[192, "c.c64_split_s", false]], "c64_split_with_overlap_p\uff08c function\uff09": [[193, "c.c64_split_with_overlap_p", false]], "c64_split_with_overlap_s\uff08c function\uff09": [[193, "c.c64_split_with_overlap_s", false]], "c64_sqrt_p\uff08c function\uff09": [[194, "c.c64_sqrt_p", false]], "c64_sqrt_s\uff08c function\uff09": [[194, "c.c64_sqrt_s", false]], "c64_sqrtgrad_p\uff08c function\uff09": [[195, "c.c64_sqrtgrad_p", false]], "c64_sqrtgrad_s\uff08c function\uff09": [[195, "c.c64_sqrtgrad_s", false]], "c64_square_p\uff08c function\uff09": [[196, "c.c64_square_p", false]], "c64_square_s\uff08c function\uff09": [[196, "c.c64_square_s", false]], "c64_squaredifference_p\uff08c function\uff09": [[197, "c.c64_squaredifference_p", false]], "c64_squaredifference_s\uff08c function\uff09": [[197, "c.c64_squaredifference_s", false]], "c64_stack_p\uff08c function\uff09": [[199, "c.c64_stack_p", false]], "c64_stack_s\uff08c function\uff09": [[199, "c.c64_stack_s", false]], "c64_subext_p\uff08c function\uff09": [[202, "c.c64_subext_p", false]], "c64_subext_s\uff08c function\uff09": [[202, "c.c64_subext_s", false]], "c64_subrelu6_p\uff08c function\uff09": [[202, "c.c64_subrelu6_p", false]], "c64_subrelu6_s\uff08c function\uff09": [[202, "c.c64_subrelu6_s", false]], "c64_subrelu_p\uff08c function\uff09": [[202, "c.c64_subrelu_p", false]], "c64_subrelu_s\uff08c function\uff09": [[202, "c.c64_subrelu_s", false]], "c64_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.c64_tensor_scatter_add_p", false]], "c64_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.c64_tensor_scatter_add_s", false]], "c64_tensorarrayread_p\uff08c function\uff09": [[208, "c.c64_tensorarrayread_p", false]], "c64_tensorarrayread_s\uff08c function\uff09": [[208, "c.c64_tensorarrayread_s", false]], "c64_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.c64_tensorlistfromtensor_p", false]], "c64_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.c64_tensorlistfromtensor_s", false]], "c64_tile_p\uff08c function\uff09": [[215, "c.c64_tile_p", false]], "c64_tile_s\uff08c function\uff09": [[215, "c.c64_tile_s", false]], "c64_transpose_p\uff08c function\uff09": [[217, "c.c64_transpose_p", false]], "c64_transpose_s\uff08c function\uff09": [[217, "c.c64_transpose_s", false]], "c64_tril_p\uff08c function\uff09": [[218, "c.c64_tril_p", false]], "c64_tril_s\uff08c function\uff09": [[218, "c.c64_tril_s", false]], "c64_triu_p\uff08c function\uff09": [[219, "c.c64_triu_p", false]], "c64_triu_s\uff08c function\uff09": [[219, "c.c64_triu_s", false]], "c64_unique_p\uff08c function\uff09": [[221, "c.c64_Unique_p", false]], "c64_unique_s\uff08c function\uff09": [[221, "c.c64_Unique_s", false]], "c64_unsorted_segment_sum_p\uff08c function\uff09": [[222, "c.c64_unsorted_segment_sum_p", false]], "c64_unsorted_segment_sum_s\uff08c function\uff09": [[222, "c.c64_unsorted_segment_sum_s", false]], "c64_where_p\uff08c function\uff09": [[225, "c.c64_where_p", false]], "c64_where_s\uff08c function\uff09": [[225, "c.c64_where_s", false]], "c64_zerolike_p\uff08c function\uff09": [[226, "c.c64_zerolike_p", false]], "c64_zerolike_s\uff08c function\uff09": [[226, "c.c64_zerolike_s", false]], "castc128toc128_p\uff08c function\uff09": [[42, "c.castc128Toc128_p", false]], "castc128toc128_s\uff08c function\uff09": [[42, "c.castc128Toc128_s", false]], "castc64toc64_p\uff08c function\uff09": [[42, "c.castc64Toc64_p", false]], "castc64toc64_s\uff08c function\uff09": [[42, "c.castc64Toc64_s", false]], "castdptodp_p\uff08c function\uff09": [[42, "c.castdpTodp_p", false]], "castdptodp_s\uff08c function\uff09": [[42, "c.castdpTodp_s", false]], "castdptofp_p\uff08c function\uff09": [[42, "c.castdpTofp_p", false]], "castdptofp_s\uff08c function\uff09": [[42, "c.castdpTofp_s", false]], "castfptofp_p\uff08c function\uff09": [[42, "c.castfpTofp_p", false]], "castfptofp_s\uff08c function\uff09": [[42, "c.castfpTofp_s", false]], "castfptoint16_p\uff08c function\uff09": [[42, "c.castfpToint16_p", false]], "castfptoint16_s\uff08c function\uff09": [[42, "c.castfpToint16_s", false]], "castfptoint32_p\uff08c function\uff09": [[42, "c.castfpToint32_p", false]], "castfptoint32_s\uff08c function\uff09": [[42, "c.castfpToint32_s", false]], "castfptoint8_p\uff08c function\uff09": [[42, "c.castfpToint8_p", false]], "castfptoint8_s\uff08c function\uff09": [[42, "c.castfpToint8_s", false]], "casti16tofp_p\uff08c function\uff09": [[42, "c.casti16Tofp_p", false]], "casti16tofp_s\uff08c function\uff09": [[42, "c.casti16Tofp_s", false]], "casti16tointi16_p\uff08c function\uff09": [[42, "c.casti16Tointi16_p", false]], "casti16tointi16_s\uff08c function\uff09": [[42, "c.casti16Tointi16_s", false]], "casti16tointi32_p\uff08c function\uff09": [[42, "c.casti16Tointi32_p", false]], "casti16tointi32_s\uff08c function\uff09": [[42, "c.casti16Tointi32_s", false]], "casti16tointi8_p\uff08c function\uff09": [[42, "c.casti16Tointi8_p", false]], "casti16tointi8_s\uff08c function\uff09": [[42, "c.casti16Tointi8_s", false]], "casti32tofp_p\uff08c function\uff09": [[42, "c.casti32Tofp_p", false]], "casti32tofp_s\uff08c function\uff09": [[42, "c.casti32Tofp_s", false]], "casti32tointi16_p\uff08c function\uff09": [[42, "c.casti32Tointi16_p", false]], "casti32tointi16_s\uff08c function\uff09": [[42, "c.casti32Tointi16_s", false]], "casti32tointi32_p\uff08c function\uff09": [[42, "c.casti32Tointi32_p", false]], "casti32tointi32_s\uff08c function\uff09": [[42, "c.casti32Tointi32_s", false]], "casti32tointi8_p\uff08c function\uff09": [[42, "c.casti32Tointi8_p", false]], "casti32tointi8_s\uff08c function\uff09": [[42, "c.casti32Tointi8_s", false]], "casti8tofp_p\uff08c function\uff09": [[42, "c.casti8Tofp_p", false]], "casti8tofp_s\uff08c function\uff09": [[42, "c.casti8Tofp_s", false]], "casti8tointi16_p\uff08c function\uff09": [[42, "c.casti8Tointi16_p", false]], "casti8tointi16_s\uff08c function\uff09": [[42, "c.casti8Tointi16_s", false]], "casti8tointi32_p\uff08c function\uff09": [[42, "c.casti8Tointi32_p", false]], "casti8tointi32_s\uff08c function\uff09": [[42, "c.casti8Tointi32_s", false]], "casti8tointi8_p\uff08c function\uff09": [[42, "c.casti8Tointi8_p", false]], "casti8tointi8_s\uff08c function\uff09": [[42, "c.casti8Tointi8_s", false]], "customnormalize_p\uff08c function\uff09": [[56, "c.customnormalize_p", false]], "customnormalize_s\uff08c function\uff09": [[56, "c.customnormalize_s", false]], "custompredict_p\uff08c function\uff09": [[57, "c.custompredict_p", false]], "custompredict_s\uff08c function\uff09": [[57, "c.custompredict_s", false]], "dp_abs_p\uff08c function\uff09": [[10, "c.dp_abs_p", false]], "dp_abs_s\uff08c function\uff09": [[10, "c.dp_abs_s", false]], "dp_addext_p\uff08c function\uff09": [[17, "c.dp_addext_p", false]], "dp_addext_s\uff08c function\uff09": [[17, "c.dp_addext_s", false]], "dp_addn_p\uff08c function\uff09": [[19, "c.dp_addn_p", false]], "dp_addn_s\uff08c function\uff09": [[19, "c.dp_addn_s", false]], "dp_addrelu6_p\uff08c function\uff09": [[17, "c.dp_addrelu6_p", false]], "dp_addrelu6_s\uff08c function\uff09": [[17, "c.dp_addrelu6_s", false]], "dp_addrelu_p\uff08c function\uff09": [[17, "c.dp_addrelu_p", false]], "dp_addrelu_s\uff08c function\uff09": [[17, "c.dp_addrelu_s", false]], "dp_allgather_p\uff08c function\uff09": [[22, "c.dp_allgather_p", false]], "dp_allgather_s\uff08c function\uff09": [[22, "c.dp_allgather_s", false]], "dp_and_p\uff08c function\uff09": [[112, "c.dp_and_p", false]], "dp_and_s\uff08c function\uff09": [[112, "c.dp_and_s", false]], "dp_assign_p\uff08c function\uff09": [[27, "c.dp_assign_p", false]], "dp_assign_s\uff08c function\uff09": [[27, "c.dp_assign_s", false]], "dp_assignadd_p\uff08c function\uff09": [[28, "c.dp_assignadd_p", false]], "dp_assignadd_s\uff08c function\uff09": [[28, "c.dp_assignadd_s", false]], "dp_batchtospace_p\uff08c function\uff09": [[35, "c.dp_batchtospace_p", false]], "dp_batchtospace_s\uff08c function\uff09": [[35, "c.dp_batchtospace_s", false]], "dp_batchtospacend_p\uff08c function\uff09": [[36, "c.dp_batchtospacend_p", false]], "dp_batchtospacend_s\uff08c function\uff09": [[36, "c.dp_batchtospacend_s", false]], "dp_biasadd_p\uff08c function\uff09": [[37, "c.dp_biasadd_p", false]], "dp_biasadd_s\uff08c function\uff09": [[37, "c.dp_biasadd_s", false]], "dp_broadcastto_p\uff08c function\uff09": [[41, "c.dp_broadcastto_p", false]], "dp_broadcastto_s\uff08c function\uff09": [[41, "c.dp_broadcastto_s", false]], "dp_ceil_p\uff08c function\uff09": [[43, "c.dp_ceil_p", false]], "dp_ceil_s\uff08c function\uff09": [[43, "c.dp_ceil_s", false]], "dp_clip_p\uff08c function\uff09": [[44, "c.dp_clip_p", false]], "dp_clip_s\uff08c function\uff09": [[44, "c.dp_clip_s", false]], "dp_concat_p\uff08c function\uff09": [[45, "c.dp_concat_p", false]], "dp_concat_s\uff08c function\uff09": [[45, "c.dp_concat_s", false]], "dp_constant_of_shape_p\uff08c function\uff09": [[46, "c.dp_constant_of_shape_p", false]], "dp_constant_of_shape_s\uff08c function\uff09": [[46, "c.dp_constant_of_shape_s", false]], "dp_cos_p\uff08c function\uff09": [[51, "c.dp_cos_p", false]], "dp_cos_s\uff08c function\uff09": [[51, "c.dp_cos_s", false]], "dp_cumsum_p\uff08c function\uff09": [[54, "c.dp_cumsum_p", false]], "dp_cumsum_s\uff08c function\uff09": [[54, "c.dp_cumsum_s", false]], "dp_depthtospace_p\uff08c function\uff09": [[59, "c.dp_depthtospace_p", false]], "dp_depthtospace_s\uff08c function\uff09": [[59, "c.dp_depthtospace_s", false]], "dp_div_fusion_p\uff08c function\uff09": [[61, "c.dp_div_fusion_p", false]], "dp_div_fusion_s\uff08c function\uff09": [[61, "c.dp_div_fusion_s", false]], "dp_eltwise_p\uff08c function\uff09": [[67, "c.dp_eltwise_p", false]], "dp_eltwise_s\uff08c function\uff09": [[67, "c.dp_eltwise_s", false]], "dp_equal_p\uff08c function\uff09": [[70, "c.dp_equal_p", false]], "dp_equal_s\uff08c function\uff09": [[70, "c.dp_equal_s", false]], "dp_erf_p\uff08c function\uff09": [[71, "c.dp_erf_p", false]], "dp_erf_s\uff08c function\uff09": [[71, "c.dp_erf_s", false]], "dp_expfusion_p\uff08c function\uff09": [[73, "c.dp_expfusion_p", false]], "dp_expfusion_s\uff08c function\uff09": [[73, "c.dp_expfusion_s", false]], "dp_extract_features_p\uff08c function\uff09": [[55, "c.dp_extract_features_p", false]], "dp_extract_features_s\uff08c function\uff09": [[55, "c.dp_extract_features_s", false]], "dp_fill_p\uff08c function\uff09": [[78, "c.dp_fill_p", false]], "dp_fill_s\uff08c function\uff09": [[78, "c.dp_fill_s", false]], "dp_floor_p\uff08c function\uff09": [[82, "c.dp_floor_p", false]], "dp_floor_s\uff08c function\uff09": [[82, "c.dp_floor_s", false]], "dp_floordiv_p\uff08c function\uff09": [[83, "c.dp_floordiv_p", false]], "dp_floordiv_s\uff08c function\uff09": [[83, "c.dp_floordiv_s", false]], "dp_floormod_p\uff08c function\uff09": [[84, "c.dp_floormod_p", false]], "dp_floormod_s\uff08c function\uff09": [[84, "c.dp_floormod_s", false]], "dp_formattranspose_p\uff08c function\uff09": [[85, "c.dp_formattranspose_p", false]], "dp_formattranspose_s\uff08c function\uff09": [[85, "c.dp_formattranspose_s", false]], "dp_gather_nd_p\uff08c function\uff09": [[89, "c.dp_gather_nd_p", false]], "dp_gather_nd_s\uff08c function\uff09": [[89, "c.dp_gather_nd_s", false]], "dp_gather_p\uff08c function\uff09": [[88, "c.dp_gather_p", false]], "dp_gather_s\uff08c function\uff09": [[88, "c.dp_gather_s", false]], "dp_gatherd_p\uff08c function\uff09": [[90, "c.dp_gatherd_p", false]], "dp_gatherd_s\uff08c function\uff09": [[90, "c.dp_gatherd_s", false]], "dp_greater_s\uff08c function\uff09": [[92, "c.dp_greater_s", false]], "dp_greaterequal_s\uff08c function\uff09": [[93, "c.dp_greaterequal_s", false]], "dp_isfinite_p\uff08c function\uff09": [[99, "c.dp_isfinite_p", false]], "dp_isfinite_s\uff08c function\uff09": [[99, "c.dp_isfinite_s", false]], "dp_less_p\uff08c function\uff09": [[104, "c.dp_less_p", false]], "dp_less_s\uff08c function\uff09": [[104, "c.dp_less_s", false]], "dp_lessequal_p\uff08c function\uff09": [[105, "c.dp_lessequal_p", false]], "dp_lessequal_s\uff08c function\uff09": [[105, "c.dp_lessequal_s", false]], "dp_log1p_p\uff08c function\uff09": [[108, "c.dp_log1p_p", false]], "dp_log1p_s\uff08c function\uff09": [[108, "c.dp_log1p_s", false]], "dp_logical_not_p\uff08c function\uff09": [[110, "c.dp_logical_not_p", false]], "dp_logical_not_s\uff08c function\uff09": [[110, "c.dp_logical_not_s", false]], "dp_logical_or_p\uff08c function\uff09": [[111, "c.dp_logical_or_p", false]], "dp_logical_or_s\uff08c function\uff09": [[111, "c.dp_logical_or_s", false]], "dp_lsh_projection_p\uff08c function\uff09": [[116, "c.dp_lsh_projection_p", false]], "dp_lsh_projection_s\uff08c function\uff09": [[116, "c.dp_lsh_projection_s", false]], "dp_matmulfusion_p\uff08c function\uff09": [[121, "c.dp_matmulfusion_p", false]], "dp_matmulfusion_s\uff08c function\uff09": [[121, "c.dp_matmulfusion_s", false]], "dp_maximum_p\uff08c function\uff09": [[122, "c.dp_maximum_p", false]], "dp_maximum_s\uff08c function\uff09": [[122, "c.dp_maximum_s", false]], "dp_maxpool_fusion_p\uff08c function\uff09": [[124, "c.dp_maxpool_fusion_p", false]], "dp_maxpool_fusion_s\uff08c function\uff09": [[124, "c.dp_maxpool_fusion_s", false]], "dp_maxpool_grad_p\uff08c function\uff09": [[125, "c.dp_maxpool_grad_p", false]], "dp_maxpool_grad_s\uff08c function\uff09": [[125, "c.dp_maxpool_grad_s", false]], "dp_minimum_p\uff08c function\uff09": [[127, "c.dp_minimum_p", false]], "dp_minimum_s\uff08c function\uff09": [[127, "c.dp_minimum_s", false]], "dp_mod_p\uff08c function\uff09": [[129, "c.dp_mod_p", false]], "dp_mod_s\uff08c function\uff09": [[129, "c.dp_mod_s", false]], "dp_mul_p\uff08c function\uff09": [[130, "c.dp_mul_p", false]], "dp_mul_s\uff08c function\uff09": [[130, "c.dp_mul_s", false]], "dp_neg_grad_p\uff08c function\uff09": [[133, "c.dp_neg_grad_p", false]], "dp_neg_grad_s\uff08c function\uff09": [[133, "c.dp_neg_grad_s", false]], "dp_neg_p\uff08c function\uff09": [[132, "c.dp_neg_p", false]], "dp_neg_s\uff08c function\uff09": [[132, "c.dp_neg_s", false]], "dp_nonzero_p\uff08c function\uff09": [[137, "c.dp_nonzero_p", false]], "dp_nonzero_s\uff08c function\uff09": [[137, "c.dp_nonzero_s", false]], "dp_not_equal_p\uff08c function\uff09": [[138, "c.dp_not_equal_p", false]], "dp_not_equal_s\uff08c function\uff09": [[138, "c.dp_not_equal_s", false]], "dp_onehot_p\uff08c function\uff09": [[139, "c.dp_onehot_p", false]], "dp_onehot_s\uff08c function\uff09": [[139, "c.dp_onehot_s", false]], "dp_ones_like_p\uff08c function\uff09": [[140, "c.dp_ones_like_p", false]], "dp_ones_like_s\uff08c function\uff09": [[140, "c.dp_ones_like_s", false]], "dp_padfusion_p\uff08c function\uff09": [[141, "c.dp_padfusion_p", false]], "dp_padfusion_s\uff08c function\uff09": [[141, "c.dp_padfusion_s", false]], "dp_pow_fusion_p\uff08c function\uff09": [[142, "c.dp_pow_fusion_p", false]], "dp_pow_fusion_s\uff08c function\uff09": [[142, "c.dp_pow_fusion_s", false]], "dp_raggedrange_p\uff08c function\uff09": [[147, "c.dp_raggedrange_p", false]], "dp_raggedrange_s\uff08c function\uff09": [[147, "c.dp_raggedrange_s", false]], "dp_range_p\uff08c function\uff09": [[150, "c.dp_range_p", false]], "dp_range_s\uff08c function\uff09": [[150, "c.dp_range_s", false]], "dp_real_div_p\uff08c function\uff09": [[152, "c.dp_real_div_p", false]], "dp_real_div_s\uff08c function\uff09": [[152, "c.dp_real_div_s", false]], "dp_reciprocal_p\uff08c function\uff09": [[153, "c.dp_reciprocal_p", false]], "dp_reciprocal_s\uff08c function\uff09": [[153, "c.dp_reciprocal_s", false]], "dp_reduce_p\uff08c function\uff09": [[154, "c.dp_reduce_p", false]], "dp_reduce_s\uff08c function\uff09": [[154, "c.dp_reduce_s", false]], "dp_reduceall_p\uff08c function\uff09": [[21, "c.dp_reduceall_p", false]], "dp_reduceall_s\uff08c function\uff09": [[21, "c.dp_reduceall_s", false]], "dp_reducescatter_p\uff08c function\uff09": [[155, "c.dp_reducescatter_p", false]], "dp_reducescatter_s\uff08c function\uff09": [[155, "c.dp_reducescatter_s", false]], "dp_reshape_p\uff08c function\uff09": [[156, "c.dp_reshape_p", false]], "dp_reshape_s\uff08c function\uff09": [[156, "c.dp_reshape_s", false]], "dp_round_p\uff08c function\uff09": [[163, "c.dp_round_p", false]], "dp_round_s\uff08c function\uff09": [[163, "c.dp_round_s", false]], "dp_rsqrt_p\uff08c function\uff09": [[164, "c.dp_rsqrt_p", false]], "dp_rsqrt_s\uff08c function\uff09": [[164, "c.dp_rsqrt_s", false]], "dp_scalefusion_p\uff08c function\uff09": [[166, "c.dp_scalefusion_p", false]], "dp_scalefusion_s\uff08c function\uff09": [[166, "c.dp_scalefusion_s", false]], "dp_scatter_elements_p\uff08c function\uff09": [[167, "c.dp_scatter_elements_p", false]], "dp_scatter_elements_s\uff08c function\uff09": [[167, "c.dp_scatter_elements_s", false]], "dp_scatter_nd_p\uff08c function\uff09": [[168, "c.dp_scatter_nd_p", false]], "dp_scatter_nd_s\uff08c function\uff09": [[168, "c.dp_scatter_nd_s", false]], "dp_scatter_nd_update_p\uff08c function\uff09": [[169, "c.dp_scatter_nd_update_p", false]], "dp_scatter_nd_update_s\uff08c function\uff09": [[169, "c.dp_scatter_nd_update_s", false]], "dp_select_p\uff08c function\uff09": [[170, "c.dp_select_p", false]], "dp_select_s\uff08c function\uff09": [[170, "c.dp_select_s", false]], "dp_sin_p\uff08c function\uff09": [[175, "c.dp_sin_p", false]], "dp_sin_s\uff08c function\uff09": [[175, "c.dp_sin_s", false]], "dp_slice_p\uff08c function\uff09": [[178, "c.dp_slice_p", false]], "dp_slice_s\uff08c function\uff09": [[178, "c.dp_slice_s", false]], "dp_spacetobatch_p\uff08c function\uff09": [[183, "c.dp_spacetobatch_p", false]], "dp_spacetobatch_s\uff08c function\uff09": [[183, "c.dp_spacetobatch_s", false]], "dp_spacetobatchnd_p\uff08c function\uff09": [[184, "c.dp_spacetobatchnd_p", false]], "dp_spacetobatchnd_s\uff08c function\uff09": [[184, "c.dp_spacetobatchnd_s", false]], "dp_spacetodepth_p\uff08c function\uff09": [[185, "c.dp_spacetodepth_p", false]], "dp_spacetodepth_s\uff08c function\uff09": [[185, "c.dp_spacetodepth_s", false]], "dp_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.dp_sparsefillemptyrows_p", false]], "dp_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.dp_sparsefillemptyrows_s", false]], "dp_sparsesegmentsum_p\uff08c function\uff09": [[189, "c.dp_sparsesegmentsum_p", false]], "dp_sparsesegmentsum_s\uff08c function\uff09": [[189, "c.dp_sparsesegmentsum_s", false]], "dp_sparsetodense_p\uff08c function\uff09": [[190, "c.dp_sparsetodense_p", false]], "dp_sparsetodense_s\uff08c function\uff09": [[190, "c.dp_sparsetodense_s", false]], "dp_splice_p\uff08c function\uff09": [[191, "c.dp_splice_p", false]], "dp_splice_s\uff08c function\uff09": [[191, "c.dp_splice_s", false]], "dp_split_p\uff08c function\uff09": [[192, "c.dp_split_p", false]], "dp_split_s\uff08c function\uff09": [[192, "c.dp_split_s", false]], "dp_split_with_overlap_p\uff08c function\uff09": [[193, "c.dp_split_with_overlap_p", false]], "dp_split_with_overlap_s\uff08c function\uff09": [[193, "c.dp_split_with_overlap_s", false]], "dp_sqrt_p\uff08c function\uff09": [[194, "c.dp_sqrt_p", false]], "dp_sqrt_s\uff08c function\uff09": [[194, "c.dp_sqrt_s", false]], "dp_sqrtgrad_p\uff08c function\uff09": [[195, "c.dp_sqrtgrad_p", false]], "dp_sqrtgrad_s\uff08c function\uff09": [[195, "c.dp_sqrtgrad_s", false]], "dp_square_p\uff08c function\uff09": [[196, "c.dp_square_p", false]], "dp_square_s\uff08c function\uff09": [[196, "c.dp_square_s", false]], "dp_squaredifference_p\uff08c function\uff09": [[197, "c.dp_squaredifference_p", false]], "dp_squaredifference_s\uff08c function\uff09": [[197, "c.dp_squaredifference_s", false]], "dp_stack_p\uff08c function\uff09": [[199, "c.dp_stack_p", false]], "dp_stack_s\uff08c function\uff09": [[199, "c.dp_stack_s", false]], "dp_subext_p\uff08c function\uff09": [[202, "c.dp_subext_p", false]], "dp_subext_s\uff08c function\uff09": [[202, "c.dp_subext_s", false]], "dp_subrelu6_p\uff08c function\uff09": [[202, "c.dp_subrelu6_p", false]], "dp_subrelu6_s\uff08c function\uff09": [[202, "c.dp_subrelu6_s", false]], "dp_subrelu_p\uff08c function\uff09": [[202, "c.dp_subrelu_p", false]], "dp_subrelu_s\uff08c function\uff09": [[202, "c.dp_subrelu_s", false]], "dp_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.dp_tensor_scatter_add_p", false]], "dp_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.dp_tensor_scatter_add_s", false]], "dp_tensorarrayread_p\uff08c function\uff09": [[208, "c.dp_tensorarrayread_p", false]], "dp_tensorarrayread_s\uff08c function\uff09": [[208, "c.dp_tensorarrayread_s", false]], "dp_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.dp_tensorlistfromtensor_p", false]], "dp_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.dp_tensorlistfromtensor_s", false]], "dp_tile_p\uff08c function\uff09": [[215, "c.dp_tile_p", false]], "dp_tile_s\uff08c function\uff09": [[215, "c.dp_tile_s", false]], "dp_transpose_p\uff08c function\uff09": [[217, "c.dp_transpose_p", false]], "dp_transpose_s\uff08c function\uff09": [[217, "c.dp_transpose_s", false]], "dp_tril_p\uff08c function\uff09": [[218, "c.dp_tril_p", false]], "dp_tril_s\uff08c function\uff09": [[218, "c.dp_tril_s", false]], "dp_triu_p\uff08c function\uff09": [[219, "c.dp_triu_p", false]], "dp_triu_s\uff08c function\uff09": [[219, "c.dp_triu_s", false]], "dp_unique_p\uff08c function\uff09": [[221, "c.dp_Unique_p", false]], "dp_unique_s\uff08c function\uff09": [[221, "c.dp_Unique_s", false]], "dp_unsorted_segment_sum_p\uff08c function\uff09": [[222, "c.dp_unsorted_segment_sum_p", false]], "dp_unsorted_segment_sum_s\uff08c function\uff09": [[222, "c.dp_unsorted_segment_sum_s", false]], "dp_where_p\uff08c function\uff09": [[225, "c.dp_where_p", false]], "dp_where_s\uff08c function\uff09": [[225, "c.dp_where_s", false]], "dp_zerolike_p\uff08c function\uff09": [[226, "c.dp_zerolike_p", false]], "dp_zerolike_s\uff08c function\uff09": [[226, "c.dp_zerolike_s", false]], "flatten\uff08c function\uff09": [[80, "c.Flatten", false]], "fp_abs_p\uff08c function\uff09": [[10, "c.fp_abs_p", false]], "fp_abs_s\uff08c function\uff09": [[10, "c.fp_abs_s", false]], "fp_absgrad_p\uff08c function\uff09": [[11, "c.fp_absgrad_p", false]], "fp_absgrad_s\uff08c function\uff09": [[11, "c.fp_absgrad_s", false]], "fp_adam_p\uff08c function\uff09": [[14, "c.fp_adam_p", false]], "fp_adam_s\uff08c function\uff09": [[14, "c.fp_adam_s", false]], "fp_adamweightdecay_p\uff08c function\uff09": [[15, "c.fp_adamweightdecay_p", false]], "fp_adamweightdecay_s\uff08c function\uff09": [[15, "c.fp_adamweightdecay_s", false]], "fp_adder_p\uff08c function\uff09": [[16, "c.fp_adder_p", false]], "fp_adder_s\uff08c function\uff09": [[16, "c.fp_adder_s", false]], "fp_addext_p\uff08c function\uff09": [[17, "c.fp_addext_p", false]], "fp_addext_s\uff08c function\uff09": [[17, "c.fp_addext_s", false]], "fp_addgrad_p\uff08c function\uff09": [[18, "c.fp_addgrad_p", false]], "fp_addgrad_s\uff08c function\uff09": [[18, "c.fp_addgrad_s", false]], "fp_addn_p\uff08c function\uff09": [[19, "c.fp_addn_p", false]], "fp_addn_s\uff08c function\uff09": [[19, "c.fp_addn_s", false]], "fp_addrelu6_p\uff08c function\uff09": [[17, "c.fp_addrelu6_p", false]], "fp_addrelu6_s\uff08c function\uff09": [[17, "c.fp_addrelu6_s", false]], "fp_addrelu_p\uff08c function\uff09": [[17, "c.fp_addrelu_p", false]], "fp_addrelu_s\uff08c function\uff09": [[17, "c.fp_addrelu_s", false]], "fp_affine_p\uff08c function\uff09": [[20, "c.fp_affine_p", false]], "fp_affine_s\uff08c function\uff09": [[20, "c.fp_affine_s", false]], "fp_allgather_p\uff08c function\uff09": [[22, "c.fp_allgather_p", false]], "fp_allgather_s\uff08c function\uff09": [[22, "c.fp_allgather_s", false]], "fp_and_p\uff08c function\uff09": [[112, "c.fp_and_p", false]], "fp_and_s\uff08c function\uff09": [[112, "c.fp_and_s", false]], "fp_applymomentum_p\uff08c function\uff09": [[23, "c.fp_applymomentum_p", false]], "fp_applymomentum_s\uff08c function\uff09": [[23, "c.fp_applymomentum_s", false]], "fp_argmax_p\uff08c function\uff09": [[24, "c.fp_argmax_p", false]], "fp_argmax_s\uff08c function\uff09": [[24, "c.fp_argmax_s", false]], "fp_argmin_p\uff08c function\uff09": [[25, "c.fp_argmin_p", false]], "fp_argmin_s\uff08c function\uff09": [[25, "c.fp_argmin_s", false]], "fp_assign_p\uff08c function\uff09": [[27, "c.fp_assign_p", false]], "fp_assign_s\uff08c function\uff09": [[27, "c.fp_assign_s", false]], "fp_assignadd_p\uff08c function\uff09": [[28, "c.fp_assignadd_p", false]], "fp_assignadd_s\uff08c function\uff09": [[28, "c.fp_assignadd_s", false]], "fp_attention_p\uff08c function\uff09": [[29, "c.fp_attention_p", false]], "fp_attention_s\uff08c function\uff09": [[29, "c.fp_attention_s", false]], "fp_audio_spectrogram_p\uff08c function\uff09": [[30, "c.fp_audio_spectrogram_p", false]], "fp_audio_spectrogram_s\uff08c function\uff09": [[30, "c.fp_audio_spectrogram_s", false]], "fp_avgpool_fusion_p\uff08c function\uff09": [[31, "c.fp_avgpool_fusion_p", false]], "fp_avgpool_fusion_s\uff08c function\uff09": [[31, "c.fp_avgpool_fusion_s", false]], "fp_avgpoolinggrad_p\uff08c function\uff09": [[32, "c.fp_avgpoolinggrad_p", false]], "fp_avgpoolinggrad_s\uff08c function\uff09": [[32, "c.fp_avgpoolinggrad_s", false]], "fp_batchnorm_p\uff08c function\uff09": [[33, "c.fp_batchnorm_p", false]], "fp_batchnorm_s\uff08c function\uff09": [[33, "c.fp_batchnorm_s", false]], "fp_batchnormgrad_p\uff08c function\uff09": [[34, "c.fp_batchnormgrad_p", false]], "fp_batchnormgrad_s\uff08c function\uff09": [[34, "c.fp_batchnormgrad_s", false]], "fp_batchtospace_p\uff08c function\uff09": [[35, "c.fp_batchtospace_p", false]], "fp_batchtospace_s\uff08c function\uff09": [[35, "c.fp_batchtospace_s", false]], "fp_batchtospacend_p\uff08c function\uff09": [[36, "c.fp_batchtospacend_p", false]], "fp_batchtospacend_s\uff08c function\uff09": [[36, "c.fp_batchtospacend_s", false]], "fp_biasadd_p\uff08c function\uff09": [[37, "c.fp_biasadd_p", false]], "fp_biasadd_s\uff08c function\uff09": [[37, "c.fp_biasadd_s", false]], "fp_biasaddgrad_p\uff08c function\uff09": [[38, "c.fp_biasaddgrad_p", false]], "fp_biasaddgrad_s\uff08c function\uff09": [[38, "c.fp_biasaddgrad_s", false]], "fp_binarycrossentropy_p\uff08c function\uff09": [[39, "c.fp_binarycrossentropy_p", false]], "fp_binarycrossentropy_s\uff08c function\uff09": [[39, "c.fp_binarycrossentropy_s", false]], "fp_binarycrossentropygrad_p\uff08c function\uff09": [[40, "c.fp_binarycrossentropygrad_p", false]], "fp_binarycrossentropygrad_s\uff08c function\uff09": [[40, "c.fp_binarycrossentropygrad_s", false]], "fp_broadcastto_p\uff08c function\uff09": [[41, "c.fp_broadcastto_p", false]], "fp_broadcastto_s\uff08c function\uff09": [[41, "c.fp_broadcastto_s", false]], "fp_ceil_p\uff08c function\uff09": [[43, "c.fp_ceil_p", false]], "fp_ceil_s\uff08c function\uff09": [[43, "c.fp_ceil_s", false]], "fp_celu_p\uff08c function\uff09": [[12, "c.fp_celu_p", false]], "fp_celu_s\uff08c function\uff09": [[12, "c.fp_celu_s", false]], "fp_clip_p\uff08c function\uff09": [[12, "c.fp_clip_p", false], [44, "c.fp_clip_p", false]], "fp_clip_s\uff08c function\uff09": [[12, "c.fp_clip_s", false], [44, "c.fp_clip_s", false]], "fp_concat_p\uff08c function\uff09": [[45, "c.fp_concat_p", false]], "fp_concat_s\uff08c function\uff09": [[45, "c.fp_concat_s", false]], "fp_constant_of_shape_p\uff08c function\uff09": [[46, "c.fp_constant_of_shape_p", false]], "fp_constant_of_shape_s\uff08c function\uff09": [[46, "c.fp_constant_of_shape_s", false]], "fp_conv2d_p\uff08c function\uff09": [[47, "c.fp_conv2d_p", false]], "fp_conv2d_s\uff08c function\uff09": [[47, "c.fp_conv2d_s", false]], "fp_conv2dbackpropfilterfusion_p\uff08c function\uff09": [[49, "c.fp_conv2dbackpropfilterfusion_p", false]], "fp_conv2dbackpropfilterfusion_s\uff08c function\uff09": [[49, "c.fp_conv2dbackpropfilterfusion_s", false]], "fp_conv2dbackpropinputfusion_p\uff08c function\uff09": [[50, "c.fp_conv2dbackpropinputfusion_p", false]], "fp_conv2dbackpropinputfusion_s\uff08c function\uff09": [[50, "c.fp_conv2dbackpropinputfusion_s", false]], "fp_convtranspose_p\uff08c function\uff09": [[48, "c.fp_convtranspose_p", false]], "fp_convtranspose_s\uff08c function\uff09": [[48, "c.fp_convtranspose_s", false]], "fp_cos_p\uff08c function\uff09": [[51, "c.fp_cos_p", false]], "fp_cos_s\uff08c function\uff09": [[51, "c.fp_cos_s", false]], "fp_crop_and_resize_anycore\uff08c function\uff09": [[53, "c.fp_crop_and_resize_anycore", false]], "fp_cumsum_p\uff08c function\uff09": [[54, "c.fp_cumsum_p", false]], "fp_cumsum_s\uff08c function\uff09": [[54, "c.fp_cumsum_s", false]], "fp_deconvgradfilter_p\uff08c function\uff09": [[58, "c.fp_deconvgradfilter_p", false]], "fp_deconvgradfilter_s\uff08c function\uff09": [[58, "c.fp_deconvgradfilter_s", false]], "fp_depthtospace_p\uff08c function\uff09": [[59, "c.fp_depthtospace_p", false]], "fp_depthtospace_s\uff08c function\uff09": [[59, "c.fp_depthtospace_s", false]], "fp_detection_post_process_p\uff08c function\uff09": [[60, "c.fp_detection_post_process_p", false]], "fp_detection_post_process_s\uff08c function\uff09": [[60, "c.fp_detection_post_process_s", false]], "fp_div_fusion_p\uff08c function\uff09": [[61, "c.fp_div_fusion_p", false]], "fp_div_fusion_s\uff08c function\uff09": [[61, "c.fp_div_fusion_s", false]], "fp_dropout_p\uff08c function\uff09": [[63, "c.fp_dropout_p", false]], "fp_dropout_s\uff08c function\uff09": [[63, "c.fp_dropout_s", false]], "fp_dropoutgrad_p\uff08c function\uff09": [[64, "c.fp_dropoutgrad_p", false]], "fp_dropoutgrad_s\uff08c function\uff09": [[64, "c.fp_dropoutgrad_s", false]], "fp_eltwise_p\uff08c function\uff09": [[67, "c.fp_eltwise_p", false]], "fp_eltwise_s\uff08c function\uff09": [[67, "c.fp_eltwise_s", false]], "fp_elu_grad_p\uff08c function\uff09": [[13, "c.fp_elu_grad_p", false]], "fp_elu_grad_s\uff08c function\uff09": [[13, "c.fp_elu_grad_s", false]], "fp_elu_p\uff08c function\uff09": [[12, "c.fp_elu_p", false], [68, "c.fp_elu_p", false]], "fp_elu_s\uff08c function\uff09": [[12, "c.fp_elu_s", false], [68, "c.fp_elu_s", false]], "fp_embeddinglookup_p\uff08c function\uff09": [[69, "c.fp_embeddinglookup_p", false]], "fp_embeddinglookup_s\uff08c function\uff09": [[69, "c.fp_embeddinglookup_s", false]], "fp_equal_p\uff08c function\uff09": [[70, "c.fp_equal_p", false]], "fp_equal_s\uff08c function\uff09": [[70, "c.fp_equal_s", false]], "fp_erf_p\uff08c function\uff09": [[71, "c.fp_erf_p", false]], "fp_erf_s\uff08c function\uff09": [[71, "c.fp_erf_s", false]], "fp_expfusion_p\uff08c function\uff09": [[73, "c.fp_expfusion_p", false]], "fp_expfusion_s\uff08c function\uff09": [[73, "c.fp_expfusion_s", false]], "fp_extract_features_p\uff08c function\uff09": [[55, "c.fp_extract_features_p", false]], "fp_extract_features_s\uff08c function\uff09": [[55, "c.fp_extract_features_s", false]], "fp_fake_quant_with_min_max_vars_per_channel_p\uff08c function\uff09": [[75, "c.fp_fake_quant_with_min_max_vars_per_channel_p", false]], "fp_fake_quant_with_min_max_vars_per_channel_s\uff08c function\uff09": [[75, "c.fp_fake_quant_with_min_max_vars_per_channel_s", false]], "fp_fake_quant_with_min_max_vars_p\uff08c function\uff09": [[74, "c.fp_fake_quant_with_min_max_vars_p", false]], "fp_fake_quant_with_min_max_vars_s\uff08c function\uff09": [[74, "c.fp_fake_quant_with_min_max_vars_s", false]], "fp_fill_p\uff08c function\uff09": [[78, "c.fp_fill_p", false]], "fp_fill_s\uff08c function\uff09": [[78, "c.fp_fill_s", false]], "fp_flattengrad_p\uff08c function\uff09": [[81, "c.fp_flattengrad_p", false]], "fp_flattengrad_s\uff08c function\uff09": [[81, "c.fp_flattengrad_s", false]], "fp_floor_p\uff08c function\uff09": [[82, "c.fp_floor_p", false]], "fp_floor_s\uff08c function\uff09": [[82, "c.fp_floor_s", false]], "fp_floordiv_p\uff08c function\uff09": [[83, "c.fp_floordiv_p", false]], "fp_floordiv_s\uff08c function\uff09": [[83, "c.fp_floordiv_s", false]], "fp_floormod_p\uff08c function\uff09": [[84, "c.fp_floormod_p", false]], "fp_floormod_s\uff08c function\uff09": [[84, "c.fp_floormod_s", false]], "fp_formattranspose_p\uff08c function\uff09": [[85, "c.fp_formattranspose_p", false]], "fp_formattranspose_s\uff08c function\uff09": [[85, "c.fp_formattranspose_s", false]], "fp_fullconnection_p\uff08c function\uff09": [[86, "c.fp_fullconnection_p", false]], "fp_fullconnection_s\uff08c function\uff09": [[86, "c.fp_fullconnection_s", false]], "fp_fusedbatchnorm_p\uff08c function\uff09": [[87, "c.fp_fusedbatchnorm_p", false]], "fp_fusedbatchnorm_s\uff08c function\uff09": [[87, "c.fp_fusedbatchnorm_s", false]], "fp_gather_nd_p\uff08c function\uff09": [[89, "c.fp_gather_nd_p", false]], "fp_gather_nd_s\uff08c function\uff09": [[89, "c.fp_gather_nd_s", false]], "fp_gather_p\uff08c function\uff09": [[88, "c.fp_gather_p", false]], "fp_gather_s\uff08c function\uff09": [[88, "c.fp_gather_s", false]], "fp_gatherd_p\uff08c function\uff09": [[90, "c.fp_gatherd_p", false]], "fp_gatherd_s\uff08c function\uff09": [[90, "c.fp_gatherd_s", false]], "fp_gelu_grad_p\uff08c function\uff09": [[13, "c.fp_gelu_grad_p", false]], "fp_gelu_grad_s\uff08c function\uff09": [[13, "c.fp_gelu_grad_s", false]], "fp_gelu_p\uff08c function\uff09": [[12, "c.fp_gelu_p", false]], "fp_gelu_s\uff08c function\uff09": [[12, "c.fp_gelu_s", false]], "fp_glu_p\uff08c function\uff09": [[91, "c.fp_glu_p", false]], "fp_glu_s\uff08c function\uff09": [[91, "c.fp_glu_s", false]], "fp_graddiv1l_p\uff08c function\uff09": [[62, "c.fp_graddiv1l_p", false]], "fp_graddiv1l_s\uff08c function\uff09": [[62, "c.fp_graddiv1l_s", false]], "fp_graddiv2l_p\uff08c function\uff09": [[62, "c.fp_graddiv2l_p", false]], "fp_graddiv2l_s\uff08c function\uff09": [[62, "c.fp_graddiv2l_s", false]], "fp_graddiv_p\uff08c function\uff09": [[62, "c.fp_graddiv_p", false]], "fp_graddiv_s\uff08c function\uff09": [[62, "c.fp_graddiv_s", false]], "fp_gradmul1l_p\uff08c function\uff09": [[131, "c.fp_gradmul1l_p", false]], "fp_gradmul1l_s\uff08c function\uff09": [[131, "c.fp_gradmul1l_s", false]], "fp_gradmul2l_p\uff08c function\uff09": [[131, "c.fp_gradmul2l_p", false]], "fp_gradmul2l_s\uff08c function\uff09": [[131, "c.fp_gradmul2l_s", false]], "fp_gradmul_p\uff08c function\uff09": [[131, "c.fp_gradmul_p", false]], "fp_gradmul_s\uff08c function\uff09": [[131, "c.fp_gradmul_s", false]], "fp_greater_p\uff08c function\uff09": [[92, "c.fp_greater_p", false]], "fp_greater_s\uff08c function\uff09": [[92, "c.fp_greater_s", false]], "fp_greaterequal_p\uff08c function\uff09": [[93, "c.fp_greaterequal_p", false]], "fp_greaterequal_s\uff08c function\uff09": [[93, "c.fp_greaterequal_s", false]], "fp_groupnormfusion_p\uff08c function\uff09": [[94, "c.fp_groupnormfusion_p", false]], "fp_groupnormfusion_s\uff08c function\uff09": [[94, "c.fp_groupnormfusion_s", false]], "fp_gru_p\uff08c function\uff09": [[95, "c.fp_Gru_p", false]], "fp_gru_s\uff08c function\uff09": [[95, "c.fp_Gru_s", false]], "fp_h_sigmoid_grad_p\uff08c function\uff09": [[13, "c.fp_h_sigmoid_grad_p", false]], "fp_h_sigmoid_grad_s\uff08c function\uff09": [[13, "c.fp_h_sigmoid_grad_s", false]], "fp_h_swish_grad_p\uff08c function\uff09": [[13, "c.fp_h_swish_grad_p", false]], "fp_h_swish_grad_s\uff08c function\uff09": [[13, "c.fp_h_swish_grad_s", false]], "fp_hard_shrink_grad_p\uff08c function\uff09": [[13, "c.fp_hard_shrink_grad_p", false]], "fp_hard_shrink_grad_s\uff08c function\uff09": [[13, "c.fp_hard_shrink_grad_s", false]], "fp_hardshrink_p\uff08c function\uff09": [[12, "c.fp_hardshrink_p", false]], "fp_hardshrink_s\uff08c function\uff09": [[12, "c.fp_hardshrink_s", false]], "fp_hardtanh_p\uff08c function\uff09": [[12, "c.fp_hardtanh_p", false]], "fp_hardtanh_s\uff08c function\uff09": [[12, "c.fp_hardtanh_s", false]], "fp_hsigmoid_p\uff08c function\uff09": [[12, "c.fp_hsigmoid_p", false]], "fp_hsigmoid_s\uff08c function\uff09": [[12, "c.fp_hsigmoid_s", false]], "fp_hswish_p\uff08c function\uff09": [[12, "c.fp_hswish_p", false]], "fp_hswish_s\uff08c function\uff09": [[12, "c.fp_hswish_s", false]], "fp_instancenorm_p\uff08c function\uff09": [[97, "c.fp_instancenorm_p", false]], "fp_instancenorm_s\uff08c function\uff09": [[97, "c.fp_instancenorm_s", false]], "fp_isfinite_p\uff08c function\uff09": [[99, "c.fp_isfinite_p", false]], "fp_isfinite_s\uff08c function\uff09": [[99, "c.fp_isfinite_s", false]], "fp_l2norm_p\uff08c function\uff09": [[100, "c.fp_l2norm_p", false]], "fp_l2norm_s\uff08c function\uff09": [[100, "c.fp_l2norm_s", false]], "fp_l_relu_grad_p\uff08c function\uff09": [[13, "c.fp_l_relu_grad_p", false]], "fp_l_relu_grad_s\uff08c function\uff09": [[13, "c.fp_l_relu_grad_s", false]], "fp_layernormfusion_p\uff08c function\uff09": [[101, "c.fp_layernormfusion_p", false]], "fp_layernormfusion_s\uff08c function\uff09": [[101, "c.fp_layernormfusion_s", false]], "fp_layernormgrad_p\uff08c function\uff09": [[102, "c.fp_layernormgrad_p", false]], "fp_layernormgrad_s\uff08c function\uff09": [[102, "c.fp_layernormgrad_s", false]], "fp_leaky_relu_p\uff08c function\uff09": [[103, "c.fp_leaky_relu_p", false]], "fp_leaky_relu_s\uff08c function\uff09": [[103, "c.fp_leaky_relu_s", false]], "fp_less_p\uff08c function\uff09": [[104, "c.fp_less_p", false]], "fp_less_s\uff08c function\uff09": [[104, "c.fp_less_s", false]], "fp_lessequal_p\uff08c function\uff09": [[105, "c.fp_lessequal_p", false]], "fp_lessequal_s\uff08c function\uff09": [[105, "c.fp_lessequal_s", false]], "fp_linspace_p\uff08c function\uff09": [[106, "c.fp_linspace_p", false]], "fp_linspace_s\uff08c function\uff09": [[106, "c.fp_linspace_s", false]], "fp_log1p_p\uff08c function\uff09": [[108, "c.fp_log1p_p", false]], "fp_log1p_s\uff08c function\uff09": [[108, "c.fp_log1p_s", false]], "fp_log_grad_p\uff08c function\uff09": [[109, "c.fp_log_grad_p", false]], "fp_log_grad_s\uff08c function\uff09": [[109, "c.fp_log_grad_s", false]], "fp_log_p\uff08c function\uff09": [[107, "c.fp_log_p", false]], "fp_log_s\uff08c function\uff09": [[107, "c.fp_log_s", false]], "fp_logical_not_p\uff08c function\uff09": [[110, "c.fp_logical_not_p", false]], "fp_logical_not_s\uff08c function\uff09": [[110, "c.fp_logical_not_s", false]], "fp_logical_or_p\uff08c function\uff09": [[111, "c.fp_logical_or_p", false]], "fp_logical_or_s\uff08c function\uff09": [[111, "c.fp_logical_or_s", false]], "fp_logsoftmax_p\uff08c function\uff09": [[113, "c.fp_logsoftmax_p", false]], "fp_logsoftmax_s\uff08c function\uff09": [[113, "c.fp_logsoftmax_s", false]], "fp_lpnorm_p\uff08c function\uff09": [[114, "c.fp_lpnorm_p", false]], "fp_lpnorm_s\uff08c function\uff09": [[114, "c.fp_lpnorm_s", false]], "fp_lrelu_p\uff08c function\uff09": [[12, "c.fp_lrelu_p", false]], "fp_lrelu_s\uff08c function\uff09": [[12, "c.fp_lrelu_s", false]], "fp_lrn_p\uff08c function\uff09": [[115, "c.fp_lrn_p", false]], "fp_lrn_s\uff08c function\uff09": [[115, "c.fp_lrn_s", false]], "fp_lsh_projection_p\uff08c function\uff09": [[116, "c.fp_lsh_projection_p", false]], "fp_lsh_projection_s\uff08c function\uff09": [[116, "c.fp_lsh_projection_s", false]], "fp_lstm_p\uff08c function\uff09": [[117, "c.fp_Lstm_p", false]], "fp_lstm_s\uff08c function\uff09": [[117, "c.fp_Lstm_s", false]], "fp_lstmgrad_p\uff08c function\uff09": [[118, "c.fp_lstmgrad_p", false]], "fp_lstmgrad_s\uff08c function\uff09": [[118, "c.fp_lstmgrad_s", false]], "fp_lstmgraddata_p\uff08c function\uff09": [[119, "c.fp_lstmgraddata_p", false]], "fp_lstmgraddata_s\uff08c function\uff09": [[119, "c.fp_lstmgraddata_s", false]], "fp_lstmgradweight_p\uff08c function\uff09": [[120, "c.fp_lstmgradweight_p", false]], "fp_lstmgradweight_s\uff08c function\uff09": [[120, "c.fp_lstmgradweight_s", false]], "fp_matmulfusion_p\uff08c function\uff09": [[121, "c.fp_matmulfusion_p", false]], "fp_matmulfusion_s\uff08c function\uff09": [[121, "c.fp_matmulfusion_s", false]], "fp_maximum_p\uff08c function\uff09": [[122, "c.fp_maximum_p", false]], "fp_maximum_s\uff08c function\uff09": [[122, "c.fp_maximum_s", false]], "fp_maximumgrad_p\uff08c function\uff09": [[123, "c.fp_maximumgrad_p", false]], "fp_maximumgrad_s\uff08c function\uff09": [[123, "c.fp_maximumgrad_s", false]], "fp_maxpool_fusion_p\uff08c function\uff09": [[124, "c.fp_maxpool_fusion_p", false]], "fp_maxpool_fusion_s\uff08c function\uff09": [[124, "c.fp_maxpool_fusion_s", false]], "fp_maxpool_grad_p\uff08c function\uff09": [[125, "c.fp_maxpool_grad_p", false]], "fp_maxpool_grad_s\uff08c function\uff09": [[125, "c.fp_maxpool_grad_s", false]], "fp_mfcc_p\uff08c function\uff09": [[126, "c.fp_mfcc_p", false]], "fp_mfcc_s\uff08c function\uff09": [[126, "c.fp_mfcc_s", false]], "fp_minimum_p\uff08c function\uff09": [[127, "c.fp_minimum_p", false]], "fp_minimum_s\uff08c function\uff09": [[127, "c.fp_minimum_s", false]], "fp_minimumgrad_p\uff08c function\uff09": [[128, "c.fp_minimumgrad_p", false]], "fp_minimumgrad_s\uff08c function\uff09": [[128, "c.fp_minimumgrad_s", false]], "fp_mod_p\uff08c function\uff09": [[129, "c.fp_mod_p", false]], "fp_mod_s\uff08c function\uff09": [[129, "c.fp_mod_s", false]], "fp_mul_p\uff08c function\uff09": [[130, "c.fp_mul_p", false]], "fp_mul_s\uff08c function\uff09": [[130, "c.fp_mul_s", false]], "fp_neg_grad_p\uff08c function\uff09": [[133, "c.fp_neg_grad_p", false]], "fp_neg_grad_s\uff08c function\uff09": [[133, "c.fp_neg_grad_s", false]], "fp_neg_p\uff08c function\uff09": [[132, "c.fp_neg_p", false]], "fp_neg_s\uff08c function\uff09": [[132, "c.fp_neg_s", false]], "fp_nllloss_p\uff08c function\uff09": [[134, "c.fp_nllloss_p", false]], "fp_nllloss_s\uff08c function\uff09": [[134, "c.fp_nllloss_s", false]], "fp_nlllossgrad_p\uff08c function\uff09": [[135, "c.fp_nlllossgrad_p", false]], "fp_nlllossgrad_s\uff08c function\uff09": [[135, "c.fp_nlllossgrad_s", false]], "fp_non_max_suppression_p\uff08c function\uff09": [[136, "c.fp_non_max_suppression_p", false]], "fp_non_max_suppression_s\uff08c function\uff09": [[136, "c.fp_non_max_suppression_s", false]], "fp_nonzero_p\uff08c function\uff09": [[137, "c.fp_nonzero_p", false]], "fp_nonzero_s\uff08c function\uff09": [[137, "c.fp_nonzero_s", false]], "fp_not_equal_p\uff08c function\uff09": [[138, "c.fp_not_equal_p", false]], "fp_not_equal_s\uff08c function\uff09": [[138, "c.fp_not_equal_s", false]], "fp_onehot_p\uff08c function\uff09": [[139, "c.fp_onehot_p", false]], "fp_onehot_s\uff08c function\uff09": [[139, "c.fp_onehot_s", false]], "fp_ones_like_p\uff08c function\uff09": [[140, "c.fp_ones_like_p", false]], "fp_ones_like_s\uff08c function\uff09": [[140, "c.fp_ones_like_s", false]], "fp_padfusion_p\uff08c function\uff09": [[141, "c.fp_padfusion_p", false]], "fp_padfusion_s\uff08c function\uff09": [[141, "c.fp_padfusion_s", false]], "fp_pow_fusion_p\uff08c function\uff09": [[142, "c.fp_pow_fusion_p", false]], "fp_pow_fusion_s\uff08c function\uff09": [[142, "c.fp_pow_fusion_s", false]], "fp_power_grad_p\uff08c function\uff09": [[143, "c.fp_power_grad_p", false]], "fp_power_grad_s\uff08c function\uff09": [[143, "c.fp_power_grad_s", false]], "fp_prelufusion_p\uff08c function\uff09": [[144, "c.fp_prelufusion_p", false]], "fp_prelufusion_s\uff08c function\uff09": [[144, "c.fp_prelufusion_s", false]], "fp_priorbox_p\uff08c function\uff09": [[145, "c.fp_priorbox_p", false]], "fp_priorbox_s\uff08c function\uff09": [[145, "c.fp_priorbox_s", false]], "fp_quantdata_p\uff08c function\uff09": [[66, "c.fp_QuantData_p", false]], "fp_quantdata_s\uff08c function\uff09": [[66, "c.fp_QuantData_s", false]], "fp_raggedrange_p\uff08c function\uff09": [[147, "c.fp_raggedrange_p", false]], "fp_raggedrange_s\uff08c function\uff09": [[147, "c.fp_raggedrange_s", false]], "fp_random_normal_p\uff08c function\uff09": [[148, "c.fp_random_normal_p", false]], "fp_random_normal_s\uff08c function\uff09": [[148, "c.fp_random_normal_s", false]], "fp_random_standard_normal_p\uff08c function\uff09": [[149, "c.fp_random_standard_normal_p", false]], "fp_random_standard_normal_s\uff08c function\uff09": [[149, "c.fp_random_standard_normal_s", false]], "fp_range_p\uff08c function\uff09": [[150, "c.fp_range_p", false]], "fp_range_s\uff08c function\uff09": [[150, "c.fp_range_s", false]], "fp_real_div_p\uff08c function\uff09": [[152, "c.fp_real_div_p", false]], "fp_real_div_s\uff08c function\uff09": [[152, "c.fp_real_div_s", false]], "fp_reciprocal_p\uff08c function\uff09": [[153, "c.fp_reciprocal_p", false]], "fp_reciprocal_s\uff08c function\uff09": [[153, "c.fp_reciprocal_s", false]], "fp_reduce_p\uff08c function\uff09": [[154, "c.fp_reduce_p", false]], "fp_reduce_s\uff08c function\uff09": [[154, "c.fp_reduce_s", false]], "fp_reduceall_p\uff08c function\uff09": [[21, "c.fp_reduceall_p", false]], "fp_reduceall_s\uff08c function\uff09": [[21, "c.fp_reduceall_s", false]], "fp_reducescatter_p\uff08c function\uff09": [[155, "c.fp_reducescatter_p", false]], "fp_reducescatter_s\uff08c function\uff09": [[155, "c.fp_reducescatter_s", false]], "fp_relu6_grad_p\uff08c function\uff09": [[13, "c.fp_relu6_grad_p", false]], "fp_relu6_grad_s\uff08c function\uff09": [[13, "c.fp_relu6_grad_s", false]], "fp_relu6_p\uff08c function\uff09": [[12, "c.fp_relu6_p", false]], "fp_relu6_s\uff08c function\uff09": [[12, "c.fp_relu6_s", false]], "fp_relu_grad_p\uff08c function\uff09": [[13, "c.fp_relu_grad_p", false]], "fp_relu_grad_s\uff08c function\uff09": [[13, "c.fp_relu_grad_s", false]], "fp_relu_p\uff08c function\uff09": [[12, "c.fp_relu_p", false]], "fp_relu_s\uff08c function\uff09": [[12, "c.fp_relu_s", false]], "fp_reshape_p\uff08c function\uff09": [[156, "c.fp_reshape_p", false]], "fp_reshape_s\uff08c function\uff09": [[156, "c.fp_reshape_s", false]], "fp_resize_anycore\uff08c function\uff09": [[157, "c.fp_resize_anycore", false]], "fp_resizebilineargrad_p\uff08c function\uff09": [[158, "c.fp_resizebilineargrad_p", false]], "fp_resizebilineargrad_s\uff08c function\uff09": [[158, "c.fp_resizebilineargrad_s", false]], "fp_resizenearestneighborgrad_p\uff08c function\uff09": [[158, "c.fp_resizenearestneighborgrad_p", false]], "fp_resizenearestneighborgrad_s\uff08c function\uff09": [[158, "c.fp_resizenearestneighborgrad_s", false]], "fp_roipooling_p\uff08c function\uff09": [[162, "c.fp_roipooling_p", false]], "fp_roipooling_s\uff08c function\uff09": [[162, "c.fp_roipooling_s", false]], "fp_round_p\uff08c function\uff09": [[163, "c.fp_round_p", false]], "fp_round_s\uff08c function\uff09": [[163, "c.fp_round_s", false]], "fp_rsqrt_p\uff08c function\uff09": [[164, "c.fp_rsqrt_p", false]], "fp_rsqrt_s\uff08c function\uff09": [[164, "c.fp_rsqrt_s", false]], "fp_rsqrtgrad_p\uff08c function\uff09": [[165, "c.fp_rsqrtgrad_p", false]], "fp_rsqrtgrad_s\uff08c function\uff09": [[165, "c.fp_rsqrtgrad_s", false]], "fp_scalefusion_p\uff08c function\uff09": [[166, "c.fp_scalefusion_p", false]], "fp_scalefusion_s\uff08c function\uff09": [[166, "c.fp_scalefusion_s", false]], "fp_scatter_elements_p\uff08c function\uff09": [[167, "c.fp_scatter_elements_p", false]], "fp_scatter_elements_s\uff08c function\uff09": [[167, "c.fp_scatter_elements_s", false]], "fp_scatter_nd_p\uff08c function\uff09": [[168, "c.fp_scatter_nd_p", false]], "fp_scatter_nd_s\uff08c function\uff09": [[168, "c.fp_scatter_nd_s", false]], "fp_scatter_nd_update_p\uff08c function\uff09": [[169, "c.fp_scatter_nd_update_p", false]], "fp_scatter_nd_update_s\uff08c function\uff09": [[169, "c.fp_scatter_nd_update_s", false]], "fp_select_p\uff08c function\uff09": [[170, "c.fp_select_p", false]], "fp_select_s\uff08c function\uff09": [[170, "c.fp_select_s", false]], "fp_sgd_p\uff08c function\uff09": [[171, "c.fp_sgd_p", false]], "fp_sgd_s\uff08c function\uff09": [[171, "c.fp_sgd_s", false]], "fp_sigmoid_grad_p\uff08c function\uff09": [[13, "c.fp_sigmoid_grad_p", false]], "fp_sigmoid_grad_s\uff08c function\uff09": [[13, "c.fp_sigmoid_grad_s", false]], "fp_sigmoid_p\uff08c function\uff09": [[12, "c.fp_sigmoid_p", false]], "fp_sigmoid_s\uff08c function\uff09": [[12, "c.fp_sigmoid_s", false]], "fp_sigmoidcrossentropywithlogits_p\uff08c function\uff09": [[174, "c.fp_sigmoidcrossentropywithlogits_p", false]], "fp_sigmoidcrossentropywithlogits_s\uff08c function\uff09": [[174, "c.fp_sigmoidcrossentropywithlogits_s", false]], "fp_sigmoidcrossentropywithlogitsgrad_p\uff08c function\uff09": [[173, "c.fp_sigmoidcrossentropywithlogitsgrad_p", false]], "fp_sigmoidcrossentropywithlogitsgrad_s\uff08c function\uff09": [[173, "c.fp_sigmoidcrossentropywithlogitsgrad_s", false]], "fp_sin_p\uff08c function\uff09": [[175, "c.fp_sin_p", false]], "fp_sin_s\uff08c function\uff09": [[175, "c.fp_sin_s", false]], "fp_slice_p\uff08c function\uff09": [[178, "c.fp_slice_p", false]], "fp_slice_s\uff08c function\uff09": [[178, "c.fp_slice_s", false]], "fp_smoothl1loss_p\uff08c function\uff09": [[179, "c.fp_smoothl1loss_p", false]], "fp_smoothl1loss_s\uff08c function\uff09": [[179, "c.fp_smoothl1loss_s", false]], "fp_smoothl1lossgrad_p\uff08c function\uff09": [[180, "c.fp_smoothl1lossgrad_p", false]], "fp_smoothl1lossgrad_s\uff08c function\uff09": [[180, "c.fp_smoothl1lossgrad_s", false]], "fp_softmax_cross_entropy_with_logits_p\uff08c function\uff09": [[182, "c.fp_softmax_cross_entropy_with_logits_p", false]], "fp_softmax_cross_entropy_with_logits_s\uff08c function\uff09": [[182, "c.fp_softmax_cross_entropy_with_logits_s", false]], "fp_softmax_p\uff08c function\uff09": [[181, "c.fp_softmax_p", false]], "fp_softmax_s\uff08c function\uff09": [[181, "c.fp_softmax_s", false]], "fp_softplus_grad_p\uff08c function\uff09": [[13, "c.fp_softplus_grad_p", false]], "fp_softplus_grad_s\uff08c function\uff09": [[13, "c.fp_softplus_grad_s", false]], "fp_softplus_p\uff08c function\uff09": [[12, "c.fp_softplus_p", false]], "fp_softplus_s\uff08c function\uff09": [[12, "c.fp_softplus_s", false]], "fp_softshrink_grad_p\uff08c function\uff09": [[13, "c.fp_softshrink_grad_p", false]], "fp_softshrink_grad_s\uff08c function\uff09": [[13, "c.fp_softshrink_grad_s", false]], "fp_softshrink_p\uff08c function\uff09": [[12, "c.fp_softshrink_p", false]], "fp_softshrink_s\uff08c function\uff09": [[12, "c.fp_softshrink_s", false]], "fp_softsignopt_p\uff08c function\uff09": [[12, "c.fp_softsignopt_p", false]], "fp_softsignopt_s\uff08c function\uff09": [[12, "c.fp_softsignopt_s", false]], "fp_spacetobatch_p\uff08c function\uff09": [[183, "c.fp_spacetobatch_p", false]], "fp_spacetobatch_s\uff08c function\uff09": [[183, "c.fp_spacetobatch_s", false]], "fp_spacetobatchnd_p\uff08c function\uff09": [[184, "c.fp_spacetobatchnd_p", false]], "fp_spacetobatchnd_s\uff08c function\uff09": [[184, "c.fp_spacetobatchnd_s", false]], "fp_spacetodepth_p\uff08c function\uff09": [[185, "c.fp_spacetodepth_p", false]], "fp_spacetodepth_s\uff08c function\uff09": [[185, "c.fp_spacetodepth_s", false]], "fp_sparse_softmax_cross_entropy_with_logits_p\uff08c function\uff09": [[186, "c.fp_sparse_softmax_cross_entropy_with_logits_p", false]], "fp_sparse_softmax_cross_entropy_with_logits_s\uff08c function\uff09": [[186, "c.fp_sparse_softmax_cross_entropy_with_logits_s", false]], "fp_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.fp_sparsefillemptyrows_p", false]], "fp_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.fp_sparsefillemptyrows_s", false]], "fp_sparsesegmentsum_p\uff08c function\uff09": [[189, "c.fp_sparsesegmentsum_p", false]], "fp_sparsesegmentsum_s\uff08c function\uff09": [[189, "c.fp_sparsesegmentsum_s", false]], "fp_sparsetodense_p\uff08c function\uff09": [[190, "c.fp_sparsetodense_p", false]], "fp_sparsetodense_s\uff08c function\uff09": [[190, "c.fp_sparsetodense_s", false]], "fp_splice_p\uff08c function\uff09": [[191, "c.fp_splice_p", false]], "fp_splice_s\uff08c function\uff09": [[191, "c.fp_splice_s", false]], "fp_split_p\uff08c function\uff09": [[192, "c.fp_split_p", false]], "fp_split_s\uff08c function\uff09": [[192, "c.fp_split_s", false]], "fp_split_with_overlap_p\uff08c function\uff09": [[193, "c.fp_split_with_overlap_p", false]], "fp_split_with_overlap_s\uff08c function\uff09": [[193, "c.fp_split_with_overlap_s", false]], "fp_sqrt_p\uff08c function\uff09": [[194, "c.fp_sqrt_p", false]], "fp_sqrt_s\uff08c function\uff09": [[194, "c.fp_sqrt_s", false]], "fp_sqrtgrad_p\uff08c function\uff09": [[195, "c.fp_sqrtgrad_p", false]], "fp_sqrtgrad_s\uff08c function\uff09": [[195, "c.fp_sqrtgrad_s", false]], "fp_square_p\uff08c function\uff09": [[196, "c.fp_square_p", false]], "fp_square_s\uff08c function\uff09": [[196, "c.fp_square_s", false]], "fp_squaredifference_p\uff08c function\uff09": [[197, "c.fp_squaredifference_p", false]], "fp_squaredifference_s\uff08c function\uff09": [[197, "c.fp_squaredifference_s", false]], "fp_stack_p\uff08c function\uff09": [[199, "c.fp_stack_p", false]], "fp_stack_s\uff08c function\uff09": [[199, "c.fp_stack_s", false]], "fp_stridedslicegrad_p\uff08c function\uff09": [[201, "c.fp_stridedslicegrad_p", false]], "fp_stridedslicegrad_s\uff08c function\uff09": [[201, "c.fp_stridedslicegrad_s", false]], "fp_subext_p\uff08c function\uff09": [[202, "c.fp_subext_p", false]], "fp_subext_s\uff08c function\uff09": [[202, "c.fp_subext_s", false]], "fp_subgrad_p\uff08c function\uff09": [[203, "c.fp_subgrad_p", false]], "fp_subgrad_s\uff08c function\uff09": [[203, "c.fp_subgrad_s", false]], "fp_subrelu6_p\uff08c function\uff09": [[202, "c.fp_subrelu6_p", false]], "fp_subrelu6_s\uff08c function\uff09": [[202, "c.fp_subrelu6_s", false]], "fp_subrelu_p\uff08c function\uff09": [[202, "c.fp_subrelu_p", false]], "fp_subrelu_s\uff08c function\uff09": [[202, "c.fp_subrelu_s", false]], "fp_swish_p\uff08c function\uff09": [[12, "c.fp_swish_p", false]], "fp_swish_s\uff08c function\uff09": [[12, "c.fp_swish_s", false]], "fp_tanh_grad_p\uff08c function\uff09": [[13, "c.fp_tanh_grad_p", false]], "fp_tanh_grad_s\uff08c function\uff09": [[13, "c.fp_tanh_grad_s", false]], "fp_tanh_p\uff08c function\uff09": [[12, "c.fp_tanh_p", false]], "fp_tanh_s\uff08c function\uff09": [[12, "c.fp_tanh_s", false]], "fp_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.fp_tensor_scatter_add_p", false]], "fp_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.fp_tensor_scatter_add_s", false]], "fp_tensorarrayread_p\uff08c function\uff09": [[208, "c.fp_tensorarrayread_p", false]], "fp_tensorarrayread_s\uff08c function\uff09": [[208, "c.fp_tensorarrayread_s", false]], "fp_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.fp_tensorlistfromtensor_p", false]], "fp_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.fp_tensorlistfromtensor_s", false]], "fp_tile_p\uff08c function\uff09": [[215, "c.fp_tile_p", false]], "fp_tile_s\uff08c function\uff09": [[215, "c.fp_tile_s", false]], "fp_to_i8_quant_p\uff08c function\uff09": [[146, "c.fp_to_i8_quant_p", false]], "fp_to_i8_quant_s\uff08c function\uff09": [[146, "c.fp_to_i8_quant_s", false]], "fp_topk_fusion_p\uff08c function\uff09": [[216, "c.fp_topk_fusion_p", false]], "fp_topk_fusion_s\uff08c function\uff09": [[216, "c.fp_topk_fusion_s", false]], "fp_transpose_p\uff08c function\uff09": [[217, "c.fp_transpose_p", false]], "fp_transpose_s\uff08c function\uff09": [[217, "c.fp_transpose_s", false]], "fp_tril_p\uff08c function\uff09": [[218, "c.fp_tril_p", false]], "fp_tril_s\uff08c function\uff09": [[218, "c.fp_tril_s", false]], "fp_triu_p\uff08c function\uff09": [[219, "c.fp_triu_p", false]], "fp_triu_s\uff08c function\uff09": [[219, "c.fp_triu_s", false]], "fp_uniform_real_p\uff08c function\uff09": [[220, "c.fp_uniform_real_p", false]], "fp_uniform_real_s\uff08c function\uff09": [[220, "c.fp_uniform_real_s", false]], "fp_unique_p\uff08c function\uff09": [[221, "c.fp_Unique_p", false]], "fp_unique_s\uff08c function\uff09": [[221, "c.fp_Unique_s", false]], "fp_unsorted_segment_sum_p\uff08c function\uff09": [[222, "c.fp_unsorted_segment_sum_p", false]], "fp_unsorted_segment_sum_s\uff08c function\uff09": [[222, "c.fp_unsorted_segment_sum_s", false]], "fp_where_p\uff08c function\uff09": [[225, "c.fp_where_p", false]], "fp_where_s\uff08c function\uff09": [[225, "c.fp_where_s", false]], "fp_zerolike_p\uff08c function\uff09": [[226, "c.fp_zerolike_p", false]], "fp_zerolike_s\uff08c function\uff09": [[226, "c.fp_zerolike_s", false]], "hp_abs_p\uff08c function\uff09": [[10, "c.hp_abs_p", false]], "hp_abs_s\uff08c function\uff09": [[10, "c.hp_abs_s", false]], "hp_absgrad_p\uff08c function\uff09": [[11, "c.hp_absgrad_p", false]], "hp_absgrad_s\uff08c function\uff09": [[11, "c.hp_absgrad_s", false]], "hp_adamweightdecay_p\uff08c function\uff09": [[15, "c.hp_adamweightdecay_p", false]], "hp_adamweightdecay_s\uff08c function\uff09": [[15, "c.hp_adamweightdecay_s", false]], "hp_adder_p\uff08c function\uff09": [[16, "c.hp_adder_p", false]], "hp_adder_s\uff08c function\uff09": [[16, "c.hp_adder_s", false]], "hp_addext_p\uff08c function\uff09": [[17, "c.hp_addext_p", false]], "hp_addext_s\uff08c function\uff09": [[17, "c.hp_addext_s", false]], "hp_addgrad_p\uff08c function\uff09": [[18, "c.hp_addgrad_p", false]], "hp_addgrad_s\uff08c function\uff09": [[18, "c.hp_addgrad_s", false]], "hp_addrelu6_p\uff08c function\uff09": [[17, "c.hp_addrelu6_p", false]], "hp_addrelu6_s\uff08c function\uff09": [[17, "c.hp_addrelu6_s", false]], "hp_addrelu_p\uff08c function\uff09": [[17, "c.hp_addrelu_p", false]], "hp_addrelu_s\uff08c function\uff09": [[17, "c.hp_addrelu_s", false]], "hp_affine_p\uff08c function\uff09": [[20, "c.hp_affine_p", false]], "hp_affine_s\uff08c function\uff09": [[20, "c.hp_affine_s", false]], "hp_and_p\uff08c function\uff09": [[112, "c.hp_and_p", false]], "hp_and_s\uff08c function\uff09": [[112, "c.hp_and_s", false]], "hp_applymomentum_p\uff08c function\uff09": [[23, "c.hp_applymomentum_p", false]], "hp_applymomentum_s\uff08c function\uff09": [[23, "c.hp_applymomentum_s", false]], "hp_argmax_p\uff08c function\uff09": [[24, "c.hp_argmax_p", false]], "hp_argmax_s\uff08c function\uff09": [[24, "c.hp_argmax_s", false]], "hp_argmin_p\uff08c function\uff09": [[25, "c.hp_argmin_p", false]], "hp_argmin_s\uff08c function\uff09": [[25, "c.hp_argmin_s", false]], "hp_assign_p\uff08c function\uff09": [[27, "c.hp_assign_p", false]], "hp_assign_s\uff08c function\uff09": [[27, "c.hp_assign_s", false]], "hp_assignadd_p\uff08c function\uff09": [[28, "c.hp_assignadd_p", false]], "hp_assignadd_s\uff08c function\uff09": [[28, "c.hp_assignadd_s", false]], "hp_avgpool_fusion_p\uff08c function\uff09": [[31, "c.hp_avgpool_fusion_p", false]], "hp_avgpool_fusion_s\uff08c function\uff09": [[31, "c.hp_avgpool_fusion_s", false]], "hp_avgpoolinggrad_p\uff08c function\uff09": [[32, "c.hp_avgpoolinggrad_p", false]], "hp_avgpoolinggrad_s\uff08c function\uff09": [[32, "c.hp_avgpoolinggrad_s", false]], "hp_batchnorm_p\uff08c function\uff09": [[33, "c.hp_batchnorm_p", false]], "hp_batchnorm_s\uff08c function\uff09": [[33, "c.hp_batchnorm_s", false]], "hp_batchnormgrad_p\uff08c function\uff09": [[34, "c.hp_batchnormgrad_p", false]], "hp_batchnormgrad_s\uff08c function\uff09": [[34, "c.hp_batchnormgrad_s", false]], "hp_batchtospace_p\uff08c function\uff09": [[35, "c.hp_batchtospace_p", false]], "hp_batchtospace_s\uff08c function\uff09": [[35, "c.hp_batchtospace_s", false]], "hp_batchtospacend_p\uff08c function\uff09": [[36, "c.hp_batchtospacend_p", false]], "hp_batchtospacend_s\uff08c function\uff09": [[36, "c.hp_batchtospacend_s", false]], "hp_biasadd_p\uff08c function\uff09": [[37, "c.hp_biasadd_p", false]], "hp_biasadd_s\uff08c function\uff09": [[37, "c.hp_biasadd_s", false]], "hp_biasaddgrad_p\uff08c function\uff09": [[38, "c.hp_biasaddgrad_p", false]], "hp_biasaddgrad_s\uff08c function\uff09": [[38, "c.hp_biasaddgrad_s", false]], "hp_binarycrossentropy_p\uff08c function\uff09": [[39, "c.hp_binarycrossentropy_p", false]], "hp_binarycrossentropy_s\uff08c function\uff09": [[39, "c.hp_binarycrossentropy_s", false]], "hp_binarycrossentropygrad_p\uff08c function\uff09": [[40, "c.hp_binarycrossentropygrad_p", false]], "hp_binarycrossentropygrad_s\uff08c function\uff09": [[40, "c.hp_binarycrossentropygrad_s", false]], "hp_broadcastto_p\uff08c function\uff09": [[41, "c.hp_broadcastto_p", false]], "hp_broadcastto_s\uff08c function\uff09": [[41, "c.hp_broadcastto_s", false]], "hp_ceil_p\uff08c function\uff09": [[43, "c.hp_ceil_p", false]], "hp_ceil_s\uff08c function\uff09": [[43, "c.hp_ceil_s", false]], "hp_celu_p\uff08c function\uff09": [[12, "c.hp_celu_p", false]], "hp_celu_s\uff08c function\uff09": [[12, "c.hp_celu_s", false]], "hp_clip_p\uff08c function\uff09": [[12, "c.hp_clip_p", false], [44, "c.hp_clip_p", false]], "hp_clip_s\uff08c function\uff09": [[12, "c.hp_clip_s", false], [44, "c.hp_clip_s", false]], "hp_concat_p\uff08c function\uff09": [[45, "c.hp_concat_p", false]], "hp_concat_s\uff08c function\uff09": [[45, "c.hp_concat_s", false]], "hp_conv2d_p\uff08c function\uff09": [[47, "c.hp_conv2d_p", false]], "hp_conv2d_s\uff08c function\uff09": [[47, "c.hp_conv2d_s", false]], "hp_conv2dbackpropfilterfusion_p\uff08c function\uff09": [[49, "c.hp_conv2dbackpropfilterfusion_p", false]], "hp_conv2dbackpropfilterfusion_s\uff08c function\uff09": [[49, "c.hp_conv2dbackpropfilterfusion_s", false]], "hp_conv2dbackpropinputfusion_p\uff08c function\uff09": [[50, "c.hp_conv2dbackpropinputfusion_p", false]], "hp_conv2dbackpropinputfusion_s\uff08c function\uff09": [[50, "c.hp_conv2dbackpropinputfusion_s", false]], "hp_convtranspose_p\uff08c function\uff09": [[48, "c.hp_convtranspose_p", false]], "hp_convtranspose_s\uff08c function\uff09": [[48, "c.hp_convtranspose_s", false]], "hp_cos_p\uff08c function\uff09": [[51, "c.hp_cos_p", false]], "hp_cos_s\uff08c function\uff09": [[51, "c.hp_cos_s", false]], "hp_crop_and_resize_anycore\uff08c function\uff09": [[53, "c.hp_crop_and_resize_anycore", false]], "hp_cumsum_p\uff08c function\uff09": [[54, "c.hp_cumsum_p", false]], "hp_cumsum_s\uff08c function\uff09": [[54, "c.hp_cumsum_s", false]], "hp_deconvgradfilter_p\uff08c function\uff09": [[58, "c.hp_deconvgradfilter_p", false]], "hp_deconvgradfilter_s\uff08c function\uff09": [[58, "c.hp_deconvgradfilter_s", false]], "hp_depthtospace_p\uff08c function\uff09": [[59, "c.hp_depthtospace_p", false]], "hp_depthtospace_s\uff08c function\uff09": [[59, "c.hp_depthtospace_s", false]], "hp_detection_post_process_p\uff08c function\uff09": [[60, "c.hp_detection_post_process_p", false]], "hp_detection_post_process_s\uff08c function\uff09": [[60, "c.hp_detection_post_process_s", false]], "hp_div_fusion_p\uff08c function\uff09": [[61, "c.hp_div_fusion_p", false]], "hp_div_fusion_s\uff08c function\uff09": [[61, "c.hp_div_fusion_s", false]], "hp_dropout_p\uff08c function\uff09": [[63, "c.hp_dropout_p", false]], "hp_dropout_s\uff08c function\uff09": [[63, "c.hp_dropout_s", false]], "hp_dropoutgrad_p\uff08c function\uff09": [[64, "c.hp_dropoutgrad_p", false]], "hp_dropoutgrad_s\uff08c function\uff09": [[64, "c.hp_dropoutgrad_s", false]], "hp_eltwise_p\uff08c function\uff09": [[67, "c.hp_eltwise_p", false]], "hp_eltwise_s\uff08c function\uff09": [[67, "c.hp_eltwise_s", false]], "hp_elu_p\uff08c function\uff09": [[12, "c.hp_elu_p", false], [68, "c.hp_elu_p", false]], "hp_elu_s\uff08c function\uff09": [[12, "c.hp_elu_s", false], [68, "c.hp_elu_s", false]], "hp_embeddinglookup_p\uff08c function\uff09": [[69, "c.hp_embeddinglookup_p", false]], "hp_embeddinglookup_s\uff08c function\uff09": [[69, "c.hp_embeddinglookup_s", false]], "hp_equal_p\uff08c function\uff09": [[70, "c.hp_equal_p", false]], "hp_equal_s\uff08c function\uff09": [[70, "c.hp_equal_s", false]], "hp_erf_p\uff08c function\uff09": [[71, "c.hp_erf_p", false]], "hp_erf_s\uff08c function\uff09": [[71, "c.hp_erf_s", false]], "hp_expfusion_p\uff08c function\uff09": [[73, "c.hp_expfusion_p", false]], "hp_expfusion_s\uff08c function\uff09": [[73, "c.hp_expfusion_s", false]], "hp_fake_quant_with_min_max_vars_per_channel_p\uff08c function\uff09": [[75, "c.hp_fake_quant_with_min_max_vars_per_channel_p", false]], "hp_fake_quant_with_min_max_vars_per_channel_s\uff08c function\uff09": [[75, "c.hp_fake_quant_with_min_max_vars_per_channel_s", false]], "hp_fake_quant_with_min_max_vars_p\uff08c function\uff09": [[74, "c.hp_fake_quant_with_min_max_vars_p", false]], "hp_fake_quant_with_min_max_vars_s\uff08c function\uff09": [[74, "c.hp_fake_quant_with_min_max_vars_s", false]], "hp_fill_p\uff08c function\uff09": [[78, "c.hp_fill_p", false]], "hp_fill_s\uff08c function\uff09": [[78, "c.hp_fill_s", false]], "hp_flattengrad_p\uff08c function\uff09": [[81, "c.hp_flattengrad_p", false]], "hp_flattengrad_s\uff08c function\uff09": [[81, "c.hp_flattengrad_s", false]], "hp_floor_p\uff08c function\uff09": [[82, "c.hp_floor_p", false]], "hp_floor_s\uff08c function\uff09": [[82, "c.hp_floor_s", false]], "hp_floordiv_p\uff08c function\uff09": [[83, "c.hp_floordiv_p", false]], "hp_floordiv_s\uff08c function\uff09": [[83, "c.hp_floordiv_s", false]], "hp_floormod_p\uff08c function\uff09": [[84, "c.hp_floormod_p", false]], "hp_floormod_s\uff08c function\uff09": [[84, "c.hp_floormod_s", false]], "hp_formattranspose_p\uff08c function\uff09": [[85, "c.hp_formattranspose_p", false]], "hp_formattranspose_s\uff08c function\uff09": [[85, "c.hp_formattranspose_s", false]], "hp_fullconnection_p\uff08c function\uff09": [[86, "c.hp_fullconnection_p", false]], "hp_fullconnection_s\uff08c function\uff09": [[86, "c.hp_fullconnection_s", false]], "hp_fusedbatchnorm_p\uff08c function\uff09": [[87, "c.hp_fusedbatchnorm_p", false]], "hp_fusedbatchnorm_s\uff08c function\uff09": [[87, "c.hp_fusedbatchnorm_s", false]], "hp_gather_nd_p\uff08c function\uff09": [[89, "c.hp_gather_nd_p", false]], "hp_gather_nd_s\uff08c function\uff09": [[89, "c.hp_gather_nd_s", false]], "hp_gather_p\uff08c function\uff09": [[88, "c.hp_gather_p", false]], "hp_gather_s\uff08c function\uff09": [[88, "c.hp_gather_s", false]], "hp_gelu_p\uff08c function\uff09": [[12, "c.hp_gelu_p", false]], "hp_gelu_s\uff08c function\uff09": [[12, "c.hp_gelu_s", false]], "hp_glu_p\uff08c function\uff09": [[91, "c.hp_glu_p", false]], "hp_glu_s\uff08c function\uff09": [[91, "c.hp_glu_s", false]], "hp_graddiv1l_p\uff08c function\uff09": [[62, "c.hp_graddiv1l_p", false]], "hp_graddiv1l_s\uff08c function\uff09": [[62, "c.hp_graddiv1l_s", false]], "hp_graddiv2l_p\uff08c function\uff09": [[62, "c.hp_graddiv2l_p", false]], "hp_graddiv2l_s\uff08c function\uff09": [[62, "c.hp_graddiv2l_s", false]], "hp_graddiv_p\uff08c function\uff09": [[62, "c.hp_graddiv_p", false]], "hp_graddiv_s\uff08c function\uff09": [[62, "c.hp_graddiv_s", false]], "hp_gradmul1l_p\uff08c function\uff09": [[131, "c.hp_gradmul1l_p", false]], "hp_gradmul1l_s\uff08c function\uff09": [[131, "c.hp_gradmul1l_s", false]], "hp_gradmul2l_p\uff08c function\uff09": [[131, "c.hp_gradmul2l_p", false]], "hp_gradmul2l_s\uff08c function\uff09": [[131, "c.hp_gradmul2l_s", false]], "hp_gradmul_p\uff08c function\uff09": [[131, "c.hp_gradmul_p", false]], "hp_gradmul_s\uff08c function\uff09": [[131, "c.hp_gradmul_s", false]], "hp_greater_p\uff08c function\uff09": [[92, "c.hp_greater_p", false]], "hp_greater_s\uff08c function\uff09": [[92, "c.hp_greater_s", false]], "hp_greaterequal_p\uff08c function\uff09": [[93, "c.hp_greaterequal_p", false]], "hp_greaterequal_s\uff08c function\uff09": [[93, "c.hp_greaterequal_s", false]], "hp_groupnormfusion_p\uff08c function\uff09": [[94, "c.hp_groupnormfusion_p", false]], "hp_groupnormfusion_s\uff08c function\uff09": [[94, "c.hp_groupnormfusion_s", false]], "hp_gru_p\uff08c function\uff09": [[95, "c.hp_Gru_p", false]], "hp_gru_s\uff08c function\uff09": [[95, "c.hp_Gru_s", false]], "hp_hardshrink_p\uff08c function\uff09": [[12, "c.hp_hardshrink_p", false]], "hp_hardshrink_s\uff08c function\uff09": [[12, "c.hp_hardshrink_s", false]], "hp_hardtanh_p\uff08c function\uff09": [[12, "c.hp_hardtanh_p", false]], "hp_hardtanh_s\uff08c function\uff09": [[12, "c.hp_hardtanh_s", false]], "hp_hsigmoid_p\uff08c function\uff09": [[12, "c.hp_hsigmoid_p", false]], "hp_hsigmoid_s\uff08c function\uff09": [[12, "c.hp_hsigmoid_s", false]], "hp_hswish_p\uff08c function\uff09": [[12, "c.hp_hswish_p", false]], "hp_hswish_s\uff08c function\uff09": [[12, "c.hp_hswish_s", false]], "hp_instancenorm_p\uff08c function\uff09": [[97, "c.hp_instancenorm_p", false]], "hp_instancenorm_s\uff08c function\uff09": [[97, "c.hp_instancenorm_s", false]], "hp_isfinite_p\uff08c function\uff09": [[99, "c.hp_isfinite_p", false]], "hp_isfinite_s\uff08c function\uff09": [[99, "c.hp_isfinite_s", false]], "hp_l2norm_p\uff08c function\uff09": [[100, "c.hp_l2norm_p", false]], "hp_l2norm_s\uff08c function\uff09": [[100, "c.hp_l2norm_s", false]], "hp_layernormfusion_p\uff08c function\uff09": [[101, "c.hp_layernormfusion_p", false]], "hp_layernormfusion_s\uff08c function\uff09": [[101, "c.hp_layernormfusion_s", false]], "hp_layernormgrad_p\uff08c function\uff09": [[102, "c.hp_layernormgrad_p", false]], "hp_layernormgrad_s\uff08c function\uff09": [[102, "c.hp_layernormgrad_s", false]], "hp_leaky_relu_p\uff08c function\uff09": [[103, "c.hp_leaky_relu_p", false]], "hp_leaky_relu_s\uff08c function\uff09": [[103, "c.hp_leaky_relu_s", false]], "hp_less_p\uff08c function\uff09": [[104, "c.hp_less_p", false]], "hp_less_s\uff08c function\uff09": [[104, "c.hp_less_s", false]], "hp_lessequal_p\uff08c function\uff09": [[105, "c.hp_lessequal_p", false]], "hp_lessequal_s\uff08c function\uff09": [[105, "c.hp_lessequal_s", false]], "hp_log1p_p\uff08c function\uff09": [[108, "c.hp_log1p_p", false]], "hp_log1p_s\uff08c function\uff09": [[108, "c.hp_log1p_s", false]], "hp_log_grad_p\uff08c function\uff09": [[109, "c.hp_log_grad_p", false]], "hp_log_grad_s\uff08c function\uff09": [[109, "c.hp_log_grad_s", false]], "hp_log_p\uff08c function\uff09": [[107, "c.hp_log_p", false]], "hp_log_s\uff08c function\uff09": [[107, "c.hp_log_s", false]], "hp_logical_not_p\uff08c function\uff09": [[110, "c.hp_logical_not_p", false]], "hp_logical_not_s\uff08c function\uff09": [[110, "c.hp_logical_not_s", false]], "hp_logical_or_p\uff08c function\uff09": [[111, "c.hp_logical_or_p", false]], "hp_logical_or_s\uff08c function\uff09": [[111, "c.hp_logical_or_s", false]], "hp_logsoftmax_p\uff08c function\uff09": [[113, "c.hp_logsoftmax_p", false]], "hp_logsoftmax_s\uff08c function\uff09": [[113, "c.hp_logsoftmax_s", false]], "hp_lpnorm_p\uff08c function\uff09": [[114, "c.hp_lpnorm_p", false]], "hp_lpnorm_s\uff08c function\uff09": [[114, "c.hp_lpnorm_s", false]], "hp_lrelu_p\uff08c function\uff09": [[12, "c.hp_lrelu_p", false]], "hp_lrelu_s\uff08c function\uff09": [[12, "c.hp_lrelu_s", false]], "hp_lrn_p\uff08c function\uff09": [[115, "c.hp_lrn_p", false]], "hp_lrn_s\uff08c function\uff09": [[115, "c.hp_lrn_s", false]], "hp_lsh_projection_p\uff08c function\uff09": [[116, "c.hp_lsh_projection_p", false]], "hp_lsh_projection_s\uff08c function\uff09": [[116, "c.hp_lsh_projection_s", false]], "hp_lstm_p\uff08c function\uff09": [[117, "c.hp_Lstm_p", false]], "hp_lstm_s\uff08c function\uff09": [[117, "c.hp_Lstm_s", false]], "hp_lstmgrad_p\uff08c function\uff09": [[118, "c.hp_lstmgrad_p", false]], "hp_lstmgrad_s\uff08c function\uff09": [[118, "c.hp_lstmgrad_s", false]], "hp_lstmgraddata_p\uff08c function\uff09": [[119, "c.hp_lstmgraddata_p", false]], "hp_lstmgraddata_s\uff08c function\uff09": [[119, "c.hp_lstmgraddata_s", false]], "hp_lstmgradweight_p\uff08c function\uff09": [[120, "c.hp_lstmgradweight_p", false]], "hp_lstmgradweight_s\uff08c function\uff09": [[120, "c.hp_lstmgradweight_s", false]], "hp_maximum_p\uff08c function\uff09": [[122, "c.hp_maximum_p", false]], "hp_maximum_s\uff08c function\uff09": [[122, "c.hp_maximum_s", false]], "hp_maximumgrad_p\uff08c function\uff09": [[123, "c.hp_maximumgrad_p", false]], "hp_maximumgrad_s\uff08c function\uff09": [[123, "c.hp_maximumgrad_s", false]], "hp_maxpool_fusion_p\uff08c function\uff09": [[124, "c.hp_maxpool_fusion_p", false]], "hp_maxpool_fusion_s\uff08c function\uff09": [[124, "c.hp_maxpool_fusion_s", false]], "hp_maxpool_grad_p\uff08c function\uff09": [[125, "c.hp_maxpool_grad_p", false]], "hp_maxpool_grad_s\uff08c function\uff09": [[125, "c.hp_maxpool_grad_s", false]], "hp_mfcc_p\uff08c function\uff09": [[126, "c.hp_mfcc_p", false]], "hp_mfcc_s\uff08c function\uff09": [[126, "c.hp_mfcc_s", false]], "hp_minimum_p\uff08c function\uff09": [[127, "c.hp_minimum_p", false]], "hp_minimum_s\uff08c function\uff09": [[127, "c.hp_minimum_s", false]], "hp_minimumgrad_p\uff08c function\uff09": [[128, "c.hp_minimumgrad_p", false]], "hp_minimumgrad_s\uff08c function\uff09": [[128, "c.hp_minimumgrad_s", false]], "hp_mod_p\uff08c function\uff09": [[129, "c.hp_mod_p", false]], "hp_mod_s\uff08c function\uff09": [[129, "c.hp_mod_s", false]], "hp_mul_p\uff08c function\uff09": [[130, "c.hp_mul_p", false]], "hp_mul_s\uff08c function\uff09": [[130, "c.hp_mul_s", false]], "hp_neg_grad_p\uff08c function\uff09": [[133, "c.hp_neg_grad_p", false]], "hp_neg_grad_s\uff08c function\uff09": [[133, "c.hp_neg_grad_s", false]], "hp_neg_p\uff08c function\uff09": [[132, "c.hp_neg_p", false]], "hp_neg_s\uff08c function\uff09": [[132, "c.hp_neg_s", false]], "hp_nllloss_p\uff08c function\uff09": [[134, "c.hp_nllloss_p", false]], "hp_nllloss_s\uff08c function\uff09": [[134, "c.hp_nllloss_s", false]], "hp_nlllossgrad_p\uff08c function\uff09": [[135, "c.hp_nlllossgrad_p", false]], "hp_nlllossgrad_s\uff08c function\uff09": [[135, "c.hp_nlllossgrad_s", false]], "hp_non_max_suppression_p\uff08c function\uff09": [[136, "c.hp_non_max_suppression_p", false]], "hp_non_max_suppression_s\uff08c function\uff09": [[136, "c.hp_non_max_suppression_s", false]], "hp_not_equal_p\uff08c function\uff09": [[138, "c.hp_not_equal_p", false]], "hp_not_equal_s\uff08c function\uff09": [[138, "c.hp_not_equal_s", false]], "hp_onehot_p\uff08c function\uff09": [[139, "c.hp_onehot_p", false]], "hp_onehot_s\uff08c function\uff09": [[139, "c.hp_onehot_s", false]], "hp_ones_like_p\uff08c function\uff09": [[140, "c.hp_ones_like_p", false]], "hp_ones_like_s\uff08c function\uff09": [[140, "c.hp_ones_like_s", false]], "hp_padfusion_p\uff08c function\uff09": [[141, "c.hp_padfusion_p", false]], "hp_padfusion_s\uff08c function\uff09": [[141, "c.hp_padfusion_s", false]], "hp_pow_fusion_p\uff08c function\uff09": [[142, "c.hp_pow_fusion_p", false]], "hp_pow_fusion_s\uff08c function\uff09": [[142, "c.hp_pow_fusion_s", false]], "hp_power_grad_p\uff08c function\uff09": [[143, "c.hp_power_grad_p", false]], "hp_power_grad_s\uff08c function\uff09": [[143, "c.hp_power_grad_s", false]], "hp_prelufusion_p\uff08c function\uff09": [[144, "c.hp_prelufusion_p", false]], "hp_prelufusion_s\uff08c function\uff09": [[144, "c.hp_prelufusion_s", false]], "hp_priorbox_p\uff08c function\uff09": [[145, "c.hp_priorbox_p", false]], "hp_priorbox_s\uff08c function\uff09": [[145, "c.hp_priorbox_s", false]], "hp_quantdata_p\uff08c function\uff09": [[66, "c.hp_QuantData_p", false]], "hp_quantdata_s\uff08c function\uff09": [[66, "c.hp_QuantData_s", false]], "hp_random_normal_p\uff08c function\uff09": [[148, "c.hp_random_normal_p", false]], "hp_random_normal_s\uff08c function\uff09": [[148, "c.hp_random_normal_s", false]], "hp_random_standard_normal_p\uff08c function\uff09": [[149, "c.hp_random_standard_normal_p", false]], "hp_random_standard_normal_s\uff08c function\uff09": [[149, "c.hp_random_standard_normal_s", false]], "hp_real_div_p\uff08c function\uff09": [[152, "c.hp_real_div_p", false]], "hp_real_div_s\uff08c function\uff09": [[152, "c.hp_real_div_s", false]], "hp_reciprocal_p\uff08c function\uff09": [[153, "c.hp_reciprocal_p", false]], "hp_reciprocal_s\uff08c function\uff09": [[153, "c.hp_reciprocal_s", false]], "hp_reduce_p\uff08c function\uff09": [[154, "c.hp_reduce_p", false]], "hp_reduce_s\uff08c function\uff09": [[154, "c.hp_reduce_s", false]], "hp_reduceall_p\uff08c function\uff09": [[21, "c.hp_reduceall_p", false]], "hp_reduceall_s\uff08c function\uff09": [[21, "c.hp_reduceall_s", false]], "hp_reducescatter_p\uff08c function\uff09": [[155, "c.hp_reducescatter_p", false]], "hp_reducescatter_s\uff08c function\uff09": [[155, "c.hp_reducescatter_s", false]], "hp_relu6_p\uff08c function\uff09": [[12, "c.hp_relu6_p", false]], "hp_relu6_s\uff08c function\uff09": [[12, "c.hp_relu6_s", false]], "hp_relu_grad_p\uff08c function\uff09": [[13, "c.hp_relu_grad_p", false]], "hp_relu_grad_s\uff08c function\uff09": [[13, "c.hp_relu_grad_s", false]], "hp_relu_p\uff08c function\uff09": [[12, "c.hp_relu_p", false]], "hp_relu_s\uff08c function\uff09": [[12, "c.hp_relu_s", false]], "hp_reshape_p\uff08c function\uff09": [[156, "c.hp_reshape_p", false]], "hp_reshape_s\uff08c function\uff09": [[156, "c.hp_reshape_s", false]], "hp_resize_anycore\uff08c function\uff09": [[157, "c.hp_resize_anycore", false]], "hp_resizebilineargrad_p\uff08c function\uff09": [[158, "c.hp_resizebilineargrad_p", false]], "hp_resizebilineargrad_s\uff08c function\uff09": [[158, "c.hp_resizebilineargrad_s", false]], "hp_resizenearestneighborgrad_p\uff08c function\uff09": [[158, "c.hp_resizenearestneighborgrad_p", false]], "hp_resizenearestneighborgrad_s\uff08c function\uff09": [[158, "c.hp_resizenearestneighborgrad_s", false]], "hp_roipooling_p\uff08c function\uff09": [[162, "c.hp_roipooling_p", false]], "hp_roipooling_s\uff08c function\uff09": [[162, "c.hp_roipooling_s", false]], "hp_round_p\uff08c function\uff09": [[163, "c.hp_round_p", false]], "hp_round_s\uff08c function\uff09": [[163, "c.hp_round_s", false]], "hp_rsqrt_p\uff08c function\uff09": [[164, "c.hp_rsqrt_p", false]], "hp_rsqrt_s\uff08c function\uff09": [[164, "c.hp_rsqrt_s", false]], "hp_rsqrtgrad_p\uff08c function\uff09": [[165, "c.hp_rsqrtgrad_p", false]], "hp_rsqrtgrad_s\uff08c function\uff09": [[165, "c.hp_rsqrtgrad_s", false]], "hp_scalefusion_p\uff08c function\uff09": [[166, "c.hp_scalefusion_p", false]], "hp_scalefusion_s\uff08c function\uff09": [[166, "c.hp_scalefusion_s", false]], "hp_scatter_elements_p\uff08c function\uff09": [[167, "c.hp_scatter_elements_p", false]], "hp_scatter_elements_s\uff08c function\uff09": [[167, "c.hp_scatter_elements_s", false]], "hp_scatter_nd_p\uff08c function\uff09": [[168, "c.hp_scatter_nd_p", false]], "hp_scatter_nd_s\uff08c function\uff09": [[168, "c.hp_scatter_nd_s", false]], "hp_scatter_nd_update_p\uff08c function\uff09": [[169, "c.hp_scatter_nd_update_p", false]], "hp_scatter_nd_update_s\uff08c function\uff09": [[169, "c.hp_scatter_nd_update_s", false]], "hp_select_p\uff08c function\uff09": [[170, "c.hp_select_p", false]], "hp_select_s\uff08c function\uff09": [[170, "c.hp_select_s", false]], "hp_sgd_p\uff08c function\uff09": [[171, "c.hp_sgd_p", false]], "hp_sgd_s\uff08c function\uff09": [[171, "c.hp_sgd_s", false]], "hp_sigmoid_p\uff08c function\uff09": [[12, "c.hp_sigmoid_p", false]], "hp_sigmoid_s\uff08c function\uff09": [[12, "c.hp_sigmoid_s", false]], "hp_sigmoidcrossentropywithlogits_p\uff08c function\uff09": [[174, "c.hp_sigmoidcrossentropywithlogits_p", false]], "hp_sigmoidcrossentropywithlogits_s\uff08c function\uff09": [[174, "c.hp_sigmoidcrossentropywithlogits_s", false]], "hp_sigmoidcrossentropywithlogitsgrad_p\uff08c function\uff09": [[173, "c.hp_sigmoidcrossentropywithlogitsgrad_p", false]], "hp_sigmoidcrossentropywithlogitsgrad_s\uff08c function\uff09": [[173, "c.hp_sigmoidcrossentropywithlogitsgrad_s", false]], "hp_sin_p\uff08c function\uff09": [[175, "c.hp_sin_p", false]], "hp_sin_s\uff08c function\uff09": [[175, "c.hp_sin_s", false]], "hp_slice_p\uff08c function\uff09": [[178, "c.hp_slice_p", false]], "hp_slice_s\uff08c function\uff09": [[178, "c.hp_slice_s", false]], "hp_smoothl1loss_p\uff08c function\uff09": [[179, "c.hp_smoothl1loss_p", false]], "hp_smoothl1loss_s\uff08c function\uff09": [[179, "c.hp_smoothl1loss_s", false]], "hp_smoothl1lossgrad_p\uff08c function\uff09": [[180, "c.hp_smoothl1lossgrad_p", false]], "hp_smoothl1lossgrad_s\uff08c function\uff09": [[180, "c.hp_smoothl1lossgrad_s", false]], "hp_softmax_cross_entropy_with_logits_p\uff08c function\uff09": [[182, "c.hp_softmax_cross_entropy_with_logits_p", false]], "hp_softmax_cross_entropy_with_logits_s\uff08c function\uff09": [[182, "c.hp_softmax_cross_entropy_with_logits_s", false]], "hp_softmax_p\uff08c function\uff09": [[181, "c.hp_softmax_p", false]], "hp_softmax_s\uff08c function\uff09": [[181, "c.hp_softmax_s", false]], "hp_softplus_p\uff08c function\uff09": [[12, "c.hp_softplus_p", false]], "hp_softplus_s\uff08c function\uff09": [[12, "c.hp_softplus_s", false]], "hp_softshrink_p\uff08c function\uff09": [[12, "c.hp_softshrink_p", false]], "hp_softshrink_s\uff08c function\uff09": [[12, "c.hp_softshrink_s", false]], "hp_softsignopt_p\uff08c function\uff09": [[12, "c.hp_softsignopt_p", false]], "hp_softsignopt_s\uff08c function\uff09": [[12, "c.hp_softsignopt_s", false]], "hp_spacetobatch_p\uff08c function\uff09": [[183, "c.hp_spacetobatch_p", false]], "hp_spacetobatch_s\uff08c function\uff09": [[183, "c.hp_spacetobatch_s", false]], "hp_spacetobatchnd_p\uff08c function\uff09": [[184, "c.hp_spacetobatchnd_p", false]], "hp_spacetobatchnd_s\uff08c function\uff09": [[184, "c.hp_spacetobatchnd_s", false]], "hp_spacetodepth_p\uff08c function\uff09": [[185, "c.hp_spacetodepth_p", false]], "hp_spacetodepth_s\uff08c function\uff09": [[185, "c.hp_spacetodepth_s", false]], "hp_sparse_softmax_cross_entropy_with_logits_p\uff08c function\uff09": [[186, "c.hp_sparse_softmax_cross_entropy_with_logits_p", false]], "hp_sparse_softmax_cross_entropy_with_logits_s\uff08c function\uff09": [[186, "c.hp_sparse_softmax_cross_entropy_with_logits_s", false]], "hp_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.hp_sparsefillemptyrows_p", false]], "hp_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.hp_sparsefillemptyrows_s", false]], "hp_sparsesegmentsum_p\uff08c function\uff09": [[189, "c.hp_sparsesegmentsum_p", false]], "hp_sparsesegmentsum_s\uff08c function\uff09": [[189, "c.hp_sparsesegmentsum_s", false]], "hp_sparsetodense_p\uff08c function\uff09": [[190, "c.hp_sparsetodense_p", false]], "hp_sparsetodense_s\uff08c function\uff09": [[190, "c.hp_sparsetodense_s", false]], "hp_splice_p\uff08c function\uff09": [[191, "c.hp_splice_p", false]], "hp_splice_s\uff08c function\uff09": [[191, "c.hp_splice_s", false]], "hp_split_p\uff08c function\uff09": [[192, "c.hp_split_p", false]], "hp_split_s\uff08c function\uff09": [[192, "c.hp_split_s", false]], "hp_split_with_overlap_p\uff08c function\uff09": [[193, "c.hp_split_with_overlap_p", false]], "hp_split_with_overlap_s\uff08c function\uff09": [[193, "c.hp_split_with_overlap_s", false]], "hp_sqrt_p\uff08c function\uff09": [[194, "c.hp_sqrt_p", false]], "hp_sqrt_s\uff08c function\uff09": [[194, "c.hp_sqrt_s", false]], "hp_sqrtgrad_p\uff08c function\uff09": [[195, "c.hp_sqrtgrad_p", false]], "hp_sqrtgrad_s\uff08c function\uff09": [[195, "c.hp_sqrtgrad_s", false]], "hp_square_p\uff08c function\uff09": [[196, "c.hp_square_p", false]], "hp_square_s\uff08c function\uff09": [[196, "c.hp_square_s", false]], "hp_squaredifference_p\uff08c function\uff09": [[197, "c.hp_squaredifference_p", false]], "hp_squaredifference_s\uff08c function\uff09": [[197, "c.hp_squaredifference_s", false]], "hp_stack_p\uff08c function\uff09": [[199, "c.hp_stack_p", false]], "hp_stack_s\uff08c function\uff09": [[199, "c.hp_stack_s", false]], "hp_stridedslicegrad_p\uff08c function\uff09": [[201, "c.hp_stridedslicegrad_p", false]], "hp_stridedslicegrad_s\uff08c function\uff09": [[201, "c.hp_stridedslicegrad_s", false]], "hp_subext_p\uff08c function\uff09": [[202, "c.hp_subext_p", false]], "hp_subext_s\uff08c function\uff09": [[202, "c.hp_subext_s", false]], "hp_subgrad_p\uff08c function\uff09": [[203, "c.hp_subgrad_p", false]], "hp_subgrad_s\uff08c function\uff09": [[203, "c.hp_subgrad_s", false]], "hp_subrelu6_p\uff08c function\uff09": [[202, "c.hp_subrelu6_p", false]], "hp_subrelu6_s\uff08c function\uff09": [[202, "c.hp_subrelu6_s", false]], "hp_subrelu_p\uff08c function\uff09": [[202, "c.hp_subrelu_p", false]], "hp_subrelu_s\uff08c function\uff09": [[202, "c.hp_subrelu_s", false]], "hp_swish_p\uff08c function\uff09": [[12, "c.hp_swish_p", false]], "hp_swish_s\uff08c function\uff09": [[12, "c.hp_swish_s", false]], "hp_tanh_p\uff08c function\uff09": [[12, "c.hp_tanh_p", false]], "hp_tanh_s\uff08c function\uff09": [[12, "c.hp_tanh_s", false]], "hp_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.hp_tensor_scatter_add_p", false]], "hp_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.hp_tensor_scatter_add_s", false]], "hp_tensorarrayread_p\uff08c function\uff09": [[208, "c.hp_tensorarrayread_p", false]], "hp_tensorarrayread_s\uff08c function\uff09": [[208, "c.hp_tensorarrayread_s", false]], "hp_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.hp_tensorlistfromtensor_p", false]], "hp_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.hp_tensorlistfromtensor_s", false]], "hp_tile_p\uff08c function\uff09": [[215, "c.hp_tile_p", false]], "hp_tile_s\uff08c function\uff09": [[215, "c.hp_tile_s", false]], "hp_to_i8_quant_p\uff08c function\uff09": [[146, "c.hp_to_i8_quant_p", false]], "hp_to_i8_quant_s\uff08c function\uff09": [[146, "c.hp_to_i8_quant_s", false]], "hp_topk_fusion_p\uff08c function\uff09": [[216, "c.hp_topk_fusion_p", false]], "hp_topk_fusion_s\uff08c function\uff09": [[216, "c.hp_topk_fusion_s", false]], "hp_transpose_p\uff08c function\uff09": [[217, "c.hp_transpose_p", false]], "hp_transpose_s\uff08c function\uff09": [[217, "c.hp_transpose_s", false]], "hp_tril_p\uff08c function\uff09": [[218, "c.hp_tril_p", false]], "hp_tril_s\uff08c function\uff09": [[218, "c.hp_tril_s", false]], "hp_triu_p\uff08c function\uff09": [[219, "c.hp_triu_p", false]], "hp_triu_s\uff08c function\uff09": [[219, "c.hp_triu_s", false]], "hp_uniform_real_p\uff08c function\uff09": [[220, "c.hp_uniform_real_p", false]], "hp_uniform_real_s\uff08c function\uff09": [[220, "c.hp_uniform_real_s", false]], "hp_unique_p\uff08c function\uff09": [[221, "c.hp_Unique_p", false]], "hp_unique_s\uff08c function\uff09": [[221, "c.hp_Unique_s", false]], "hp_where_p\uff08c function\uff09": [[225, "c.hp_where_p", false]], "hp_where_s\uff08c function\uff09": [[225, "c.hp_where_s", false]], "hp_zerolike_p\uff08c function\uff09": [[226, "c.hp_zerolike_p", false]], "hp_zerolike_s\uff08c function\uff09": [[226, "c.hp_zerolike_s", false]], "i16_abs_p\uff08c function\uff09": [[10, "c.i16_abs_p", false]], "i16_abs_s\uff08c function\uff09": [[10, "c.i16_abs_s", false]], "i16_addext_p\uff08c function\uff09": [[17, "c.i16_addext_p", false]], "i16_addext_s\uff08c function\uff09": [[17, "c.i16_addext_s", false]], "i16_addn_p\uff08c function\uff09": [[19, "c.i16_addn_p", false]], "i16_addn_s\uff08c function\uff09": [[19, "c.i16_addn_s", false]], "i16_addrelu6_p\uff08c function\uff09": [[17, "c.i16_addrelu6_p", false]], "i16_addrelu6_s\uff08c function\uff09": [[17, "c.i16_addrelu6_s", false]], "i16_addrelu_p\uff08c function\uff09": [[17, "c.i16_addrelu_p", false]], "i16_addrelu_s\uff08c function\uff09": [[17, "c.i16_addrelu_s", false]], "i16_allgather_p\uff08c function\uff09": [[22, "c.i16_allgather_p", false]], "i16_allgather_s\uff08c function\uff09": [[22, "c.i16_allgather_s", false]], "i16_and_p\uff08c function\uff09": [[112, "c.i16_and_p", false]], "i16_and_s\uff08c function\uff09": [[112, "c.i16_and_s", false]], "i16_assign_p\uff08c function\uff09": [[27, "c.i16_assign_p", false]], "i16_assign_s\uff08c function\uff09": [[27, "c.i16_assign_s", false]], "i16_assignadd_p\uff08c function\uff09": [[28, "c.i16_assignadd_p", false]], "i16_assignadd_s\uff08c function\uff09": [[28, "c.i16_assignadd_s", false]], "i16_batchtospace_p\uff08c function\uff09": [[35, "c.i16_batchtospace_p", false]], "i16_batchtospace_s\uff08c function\uff09": [[35, "c.i16_batchtospace_s", false]], "i16_batchtospacend_p\uff08c function\uff09": [[36, "c.i16_batchtospacend_p", false]], "i16_batchtospacend_s\uff08c function\uff09": [[36, "c.i16_batchtospacend_s", false]], "i16_biasadd_p\uff08c function\uff09": [[37, "c.i16_biasadd_p", false]], "i16_biasadd_s\uff08c function\uff09": [[37, "c.i16_biasadd_s", false]], "i16_broadcastto_p\uff08c function\uff09": [[41, "c.i16_broadcastto_p", false]], "i16_broadcastto_s\uff08c function\uff09": [[41, "c.i16_broadcastto_s", false]], "i16_clip_p\uff08c function\uff09": [[44, "c.i16_clip_p", false]], "i16_clip_s\uff08c function\uff09": [[44, "c.i16_clip_s", false]], "i16_concat_p\uff08c function\uff09": [[45, "c.i16_concat_p", false]], "i16_concat_s\uff08c function\uff09": [[45, "c.i16_concat_s", false]], "i16_constant_of_shape_p\uff08c function\uff09": [[46, "c.i16_constant_of_shape_p", false]], "i16_constant_of_shape_s\uff08c function\uff09": [[46, "c.i16_constant_of_shape_s", false]], "i16_cos_p\uff08c function\uff09": [[51, "c.i16_cos_p", false]], "i16_cos_s\uff08c function\uff09": [[51, "c.i16_cos_s", false]], "i16_cumsum_p\uff08c function\uff09": [[54, "c.i16_cumsum_p", false]], "i16_cumsum_s\uff08c function\uff09": [[54, "c.i16_cumsum_s", false]], "i16_depthtospace_p\uff08c function\uff09": [[59, "c.i16_depthtospace_p", false]], "i16_depthtospace_s\uff08c function\uff09": [[59, "c.i16_depthtospace_s", false]], "i16_div_fusion_p\uff08c function\uff09": [[61, "c.i16_div_fusion_p", false]], "i16_div_fusion_s\uff08c function\uff09": [[61, "c.i16_div_fusion_s", false]], "i16_eltwise_p\uff08c function\uff09": [[67, "c.i16_eltwise_p", false]], "i16_eltwise_s\uff08c function\uff09": [[67, "c.i16_eltwise_s", false]], "i16_equal_p\uff08c function\uff09": [[70, "c.i16_equal_p", false]], "i16_equal_s\uff08c function\uff09": [[70, "c.i16_equal_s", false]], "i16_expfusion_p\uff08c function\uff09": [[73, "c.i16_expfusion_p", false]], "i16_expfusion_s\uff08c function\uff09": [[73, "c.i16_expfusion_s", false]], "i16_extract_features_p\uff08c function\uff09": [[55, "c.i16_extract_features_p", false]], "i16_extract_features_s\uff08c function\uff09": [[55, "c.i16_extract_features_s", false]], "i16_fill_p\uff08c function\uff09": [[78, "c.i16_fill_p", false]], "i16_fill_s\uff08c function\uff09": [[78, "c.i16_fill_s", false]], "i16_formattranspose_p\uff08c function\uff09": [[85, "c.i16_formattranspose_p", false]], "i16_formattranspose_s\uff08c function\uff09": [[85, "c.i16_formattranspose_s", false]], "i16_gather_nd_p\uff08c function\uff09": [[89, "c.i16_gather_nd_p", false]], "i16_gather_nd_s\uff08c function\uff09": [[89, "c.i16_gather_nd_s", false]], "i16_gather_p\uff08c function\uff09": [[88, "c.i16_gather_p", false]], "i16_gather_s\uff08c function\uff09": [[88, "c.i16_gather_s", false]], "i16_gatherd_p\uff08c function\uff09": [[90, "c.i16_gatherd_p", false]], "i16_gatherd_s\uff08c function\uff09": [[90, "c.i16_gatherd_s", false]], "i16_greater_p\uff08c function\uff09": [[92, "c.i16_greater_p", false]], "i16_greater_s\uff08c function\uff09": [[92, "c.i16_greater_s", false]], "i16_greaterequal_p\uff08c function\uff09": [[93, "c.i16_greaterequal_p", false]], "i16_greaterequal_s\uff08c function\uff09": [[93, "c.i16_greaterequal_s", false]], "i16_invertpermutation_p\uff08c function\uff09": [[98, "c.i16_invertpermutation_p", false]], "i16_invertpermutation_s\uff08c function\uff09": [[98, "c.i16_invertpermutation_s", false]], "i16_isfinite_p\uff08c function\uff09": [[99, "c.i16_isfinite_p", false]], "i16_isfinite_s\uff08c function\uff09": [[99, "c.i16_isfinite_s", false]], "i16_less_p\uff08c function\uff09": [[104, "c.i16_less_p", false]], "i16_less_s\uff08c function\uff09": [[104, "c.i16_less_s", false]], "i16_lessequal_p\uff08c function\uff09": [[105, "c.i16_lessequal_p", false]], "i16_lessequal_s\uff08c function\uff09": [[105, "c.i16_lessequal_s", false]], "i16_log1p_p\uff08c function\uff09": [[108, "c.i16_log1p_p", false]], "i16_log1p_s\uff08c function\uff09": [[108, "c.i16_log1p_s", false]], "i16_log_p\uff08c function\uff09": [[107, "c.i16_log_p", false]], "i16_log_s\uff08c function\uff09": [[107, "c.i16_log_s", false]], "i16_logical_not_p\uff08c function\uff09": [[110, "c.i16_logical_not_p", false]], "i16_logical_not_s\uff08c function\uff09": [[110, "c.i16_logical_not_s", false]], "i16_logical_or_p\uff08c function\uff09": [[111, "c.i16_logical_or_p", false]], "i16_logical_or_s\uff08c function\uff09": [[111, "c.i16_logical_or_s", false]], "i16_lsh_projection_p\uff08c function\uff09": [[116, "c.i16_lsh_projection_p", false]], "i16_lsh_projection_s\uff08c function\uff09": [[116, "c.i16_lsh_projection_s", false]], "i16_matmulfusion_p\uff08c function\uff09": [[121, "c.i16_matmulfusion_p", false]], "i16_matmulfusion_s\uff08c function\uff09": [[121, "c.i16_matmulfusion_s", false]], "i16_maximum_p\uff08c function\uff09": [[122, "c.i16_maximum_p", false]], "i16_maximum_s\uff08c function\uff09": [[122, "c.i16_maximum_s", false]], "i16_minimum_p\uff08c function\uff09": [[127, "c.i16_minimum_p", false]], "i16_minimum_s\uff08c function\uff09": [[127, "c.i16_minimum_s", false]], "i16_mod_p\uff08c function\uff09": [[129, "c.i16_mod_p", false]], "i16_mod_s\uff08c function\uff09": [[129, "c.i16_mod_s", false]], "i16_mul_p\uff08c function\uff09": [[130, "c.i16_mul_p", false]], "i16_mul_s\uff08c function\uff09": [[130, "c.i16_mul_s", false]], "i16_neg_grad_p\uff08c function\uff09": [[133, "c.i16_neg_grad_p", false]], "i16_neg_grad_s\uff08c function\uff09": [[133, "c.i16_neg_grad_s", false]], "i16_neg_p\uff08c function\uff09": [[132, "c.i16_neg_p", false]], "i16_neg_s\uff08c function\uff09": [[132, "c.i16_neg_s", false]], "i16_nonzero_p\uff08c function\uff09": [[137, "c.i16_nonzero_p", false]], "i16_nonzero_s\uff08c function\uff09": [[137, "c.i16_nonzero_s", false]], "i16_not_equal_p\uff08c function\uff09": [[138, "c.i16_not_equal_p", false]], "i16_not_equal_s\uff08c function\uff09": [[138, "c.i16_not_equal_s", false]], "i16_onehot_p\uff08c function\uff09": [[139, "c.i16_onehot_p", false]], "i16_onehot_s\uff08c function\uff09": [[139, "c.i16_onehot_s", false]], "i16_ones_like_p\uff08c function\uff09": [[140, "c.i16_ones_like_p", false]], "i16_ones_like_s\uff08c function\uff09": [[140, "c.i16_ones_like_s", false]], "i16_padfusion_p\uff08c function\uff09": [[141, "c.i16_padfusion_p", false]], "i16_padfusion_s\uff08c function\uff09": [[141, "c.i16_padfusion_s", false]], "i16_pow_fusion_p\uff08c function\uff09": [[142, "c.i16_pow_fusion_p", false]], "i16_pow_fusion_s\uff08c function\uff09": [[142, "c.i16_pow_fusion_s", false]], "i16_raggedrange_p\uff08c function\uff09": [[147, "c.i16_raggedrange_p", false]], "i16_raggedrange_s\uff08c function\uff09": [[147, "c.i16_raggedrange_s", false]], "i16_range_p\uff08c function\uff09": [[150, "c.i16_range_p", false]], "i16_range_s\uff08c function\uff09": [[150, "c.i16_range_s", false]], "i16_real_div_p\uff08c function\uff09": [[152, "c.i16_real_div_p", false]], "i16_real_div_s\uff08c function\uff09": [[152, "c.i16_real_div_s", false]], "i16_reciprocal_p\uff08c function\uff09": [[153, "c.i16_reciprocal_p", false]], "i16_reciprocal_s\uff08c function\uff09": [[153, "c.i16_reciprocal_s", false]], "i16_reduce_p\uff08c function\uff09": [[154, "c.i16_reduce_p", false]], "i16_reduce_s\uff08c function\uff09": [[154, "c.i16_reduce_s", false]], "i16_reduceall_p\uff08c function\uff09": [[21, "c.i16_reduceall_p", false]], "i16_reduceall_s\uff08c function\uff09": [[21, "c.i16_reduceall_s", false]], "i16_reducescatter_p\uff08c function\uff09": [[155, "c.i16_reducescatter_p", false]], "i16_reducescatter_s\uff08c function\uff09": [[155, "c.i16_reducescatter_s", false]], "i16_reshape_p\uff08c function\uff09": [[156, "c.i16_reshape_p", false]], "i16_reshape_s\uff08c function\uff09": [[156, "c.i16_reshape_s", false]], "i16_rsqrt_p\uff08c function\uff09": [[164, "c.i16_rsqrt_p", false]], "i16_rsqrt_s\uff08c function\uff09": [[164, "c.i16_rsqrt_s", false]], "i16_scalefusion_p\uff08c function\uff09": [[166, "c.i16_scalefusion_p", false]], "i16_scalefusion_s\uff08c function\uff09": [[166, "c.i16_scalefusion_s", false]], "i16_scatter_elements_p\uff08c function\uff09": [[167, "c.i16_scatter_elements_p", false]], "i16_scatter_elements_s\uff08c function\uff09": [[167, "c.i16_scatter_elements_s", false]], "i16_scatter_nd_p\uff08c function\uff09": [[168, "c.i16_scatter_nd_p", false]], "i16_scatter_nd_s\uff08c function\uff09": [[168, "c.i16_scatter_nd_s", false]], "i16_scatter_nd_update_p\uff08c function\uff09": [[169, "c.i16_scatter_nd_update_p", false]], "i16_scatter_nd_update_s\uff08c function\uff09": [[169, "c.i16_scatter_nd_update_s", false]], "i16_select_p\uff08c function\uff09": [[170, "c.i16_select_p", false]], "i16_select_s\uff08c function\uff09": [[170, "c.i16_select_s", false]], "i16_sin_p\uff08c function\uff09": [[175, "c.i16_sin_p", false]], "i16_sin_s\uff08c function\uff09": [[175, "c.i16_sin_s", false]], "i16_slice_p\uff08c function\uff09": [[178, "c.i16_slice_p", false]], "i16_slice_s\uff08c function\uff09": [[178, "c.i16_slice_s", false]], "i16_spacetobatch_p\uff08c function\uff09": [[183, "c.i16_spacetobatch_p", false]], "i16_spacetobatch_s\uff08c function\uff09": [[183, "c.i16_spacetobatch_s", false]], "i16_spacetobatchnd_p\uff08c function\uff09": [[184, "c.i16_spacetobatchnd_p", false]], "i16_spacetobatchnd_s\uff08c function\uff09": [[184, "c.i16_spacetobatchnd_s", false]], "i16_spacetodepth_p\uff08c function\uff09": [[185, "c.i16_spacetodepth_p", false]], "i16_spacetodepth_s\uff08c function\uff09": [[185, "c.i16_spacetodepth_s", false]], "i16_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.i16_sparsefillemptyrows_p", false]], "i16_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.i16_sparsefillemptyrows_s", false]], "i16_sparsesegmentsum_p\uff08c function\uff09": [[189, "c.i16_sparsesegmentsum_p", false]], "i16_sparsesegmentsum_s\uff08c function\uff09": [[189, "c.i16_sparsesegmentsum_s", false]], "i16_sparsetodense_p\uff08c function\uff09": [[190, "c.i16_sparsetodense_p", false]], "i16_sparsetodense_s\uff08c function\uff09": [[190, "c.i16_sparsetodense_s", false]], "i16_splice_p\uff08c function\uff09": [[191, "c.i16_splice_p", false]], "i16_splice_s\uff08c function\uff09": [[191, "c.i16_splice_s", false]], "i16_split_p\uff08c function\uff09": [[192, "c.i16_split_p", false]], "i16_split_s\uff08c function\uff09": [[192, "c.i16_split_s", false]], "i16_split_with_overlap_p\uff08c function\uff09": [[193, "c.i16_split_with_overlap_p", false]], "i16_split_with_overlap_s\uff08c function\uff09": [[193, "c.i16_split_with_overlap_s", false]], "i16_sqrt_p\uff08c function\uff09": [[194, "c.i16_sqrt_p", false]], "i16_sqrt_s\uff08c function\uff09": [[194, "c.i16_sqrt_s", false]], "i16_sqrtgrad_p\uff08c function\uff09": [[195, "c.i16_sqrtgrad_p", false]], "i16_sqrtgrad_s\uff08c function\uff09": [[195, "c.i16_sqrtgrad_s", false]], "i16_square_p\uff08c function\uff09": [[196, "c.i16_square_p", false]], "i16_square_s\uff08c function\uff09": [[196, "c.i16_square_s", false]], "i16_squaredifference_p\uff08c function\uff09": [[197, "c.i16_squaredifference_p", false]], "i16_squaredifference_s\uff08c function\uff09": [[197, "c.i16_squaredifference_s", false]], "i16_stack_p\uff08c function\uff09": [[199, "c.i16_stack_p", false]], "i16_stack_s\uff08c function\uff09": [[199, "c.i16_stack_s", false]], "i16_subrelu6_p\uff08c function\uff09": [[202, "c.i16_subrelu6_p", false]], "i16_subrelu6_s\uff08c function\uff09": [[202, "c.i16_subrelu6_s", false]], "i16_subrelu_p\uff08c function\uff09": [[202, "c.i16_subrelu_p", false]], "i16_subrelu_s\uff08c function\uff09": [[202, "c.i16_subrelu_s", false]], "i16_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.i16_tensor_scatter_add_p", false]], "i16_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.i16_tensor_scatter_add_s", false]], "i16_tensorarrayread_p\uff08c function\uff09": [[208, "c.i16_tensorarrayread_p", false]], "i16_tensorarrayread_s\uff08c function\uff09": [[208, "c.i16_tensorarrayread_s", false]], "i16_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.i16_tensorlistfromtensor_p", false]], "i16_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.i16_tensorlistfromtensor_s", false]], "i16_tile_p\uff08c function\uff09": [[215, "c.i16_tile_p", false]], "i16_tile_s\uff08c function\uff09": [[215, "c.i16_tile_s", false]], "i16_topk_fusion_p\uff08c function\uff09": [[216, "c.i16_topk_fusion_p", false]], "i16_topk_fusion_s\uff08c function\uff09": [[216, "c.i16_topk_fusion_s", false]], "i16_transpose_p\uff08c function\uff09": [[217, "c.i16_transpose_p", false]], "i16_transpose_s\uff08c function\uff09": [[217, "c.i16_transpose_s", false]], "i16_tril_p\uff08c function\uff09": [[218, "c.i16_tril_p", false]], "i16_tril_s\uff08c function\uff09": [[218, "c.i16_tril_s", false]], "i16_triu_p\uff08c function\uff09": [[219, "c.i16_triu_p", false]], "i16_triu_s\uff08c function\uff09": [[219, "c.i16_triu_s", false]], "i16_unique_p\uff08c function\uff09": [[221, "c.i16_Unique_p", false]], "i16_unique_s\uff08c function\uff09": [[221, "c.i16_Unique_s", false]], "i16_unsorted_segment_sum_p\uff08c function\uff09": [[222, "c.i16_unsorted_segment_sum_p", false]], "i16_unsorted_segment_sum_s\uff08c function\uff09": [[222, "c.i16_unsorted_segment_sum_s", false]], "i16_where_p\uff08c function\uff09": [[225, "c.i16_where_p", false]], "i16_where_s\uff08c function\uff09": [[225, "c.i16_where_s", false]], "i16_zerolike_p\uff08c function\uff09": [[226, "c.i16_zerolike_p", false]], "i16_zerolike_s\uff08c function\uff09": [[226, "c.i16_zerolike_s", false]], "i32_abs_p\uff08c function\uff09": [[10, "c.i32_abs_p", false]], "i32_abs_s\uff08c function\uff09": [[10, "c.i32_abs_s", false]], "i32_addext_p\uff08c function\uff09": [[17, "c.i32_addext_p", false]], "i32_addext_s\uff08c function\uff09": [[17, "c.i32_addext_s", false]], "i32_addn_p\uff08c function\uff09": [[19, "c.i32_addn_p", false]], "i32_addn_s\uff08c function\uff09": [[19, "c.i32_addn_s", false]], "i32_addrelu6_p\uff08c function\uff09": [[17, "c.i32_addrelu6_p", false]], "i32_addrelu6_s\uff08c function\uff09": [[17, "c.i32_addrelu6_s", false]], "i32_addrelu_p\uff08c function\uff09": [[17, "c.i32_addrelu_p", false]], "i32_addrelu_s\uff08c function\uff09": [[17, "c.i32_addrelu_s", false]], "i32_allgather_p\uff08c function\uff09": [[22, "c.i32_allgather_p", false]], "i32_allgather_s\uff08c function\uff09": [[22, "c.i32_allgather_s", false]], "i32_and_p\uff08c function\uff09": [[112, "c.i32_and_p", false]], "i32_and_s\uff08c function\uff09": [[112, "c.i32_and_s", false]], "i32_assign_p\uff08c function\uff09": [[27, "c.i32_assign_p", false]], "i32_assign_s\uff08c function\uff09": [[27, "c.i32_assign_s", false]], "i32_assignadd_p\uff08c function\uff09": [[28, "c.i32_assignadd_p", false]], "i32_assignadd_s\uff08c function\uff09": [[28, "c.i32_assignadd_s", false]], "i32_batchtospace_p\uff08c function\uff09": [[35, "c.i32_batchtospace_p", false]], "i32_batchtospace_s\uff08c function\uff09": [[35, "c.i32_batchtospace_s", false]], "i32_batchtospacend_p\uff08c function\uff09": [[36, "c.i32_batchtospacend_p", false]], "i32_batchtospacend_s\uff08c function\uff09": [[36, "c.i32_batchtospacend_s", false]], "i32_biasadd_p\uff08c function\uff09": [[37, "c.i32_biasadd_p", false]], "i32_biasadd_s\uff08c function\uff09": [[37, "c.i32_biasadd_s", false]], "i32_broadcastto_p\uff08c function\uff09": [[41, "c.i32_broadcastto_p", false]], "i32_broadcastto_s\uff08c function\uff09": [[41, "c.i32_broadcastto_s", false]], "i32_clip_p\uff08c function\uff09": [[44, "c.i32_clip_p", false]], "i32_clip_s\uff08c function\uff09": [[44, "c.i32_clip_s", false]], "i32_concat_p\uff08c function\uff09": [[45, "c.i32_concat_p", false]], "i32_concat_s\uff08c function\uff09": [[45, "c.i32_concat_s", false]], "i32_constant_of_shape_p\uff08c function\uff09": [[46, "c.i32_constant_of_shape_p", false]], "i32_constant_of_shape_s\uff08c function\uff09": [[46, "c.i32_constant_of_shape_s", false]], "i32_cos_p\uff08c function\uff09": [[51, "c.i32_cos_p", false]], "i32_cos_s\uff08c function\uff09": [[51, "c.i32_cos_s", false]], "i32_cumsum_p\uff08c function\uff09": [[54, "c.i32_cumsum_p", false]], "i32_cumsum_s\uff08c function\uff09": [[54, "c.i32_cumsum_s", false]], "i32_depthtospace_p\uff08c function\uff09": [[59, "c.i32_depthtospace_p", false]], "i32_depthtospace_s\uff08c function\uff09": [[59, "c.i32_depthtospace_s", false]], "i32_div_fusion_p\uff08c function\uff09": [[61, "c.i32_div_fusion_p", false]], "i32_div_fusion_s\uff08c function\uff09": [[61, "c.i32_div_fusion_s", false]], "i32_eltwise_p\uff08c function\uff09": [[67, "c.i32_eltwise_p", false]], "i32_eltwise_s\uff08c function\uff09": [[67, "c.i32_eltwise_s", false]], "i32_equal_p\uff08c function\uff09": [[70, "c.i32_equal_p", false]], "i32_equal_s\uff08c function\uff09": [[70, "c.i32_equal_s", false]], "i32_expfusion_p\uff08c function\uff09": [[73, "c.i32_expfusion_p", false]], "i32_expfusion_s\uff08c function\uff09": [[73, "c.i32_expfusion_s", false]], "i32_extract_features_p\uff08c function\uff09": [[55, "c.i32_extract_features_p", false]], "i32_extract_features_s\uff08c function\uff09": [[55, "c.i32_extract_features_s", false]], "i32_fill_p\uff08c function\uff09": [[78, "c.i32_fill_p", false]], "i32_fill_s\uff08c function\uff09": [[78, "c.i32_fill_s", false]], "i32_formattranspose_p\uff08c function\uff09": [[85, "c.i32_formattranspose_p", false]], "i32_formattranspose_s\uff08c function\uff09": [[85, "c.i32_formattranspose_s", false]], "i32_gather_nd_p\uff08c function\uff09": [[89, "c.i32_gather_nd_p", false]], "i32_gather_nd_s\uff08c function\uff09": [[89, "c.i32_gather_nd_s", false]], "i32_gather_p\uff08c function\uff09": [[88, "c.i32_gather_p", false]], "i32_gather_s\uff08c function\uff09": [[88, "c.i32_gather_s", false]], "i32_gatherd_p\uff08c function\uff09": [[90, "c.i32_gatherd_p", false]], "i32_gatherd_s\uff08c function\uff09": [[90, "c.i32_gatherd_s", false]], "i32_greater_p\uff08c function\uff09": [[92, "c.i32_greater_p", false]], "i32_greater_s\uff08c function\uff09": [[92, "c.i32_greater_s", false]], "i32_greaterequal_p\uff08c function\uff09": [[93, "c.i32_greaterequal_p", false]], "i32_greaterequal_s\uff08c function\uff09": [[93, "c.i32_greaterequal_s", false]], "i32_hashtablelookup_p\uff08c function\uff09": [[96, "c.i32_hashtablelookup_p", false]], "i32_hashtablelookup_s\uff08c function\uff09": [[96, "c.i32_hashtablelookup_s", false]], "i32_invertpermutation_p\uff08c function\uff09": [[98, "c.i32_invertpermutation_p", false]], "i32_invertpermutation_s\uff08c function\uff09": [[98, "c.i32_invertpermutation_s", false]], "i32_isfinite_p\uff08c function\uff09": [[99, "c.i32_isfinite_p", false]], "i32_isfinite_s\uff08c function\uff09": [[99, "c.i32_isfinite_s", false]], "i32_less_p\uff08c function\uff09": [[104, "c.i32_less_p", false]], "i32_less_s\uff08c function\uff09": [[104, "c.i32_less_s", false]], "i32_lessequal_p\uff08c function\uff09": [[105, "c.i32_lessequal_p", false]], "i32_lessequal_s\uff08c function\uff09": [[105, "c.i32_lessequal_s", false]], "i32_log1p_p\uff08c function\uff09": [[108, "c.i32_log1p_p", false]], "i32_log1p_s\uff08c function\uff09": [[108, "c.i32_log1p_s", false]], "i32_log_p\uff08c function\uff09": [[107, "c.i32_log_p", false]], "i32_log_s\uff08c function\uff09": [[107, "c.i32_log_s", false]], "i32_logical_not_p\uff08c function\uff09": [[110, "c.i32_logical_not_p", false]], "i32_logical_not_s\uff08c function\uff09": [[110, "c.i32_logical_not_s", false]], "i32_logical_or_p\uff08c function\uff09": [[111, "c.i32_logical_or_p", false]], "i32_logical_or_s\uff08c function\uff09": [[111, "c.i32_logical_or_s", false]], "i32_lsh_projection_p\uff08c function\uff09": [[116, "c.i32_lsh_projection_p", false]], "i32_lsh_projection_s\uff08c function\uff09": [[116, "c.i32_lsh_projection_s", false]], "i32_matmulfusion_p\uff08c function\uff09": [[121, "c.i32_matmulfusion_p", false]], "i32_matmulfusion_s\uff08c function\uff09": [[121, "c.i32_matmulfusion_s", false]], "i32_maximum_p\uff08c function\uff09": [[122, "c.i32_maximum_p", false]], "i32_maximum_s\uff08c function\uff09": [[122, "c.i32_maximum_s", false]], "i32_minimum_p\uff08c function\uff09": [[127, "c.i32_minimum_p", false]], "i32_minimum_s\uff08c function\uff09": [[127, "c.i32_minimum_s", false]], "i32_mod_p\uff08c function\uff09": [[129, "c.i32_mod_p", false]], "i32_mod_s\uff08c function\uff09": [[129, "c.i32_mod_s", false]], "i32_mul_p\uff08c function\uff09": [[130, "c.i32_mul_p", false]], "i32_mul_s\uff08c function\uff09": [[130, "c.i32_mul_s", false]], "i32_neg_grad_p\uff08c function\uff09": [[133, "c.i32_neg_grad_p", false]], "i32_neg_grad_s\uff08c function\uff09": [[133, "c.i32_neg_grad_s", false]], "i32_neg_p\uff08c function\uff09": [[132, "c.i32_neg_p", false]], "i32_neg_s\uff08c function\uff09": [[132, "c.i32_neg_s", false]], "i32_nonzero_p\uff08c function\uff09": [[137, "c.i32_nonzero_p", false]], "i32_nonzero_s\uff08c function\uff09": [[137, "c.i32_nonzero_s", false]], "i32_not_equal_p\uff08c function\uff09": [[138, "c.i32_not_equal_p", false]], "i32_not_equal_s\uff08c function\uff09": [[138, "c.i32_not_equal_s", false]], "i32_onehot_p\uff08c function\uff09": [[139, "c.i32_onehot_p", false]], "i32_onehot_s\uff08c function\uff09": [[139, "c.i32_onehot_s", false]], "i32_ones_like_p\uff08c function\uff09": [[140, "c.i32_ones_like_p", false]], "i32_ones_like_s\uff08c function\uff09": [[140, "c.i32_ones_like_s", false]], "i32_padfusion_p\uff08c function\uff09": [[141, "c.i32_padfusion_p", false]], "i32_padfusion_s\uff08c function\uff09": [[141, "c.i32_padfusion_s", false]], "i32_pow_fusion_p\uff08c function\uff09": [[142, "c.i32_pow_fusion_p", false]], "i32_pow_fusion_s\uff08c function\uff09": [[142, "c.i32_pow_fusion_s", false]], "i32_raggedrange_p\uff08c function\uff09": [[147, "c.i32_raggedrange_p", false]], "i32_raggedrange_s\uff08c function\uff09": [[147, "c.i32_raggedrange_s", false]], "i32_range_p\uff08c function\uff09": [[150, "c.i32_range_p", false]], "i32_range_s\uff08c function\uff09": [[150, "c.i32_range_s", false]], "i32_real_div_p\uff08c function\uff09": [[152, "c.i32_real_div_p", false]], "i32_real_div_s\uff08c function\uff09": [[152, "c.i32_real_div_s", false]], "i32_reciprocal_p\uff08c function\uff09": [[153, "c.i32_reciprocal_p", false]], "i32_reciprocal_s\uff08c function\uff09": [[153, "c.i32_reciprocal_s", false]], "i32_reduce_p\uff08c function\uff09": [[154, "c.i32_reduce_p", false]], "i32_reduce_s\uff08c function\uff09": [[154, "c.i32_reduce_s", false]], "i32_reduceall_p\uff08c function\uff09": [[21, "c.i32_reduceall_p", false]], "i32_reduceall_s\uff08c function\uff09": [[21, "c.i32_reduceall_s", false]], "i32_reducescatter_p\uff08c function\uff09": [[155, "c.i32_reducescatter_p", false]], "i32_reducescatter_s\uff08c function\uff09": [[155, "c.i32_reducescatter_s", false]], "i32_reshape_p\uff08c function\uff09": [[156, "c.i32_reshape_p", false]], "i32_reshape_s\uff08c function\uff09": [[156, "c.i32_reshape_s", false]], "i32_rsqrt_p\uff08c function\uff09": [[164, "c.i32_rsqrt_p", false]], "i32_rsqrt_s\uff08c function\uff09": [[164, "c.i32_rsqrt_s", false]], "i32_scalefusion_p\uff08c function\uff09": [[166, "c.i32_scalefusion_p", false]], "i32_scalefusion_s\uff08c function\uff09": [[166, "c.i32_scalefusion_s", false]], "i32_scatter_elements_p\uff08c function\uff09": [[167, "c.i32_scatter_elements_p", false]], "i32_scatter_elements_s\uff08c function\uff09": [[167, "c.i32_scatter_elements_s", false]], "i32_scatter_nd_p\uff08c function\uff09": [[168, "c.i32_scatter_nd_p", false]], "i32_scatter_nd_s\uff08c function\uff09": [[168, "c.i32_scatter_nd_s", false]], "i32_scatter_nd_update_p\uff08c function\uff09": [[169, "c.i32_scatter_nd_update_p", false]], "i32_scatter_nd_update_s\uff08c function\uff09": [[169, "c.i32_scatter_nd_update_s", false]], "i32_select_p\uff08c function\uff09": [[170, "c.i32_select_p", false]], "i32_select_s\uff08c function\uff09": [[170, "c.i32_select_s", false]], "i32_sin_p\uff08c function\uff09": [[175, "c.i32_sin_p", false]], "i32_sin_s\uff08c function\uff09": [[175, "c.i32_sin_s", false]], "i32_slice_p\uff08c function\uff09": [[178, "c.i32_slice_p", false]], "i32_slice_s\uff08c function\uff09": [[178, "c.i32_slice_s", false]], "i32_spacetobatch_p\uff08c function\uff09": [[183, "c.i32_spacetobatch_p", false]], "i32_spacetobatch_s\uff08c function\uff09": [[183, "c.i32_spacetobatch_s", false]], "i32_spacetobatchnd_p\uff08c function\uff09": [[184, "c.i32_spacetobatchnd_p", false]], "i32_spacetobatchnd_s\uff08c function\uff09": [[184, "c.i32_spacetobatchnd_s", false]], "i32_spacetodepth_p\uff08c function\uff09": [[185, "c.i32_spacetodepth_p", false]], "i32_spacetodepth_s\uff08c function\uff09": [[185, "c.i32_spacetodepth_s", false]], "i32_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.i32_sparsefillemptyrows_p", false]], "i32_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.i32_sparsefillemptyrows_s", false]], "i32_sparsesegmentsum_p\uff08c function\uff09": [[189, "c.i32_sparsesegmentsum_p", false]], "i32_sparsesegmentsum_s\uff08c function\uff09": [[189, "c.i32_sparsesegmentsum_s", false]], "i32_sparsetodense_p\uff08c function\uff09": [[190, "c.i32_sparsetodense_p", false]], "i32_sparsetodense_s\uff08c function\uff09": [[190, "c.i32_sparsetodense_s", false]], "i32_splice_p\uff08c function\uff09": [[191, "c.i32_splice_p", false]], "i32_splice_s\uff08c function\uff09": [[191, "c.i32_splice_s", false]], "i32_split_p\uff08c function\uff09": [[192, "c.i32_split_p", false]], "i32_split_s\uff08c function\uff09": [[192, "c.i32_split_s", false]], "i32_split_with_overlap_p\uff08c function\uff09": [[193, "c.i32_split_with_overlap_p", false]], "i32_split_with_overlap_s\uff08c function\uff09": [[193, "c.i32_split_with_overlap_s", false]], "i32_sqrt_p\uff08c function\uff09": [[194, "c.i32_sqrt_p", false]], "i32_sqrt_s\uff08c function\uff09": [[194, "c.i32_sqrt_s", false]], "i32_sqrtgrad_p\uff08c function\uff09": [[195, "c.i32_sqrtgrad_p", false]], "i32_sqrtgrad_s\uff08c function\uff09": [[195, "c.i32_sqrtgrad_s", false]], "i32_square_p\uff08c function\uff09": [[196, "c.i32_square_p", false]], "i32_square_s\uff08c function\uff09": [[196, "c.i32_square_s", false]], "i32_squaredifference_p\uff08c function\uff09": [[197, "c.i32_squaredifference_p", false]], "i32_squaredifference_s\uff08c function\uff09": [[197, "c.i32_squaredifference_s", false]], "i32_stack_p\uff08c function\uff09": [[199, "c.i32_stack_p", false]], "i32_stack_s\uff08c function\uff09": [[199, "c.i32_stack_s", false]], "i32_subrelu6_p\uff08c function\uff09": [[202, "c.i32_subrelu6_p", false]], "i32_subrelu6_s\uff08c function\uff09": [[202, "c.i32_subrelu6_s", false]], "i32_subrelu_p\uff08c function\uff09": [[202, "c.i32_subrelu_p", false]], "i32_subrelu_s\uff08c function\uff09": [[202, "c.i32_subrelu_s", false]], "i32_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.i32_tensor_scatter_add_p", false]], "i32_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.i32_tensor_scatter_add_s", false]], "i32_tensorarrayread_p\uff08c function\uff09": [[208, "c.i32_tensorarrayread_p", false]], "i32_tensorarrayread_s\uff08c function\uff09": [[208, "c.i32_tensorarrayread_s", false]], "i32_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.i32_tensorlistfromtensor_p", false]], "i32_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.i32_tensorlistfromtensor_s", false]], "i32_tile_p\uff08c function\uff09": [[215, "c.i32_tile_p", false]], "i32_tile_s\uff08c function\uff09": [[215, "c.i32_tile_s", false]], "i32_topk_fusion_p\uff08c function\uff09": [[216, "c.i32_topk_fusion_p", false]], "i32_topk_fusion_s\uff08c function\uff09": [[216, "c.i32_topk_fusion_s", false]], "i32_transpose_p\uff08c function\uff09": [[217, "c.i32_transpose_p", false]], "i32_transpose_s\uff08c function\uff09": [[217, "c.i32_transpose_s", false]], "i32_tril_p\uff08c function\uff09": [[218, "c.i32_tril_p", false]], "i32_tril_s\uff08c function\uff09": [[218, "c.i32_tril_s", false]], "i32_triu_p\uff08c function\uff09": [[219, "c.i32_triu_p", false]], "i32_triu_s\uff08c function\uff09": [[219, "c.i32_triu_s", false]], "i32_unique_p\uff08c function\uff09": [[221, "c.i32_Unique_p", false]], "i32_unique_s\uff08c function\uff09": [[221, "c.i32_Unique_s", false]], "i32_unsorted_segment_sum_p\uff08c function\uff09": [[222, "c.i32_unsorted_segment_sum_p", false]], "i32_unsorted_segment_sum_s\uff08c function\uff09": [[222, "c.i32_unsorted_segment_sum_s", false]], "i32_where_p\uff08c function\uff09": [[225, "c.i32_where_p", false]], "i32_where_s\uff08c function\uff09": [[225, "c.i32_where_s", false]], "i32_zerolike_p\uff08c function\uff09": [[226, "c.i32_zerolike_p", false]], "i32_zerolike_s\uff08c function\uff09": [[226, "c.i32_zerolike_s", false]], "i8_abs_p\uff08c function\uff09": [[10, "c.i8_abs_p", false]], "i8_abs_s\uff08c function\uff09": [[10, "c.i8_abs_s", false]], "i8_adder_p\uff08c function\uff09": [[16, "c.i8_adder_p", false]], "i8_adder_s\uff08c function\uff09": [[16, "c.i8_adder_s", false]], "i8_addext_p\uff08c function\uff09": [[17, "c.i8_addext_p", false]], "i8_addext_s\uff08c function\uff09": [[17, "c.i8_addext_s", false]], "i8_addn_p\uff08c function\uff09": [[19, "c.i8_addn_p", false]], "i8_addn_s\uff08c function\uff09": [[19, "c.i8_addn_s", false]], "i8_addrelu6_p\uff08c function\uff09": [[17, "c.i8_addrelu6_p", false]], "i8_addrelu6_s\uff08c function\uff09": [[17, "c.i8_addrelu6_s", false]], "i8_addrelu_p\uff08c function\uff09": [[17, "c.i8_addrelu_p", false]], "i8_addrelu_s\uff08c function\uff09": [[17, "c.i8_addrelu_s", false]], "i8_affine_p\uff08c function\uff09": [[20, "c.i8_affine_p", false]], "i8_affine_s\uff08c function\uff09": [[20, "c.i8_affine_s", false]], "i8_allgather_p\uff08c function\uff09": [[22, "c.i8_allgather_p", false]], "i8_allgather_s\uff08c function\uff09": [[22, "c.i8_allgather_s", false]], "i8_and_p\uff08c function\uff09": [[112, "c.i8_and_p", false]], "i8_and_s\uff08c function\uff09": [[112, "c.i8_and_s", false]], "i8_assign_p\uff08c function\uff09": [[27, "c.i8_assign_p", false]], "i8_assign_s\uff08c function\uff09": [[27, "c.i8_assign_s", false]], "i8_assignadd_p\uff08c function\uff09": [[28, "c.i8_assignadd_p", false]], "i8_assignadd_s\uff08c function\uff09": [[28, "c.i8_assignadd_s", false]], "i8_avgpool_fusion_p\uff08c function\uff09": [[31, "c.i8_avgpool_fusion_p", false]], "i8_avgpool_fusion_s\uff08c function\uff09": [[31, "c.i8_avgpool_fusion_s", false]], "i8_batchnorm_p\uff08c function\uff09": [[33, "c.i8_batchnorm_p", false]], "i8_batchnorm_s\uff08c function\uff09": [[33, "c.i8_batchnorm_s", false]], "i8_batchtospace_p\uff08c function\uff09": [[35, "c.i8_batchtospace_p", false]], "i8_batchtospace_s\uff08c function\uff09": [[35, "c.i8_batchtospace_s", false]], "i8_batchtospacend_p\uff08c function\uff09": [[36, "c.i8_batchtospacend_p", false]], "i8_batchtospacend_s\uff08c function\uff09": [[36, "c.i8_batchtospacend_s", false]], "i8_biasadd_p\uff08c function\uff09": [[37, "c.i8_biasadd_p", false]], "i8_biasadd_s\uff08c function\uff09": [[37, "c.i8_biasadd_s", false]], "i8_binarycrossentropy_p\uff08c function\uff09": [[39, "c.i8_binarycrossentropy_p", false]], "i8_binarycrossentropy_s\uff08c function\uff09": [[39, "c.i8_binarycrossentropy_s", false]], "i8_broadcastto_p\uff08c function\uff09": [[41, "c.i8_broadcastto_p", false]], "i8_broadcastto_s\uff08c function\uff09": [[41, "c.i8_broadcastto_s", false]], "i8_celu_p\uff08c function\uff09": [[12, "c.i8_celu_p", false]], "i8_celu_s\uff08c function\uff09": [[12, "c.i8_celu_s", false]], "i8_clip_p\uff08c function\uff09": [[12, "c.i8_clip_p", false], [44, "c.i8_clip_p", false]], "i8_clip_s\uff08c function\uff09": [[12, "c.i8_clip_s", false], [44, "c.i8_clip_s", false]], "i8_concat_p\uff08c function\uff09": [[45, "c.i8_concat_p", false]], "i8_concat_s\uff08c function\uff09": [[45, "c.i8_concat_s", false]], "i8_constant_of_shape_p\uff08c function\uff09": [[46, "c.i8_constant_of_shape_p", false]], "i8_constant_of_shape_s\uff08c function\uff09": [[46, "c.i8_constant_of_shape_s", false]], "i8_conv2d_p\uff08c function\uff09": [[47, "c.i8_conv2d_p", false]], "i8_conv2d_s\uff08c function\uff09": [[47, "c.i8_conv2d_s", false]], "i8_convtranspose_p\uff08c function\uff09": [[48, "c.i8_convtranspose_p", false]], "i8_convtranspose_s\uff08c function\uff09": [[48, "c.i8_convtranspose_s", false]], "i8_cos_p\uff08c function\uff09": [[51, "c.i8_cos_p", false]], "i8_cos_s\uff08c function\uff09": [[51, "c.i8_cos_s", false]], "i8_crop_and_resize_anycore\uff08c function\uff09": [[53, "c.i8_crop_and_resize_anycore", false]], "i8_cumsum_p\uff08c function\uff09": [[54, "c.i8_cumsum_p", false]], "i8_cumsum_s\uff08c function\uff09": [[54, "c.i8_cumsum_s", false]], "i8_depthtospace_p\uff08c function\uff09": [[59, "c.i8_depthtospace_p", false]], "i8_depthtospace_s\uff08c function\uff09": [[59, "c.i8_depthtospace_s", false]], "i8_detection_post_process_p\uff08c function\uff09": [[60, "c.i8_detection_post_process_p", false]], "i8_detection_post_process_s\uff08c function\uff09": [[60, "c.i8_detection_post_process_s", false]], "i8_div_fusion_p\uff08c function\uff09": [[61, "c.i8_div_fusion_p", false]], "i8_div_fusion_s\uff08c function\uff09": [[61, "c.i8_div_fusion_s", false]], "i8_eltwise_p\uff08c function\uff09": [[67, "c.i8_eltwise_p", false]], "i8_eltwise_s\uff08c function\uff09": [[67, "c.i8_eltwise_s", false]], "i8_elu_p\uff08c function\uff09": [[12, "c.i8_elu_p", false], [68, "c.i8_elu_p", false]], "i8_elu_s\uff08c function\uff09": [[12, "c.i8_elu_s", false], [68, "c.i8_elu_s", false]], "i8_equal_p\uff08c function\uff09": [[70, "c.i8_equal_p", false]], "i8_equal_s\uff08c function\uff09": [[70, "c.i8_equal_s", false]], "i8_expfusion_p\uff08c function\uff09": [[73, "c.i8_expfusion_p", false]], "i8_expfusion_s\uff08c function\uff09": [[73, "c.i8_expfusion_s", false]], "i8_extract_features_p\uff08c function\uff09": [[55, "c.i8_extract_features_p", false]], "i8_extract_features_s\uff08c function\uff09": [[55, "c.i8_extract_features_s", false]], "i8_fill_p\uff08c function\uff09": [[78, "c.i8_fill_p", false]], "i8_fill_s\uff08c function\uff09": [[78, "c.i8_fill_s", false]], "i8_formattranspose_p\uff08c function\uff09": [[85, "c.i8_formattranspose_p", false]], "i8_formattranspose_s\uff08c function\uff09": [[85, "c.i8_formattranspose_s", false]], "i8_fullconnection_p\uff08c function\uff09": [[86, "c.i8_fullconnection_p", false]], "i8_fullconnection_s\uff08c function\uff09": [[86, "c.i8_fullconnection_s", false]], "i8_gather_nd_p\uff08c function\uff09": [[89, "c.i8_gather_nd_p", false]], "i8_gather_nd_s\uff08c function\uff09": [[89, "c.i8_gather_nd_s", false]], "i8_gather_p\uff08c function\uff09": [[88, "c.i8_gather_p", false]], "i8_gather_s\uff08c function\uff09": [[88, "c.i8_gather_s", false]], "i8_gatherd_p\uff08c function\uff09": [[90, "c.i8_gatherd_p", false]], "i8_gatherd_s\uff08c function\uff09": [[90, "c.i8_gatherd_s", false]], "i8_gelu_p\uff08c function\uff09": [[12, "c.i8_gelu_p", false]], "i8_gelu_s\uff08c function\uff09": [[12, "c.i8_gelu_s", false]], "i8_glu_p\uff08c function\uff09": [[91, "c.i8_glu_p", false]], "i8_glu_s\uff08c function\uff09": [[91, "c.i8_glu_s", false]], "i8_greater_p\uff08c function\uff09": [[92, "c.i8_greater_p", false]], "i8_greater_s\uff08c function\uff09": [[92, "c.i8_greater_s", false]], "i8_greaterequal_p\uff08c function\uff09": [[93, "c.i8_greaterequal_p", false]], "i8_greaterequal_s\uff08c function\uff09": [[93, "c.i8_greaterequal_s", false]], "i8_gru_p\uff08c function\uff09": [[95, "c.i8_Gru_p", false]], "i8_gru_s\uff08c function\uff09": [[95, "c.i8_Gru_s", false]], "i8_hardshrink_p\uff08c function\uff09": [[12, "c.i8_hardshrink_p", false]], "i8_hardshrink_s\uff08c function\uff09": [[12, "c.i8_hardshrink_s", false]], "i8_hardtanh_p\uff08c function\uff09": [[12, "c.i8_hardtanh_p", false]], "i8_hardtanh_s\uff08c function\uff09": [[12, "c.i8_hardtanh_s", false]], "i8_hsigmoid_p\uff08c function\uff09": [[12, "c.i8_hsigmoid_p", false]], "i8_hsigmoid_s\uff08c function\uff09": [[12, "c.i8_hsigmoid_s", false]], "i8_hswish_p\uff08c function\uff09": [[12, "c.i8_hswish_p", false]], "i8_hswish_s\uff08c function\uff09": [[12, "c.i8_hswish_s", false]], "i8_invertpermutation_p\uff08c function\uff09": [[98, "c.i8_invertpermutation_p", false]], "i8_invertpermutation_s\uff08c function\uff09": [[98, "c.i8_invertpermutation_s", false]], "i8_isfinite_p\uff08c function\uff09": [[99, "c.i8_isfinite_p", false]], "i8_isfinite_s\uff08c function\uff09": [[99, "c.i8_isfinite_s", false]], "i8_layernormfusion_p\uff08c function\uff09": [[101, "c.i8_layernormfusion_p", false]], "i8_layernormfusion_s\uff08c function\uff09": [[101, "c.i8_layernormfusion_s", false]], "i8_leaky_relu_p\uff08c function\uff09": [[103, "c.i8_leaky_relu_p", false]], "i8_leaky_relu_s\uff08c function\uff09": [[103, "c.i8_leaky_relu_s", false]], "i8_less_p\uff08c function\uff09": [[104, "c.i8_less_p", false]], "i8_less_s\uff08c function\uff09": [[104, "c.i8_less_s", false]], "i8_lessequal_p\uff08c function\uff09": [[105, "c.i8_lessequal_p", false]], "i8_lessequal_s\uff08c function\uff09": [[105, "c.i8_lessequal_s", false]], "i8_log1p_p\uff08c function\uff09": [[108, "c.i8_log1p_p", false]], "i8_log1p_s\uff08c function\uff09": [[108, "c.i8_log1p_s", false]], "i8_logical_not_p\uff08c function\uff09": [[110, "c.i8_logical_not_p", false]], "i8_logical_not_s\uff08c function\uff09": [[110, "c.i8_logical_not_s", false]], "i8_logical_or_p\uff08c function\uff09": [[111, "c.i8_logical_or_p", false]], "i8_logical_or_s\uff08c function\uff09": [[111, "c.i8_logical_or_s", false]], "i8_logsoftmax_p\uff08c function\uff09": [[113, "c.i8_logsoftmax_p", false]], "i8_logsoftmax_s\uff08c function\uff09": [[113, "c.i8_logsoftmax_s", false]], "i8_lrelu_p\uff08c function\uff09": [[12, "c.i8_lrelu_p", false]], "i8_lrelu_s\uff08c function\uff09": [[12, "c.i8_lrelu_s", false]], "i8_lsh_projection_p\uff08c function\uff09": [[116, "c.i8_lsh_projection_p", false]], "i8_lsh_projection_s\uff08c function\uff09": [[116, "c.i8_lsh_projection_s", false]], "i8_maximum_p\uff08c function\uff09": [[122, "c.i8_maximum_p", false]], "i8_maximum_s\uff08c function\uff09": [[122, "c.i8_maximum_s", false]], "i8_minimum_p\uff08c function\uff09": [[127, "c.i8_minimum_p", false]], "i8_minimum_s\uff08c function\uff09": [[127, "c.i8_minimum_s", false]], "i8_mod_p\uff08c function\uff09": [[129, "c.i8_mod_p", false]], "i8_mod_s\uff08c function\uff09": [[129, "c.i8_mod_s", false]], "i8_mul_p\uff08c function\uff09": [[130, "c.i8_mul_p", false]], "i8_mul_s\uff08c function\uff09": [[130, "c.i8_mul_s", false]], "i8_neg_grad_p\uff08c function\uff09": [[133, "c.i8_neg_grad_p", false]], "i8_neg_grad_s\uff08c function\uff09": [[133, "c.i8_neg_grad_s", false]], "i8_neg_p\uff08c function\uff09": [[132, "c.i8_neg_p", false]], "i8_neg_s\uff08c function\uff09": [[132, "c.i8_neg_s", false]], "i8_nllloss_p\uff08c function\uff09": [[134, "c.i8_nllloss_p", false]], "i8_nllloss_s\uff08c function\uff09": [[134, "c.i8_nllloss_s", false]], "i8_non_max_suppression_p\uff08c function\uff09": [[136, "c.i8_non_max_suppression_p", false]], "i8_non_max_suppression_s\uff08c function\uff09": [[136, "c.i8_non_max_suppression_s", false]], "i8_nonzero_p\uff08c function\uff09": [[137, "c.i8_nonzero_p", false]], "i8_nonzero_s\uff08c function\uff09": [[137, "c.i8_nonzero_s", false]], "i8_not_equal_p\uff08c function\uff09": [[138, "c.i8_not_equal_p", false]], "i8_not_equal_s\uff08c function\uff09": [[138, "c.i8_not_equal_s", false]], "i8_onehot_p\uff08c function\uff09": [[139, "c.i8_onehot_p", false]], "i8_onehot_s\uff08c function\uff09": [[139, "c.i8_onehot_s", false]], "i8_ones_like_p\uff08c function\uff09": [[140, "c.i8_ones_like_p", false]], "i8_ones_like_s\uff08c function\uff09": [[140, "c.i8_ones_like_s", false]], "i8_padfusion_p\uff08c function\uff09": [[141, "c.i8_padfusion_p", false]], "i8_padfusion_s\uff08c function\uff09": [[141, "c.i8_padfusion_s", false]], "i8_pow_fusion_p\uff08c function\uff09": [[142, "c.i8_pow_fusion_p", false]], "i8_pow_fusion_s\uff08c function\uff09": [[142, "c.i8_pow_fusion_s", false]], "i8_prelufusion_p\uff08c function\uff09": [[144, "c.i8_prelufusion_p", false]], "i8_prelufusion_s\uff08c function\uff09": [[144, "c.i8_prelufusion_s", false]], "i8_raggedrange_p\uff08c function\uff09": [[147, "c.i8_raggedrange_p", false]], "i8_raggedrange_s\uff08c function\uff09": [[147, "c.i8_raggedrange_s", false]], "i8_range_p\uff08c function\uff09": [[150, "c.i8_range_p", false]], "i8_range_s\uff08c function\uff09": [[150, "c.i8_range_s", false]], "i8_real_div_p\uff08c function\uff09": [[152, "c.i8_real_div_p", false]], "i8_real_div_s\uff08c function\uff09": [[152, "c.i8_real_div_s", false]], "i8_reciprocal_p\uff08c function\uff09": [[153, "c.i8_reciprocal_p", false]], "i8_reciprocal_s\uff08c function\uff09": [[153, "c.i8_reciprocal_s", false]], "i8_reduce_p\uff08c function\uff09": [[154, "c.i8_reduce_p", false]], "i8_reduce_s\uff08c function\uff09": [[154, "c.i8_reduce_s", false]], "i8_reduceall_p\uff08c function\uff09": [[21, "c.i8_reduceall_p", false]], "i8_reduceall_s\uff08c function\uff09": [[21, "c.i8_reduceall_s", false]], "i8_reducescatter_p\uff08c function\uff09": [[155, "c.i8_reducescatter_p", false]], "i8_reducescatter_s\uff08c function\uff09": [[155, "c.i8_reducescatter_s", false]], "i8_relu6_p\uff08c function\uff09": [[12, "c.i8_relu6_p", false]], "i8_relu6_s\uff08c function\uff09": [[12, "c.i8_relu6_s", false]], "i8_relu_p\uff08c function\uff09": [[12, "c.i8_relu_p", false]], "i8_relu_s\uff08c function\uff09": [[12, "c.i8_relu_s", false]], "i8_reshape_p\uff08c function\uff09": [[156, "c.i8_reshape_p", false]], "i8_reshape_s\uff08c function\uff09": [[156, "c.i8_reshape_s", false]], "i8_resize_anycore\uff08c function\uff09": [[157, "c.i8_resize_anycore", false]], "i8_roipooling_p\uff08c function\uff09": [[162, "c.i8_roipooling_p", false]], "i8_roipooling_s\uff08c function\uff09": [[162, "c.i8_roipooling_s", false]], "i8_rsqrt_p\uff08c function\uff09": [[164, "c.i8_rsqrt_p", false]], "i8_rsqrt_s\uff08c function\uff09": [[164, "c.i8_rsqrt_s", false]], "i8_scalefusion_p\uff08c function\uff09": [[166, "c.i8_scalefusion_p", false]], "i8_scalefusion_s\uff08c function\uff09": [[166, "c.i8_scalefusion_s", false]], "i8_scatter_elements_p\uff08c function\uff09": [[167, "c.i8_scatter_elements_p", false]], "i8_scatter_elements_s\uff08c function\uff09": [[167, "c.i8_scatter_elements_s", false]], "i8_scatter_nd_p\uff08c function\uff09": [[168, "c.i8_scatter_nd_p", false]], "i8_scatter_nd_s\uff08c function\uff09": [[168, "c.i8_scatter_nd_s", false]], "i8_scatter_nd_update_p\uff08c function\uff09": [[169, "c.i8_scatter_nd_update_p", false]], "i8_scatter_nd_update_s\uff08c function\uff09": [[169, "c.i8_scatter_nd_update_s", false]], "i8_select_p\uff08c function\uff09": [[170, "c.i8_select_p", false]], "i8_select_s\uff08c function\uff09": [[170, "c.i8_select_s", false]], "i8_sigmoid_p\uff08c function\uff09": [[12, "c.i8_sigmoid_p", false]], "i8_sigmoid_s\uff08c function\uff09": [[12, "c.i8_sigmoid_s", false]], "i8_sigmoidcrossentropywithlogits_p\uff08c function\uff09": [[174, "c.i8_sigmoidcrossentropywithlogits_p", false]], "i8_sigmoidcrossentropywithlogits_s\uff08c function\uff09": [[174, "c.i8_sigmoidcrossentropywithlogits_s", false]], "i8_sin_p\uff08c function\uff09": [[175, "c.i8_sin_p", false]], "i8_sin_s\uff08c function\uff09": [[175, "c.i8_sin_s", false]], "i8_slice_p\uff08c function\uff09": [[178, "c.i8_slice_p", false]], "i8_slice_s\uff08c function\uff09": [[178, "c.i8_slice_s", false]], "i8_smoothl1loss_p\uff08c function\uff09": [[179, "c.i8_smoothl1loss_p", false]], "i8_smoothl1loss_s\uff08c function\uff09": [[179, "c.i8_smoothl1loss_s", false]], "i8_softmax_cross_entropy_with_logits_p\uff08c function\uff09": [[182, "c.i8_softmax_cross_entropy_with_logits_p", false]], "i8_softmax_cross_entropy_with_logits_s\uff08c function\uff09": [[182, "c.i8_softmax_cross_entropy_with_logits_s", false]], "i8_softmax_p\uff08c function\uff09": [[181, "c.i8_softmax_p", false]], "i8_softmax_s\uff08c function\uff09": [[181, "c.i8_softmax_s", false]], "i8_softplus_p\uff08c function\uff09": [[12, "c.i8_softplus_p", false]], "i8_softplus_s\uff08c function\uff09": [[12, "c.i8_softplus_s", false]], "i8_softshrink_p\uff08c function\uff09": [[12, "c.i8_softshrink_p", false]], "i8_softshrink_s\uff08c function\uff09": [[12, "c.i8_softshrink_s", false]], "i8_softsignopt_p\uff08c function\uff09": [[12, "c.i8_softsignopt_p", false]], "i8_softsignopt_s\uff08c function\uff09": [[12, "c.i8_softsignopt_s", false]], "i8_spacetobatch_p\uff08c function\uff09": [[183, "c.i8_spacetobatch_p", false]], "i8_spacetobatch_s\uff08c function\uff09": [[183, "c.i8_spacetobatch_s", false]], "i8_spacetobatchnd_p\uff08c function\uff09": [[184, "c.i8_spacetobatchnd_p", false]], "i8_spacetobatchnd_s\uff08c function\uff09": [[184, "c.i8_spacetobatchnd_s", false]], "i8_spacetodepth_p\uff08c function\uff09": [[185, "c.i8_spacetodepth_p", false]], "i8_spacetodepth_s\uff08c function\uff09": [[185, "c.i8_spacetodepth_s", false]], "i8_sparsefillemptyrows_p\uff08c function\uff09": [[187, "c.i8_sparsefillemptyrows_p", false]], "i8_sparsefillemptyrows_s\uff08c function\uff09": [[187, "c.i8_sparsefillemptyrows_s", false]], "i8_sparsetodense_p\uff08c function\uff09": [[190, "c.i8_sparsetodense_p", false]], "i8_sparsetodense_s\uff08c function\uff09": [[190, "c.i8_sparsetodense_s", false]], "i8_splice_p\uff08c function\uff09": [[191, "c.i8_splice_p", false]], "i8_splice_s\uff08c function\uff09": [[191, "c.i8_splice_s", false]], "i8_split_p\uff08c function\uff09": [[192, "c.i8_split_p", false]], "i8_split_s\uff08c function\uff09": [[192, "c.i8_split_s", false]], "i8_split_with_overlap_p\uff08c function\uff09": [[193, "c.i8_split_with_overlap_p", false]], "i8_split_with_overlap_s\uff08c function\uff09": [[193, "c.i8_split_with_overlap_s", false]], "i8_sqrt_p\uff08c function\uff09": [[194, "c.i8_sqrt_p", false]], "i8_sqrt_s\uff08c function\uff09": [[194, "c.i8_sqrt_s", false]], "i8_sqrtgrad_p\uff08c function\uff09": [[195, "c.i8_sqrtgrad_p", false]], "i8_sqrtgrad_s\uff08c function\uff09": [[195, "c.i8_sqrtgrad_s", false]], "i8_square_p\uff08c function\uff09": [[196, "c.i8_square_p", false]], "i8_square_s\uff08c function\uff09": [[196, "c.i8_square_s", false]], "i8_squaredifference_p\uff08c function\uff09": [[197, "c.i8_squaredifference_p", false]], "i8_squaredifference_s\uff08c function\uff09": [[197, "c.i8_squaredifference_s", false]], "i8_stack_p\uff08c function\uff09": [[199, "c.i8_stack_p", false]], "i8_stack_s\uff08c function\uff09": [[199, "c.i8_stack_s", false]], "i8_subrelu6_p\uff08c function\uff09": [[202, "c.i8_subrelu6_p", false]], "i8_subrelu6_s\uff08c function\uff09": [[202, "c.i8_subrelu6_s", false]], "i8_subrelu_p\uff08c function\uff09": [[202, "c.i8_subrelu_p", false]], "i8_subrelu_s\uff08c function\uff09": [[202, "c.i8_subrelu_s", false]], "i8_swish_p\uff08c function\uff09": [[12, "c.i8_swish_p", false]], "i8_swish_s\uff08c function\uff09": [[12, "c.i8_swish_s", false]], "i8_tanh_p\uff08c function\uff09": [[12, "c.i8_tanh_p", false]], "i8_tanh_s\uff08c function\uff09": [[12, "c.i8_tanh_s", false]], "i8_tensor_scatter_add_p\uff08c function\uff09": [[206, "c.i8_tensor_scatter_add_p", false]], "i8_tensor_scatter_add_s\uff08c function\uff09": [[206, "c.i8_tensor_scatter_add_s", false]], "i8_tensorarrayread_p\uff08c function\uff09": [[208, "c.i8_tensorarrayread_p", false]], "i8_tensorarrayread_s\uff08c function\uff09": [[208, "c.i8_tensorarrayread_s", false]], "i8_tensorlistfromtensor_p\uff08c function\uff09": [[210, "c.i8_tensorlistfromtensor_p", false]], "i8_tensorlistfromtensor_s\uff08c function\uff09": [[210, "c.i8_tensorlistfromtensor_s", false]], "i8_tile_p\uff08c function\uff09": [[215, "c.i8_tile_p", false]], "i8_tile_s\uff08c function\uff09": [[215, "c.i8_tile_s", false]], "i8_to_fp_dequant_p\uff08c function\uff09": [[146, "c.i8_to_fp_dequant_p", false]], "i8_to_fp_dequant_s\uff08c function\uff09": [[146, "c.i8_to_fp_dequant_s", false]], "i8_to_hp_dequant_p\uff08c function\uff09": [[146, "c.i8_to_hp_dequant_p", false]], "i8_to_hp_dequant_s\uff08c function\uff09": [[146, "c.i8_to_hp_dequant_s", false]], "i8_transpose_p\uff08c function\uff09": [[217, "c.i8_transpose_p", false]], "i8_transpose_s\uff08c function\uff09": [[217, "c.i8_transpose_s", false]], "i8_tril_p\uff08c function\uff09": [[218, "c.i8_tril_p", false]], "i8_tril_s\uff08c function\uff09": [[218, "c.i8_tril_s", false]], "i8_triu_p\uff08c function\uff09": [[219, "c.i8_triu_p", false]], "i8_triu_s\uff08c function\uff09": [[219, "c.i8_triu_s", false]], "i8_unique_p\uff08c function\uff09": [[221, "c.i8_Unique_p", false]], "i8_unique_s\uff08c function\uff09": [[221, "c.i8_Unique_s", false]], "i8_unsorted_segment_sum_p\uff08c function\uff09": [[222, "c.i8_unsorted_segment_sum_p", false]], "i8_unsorted_segment_sum_s\uff08c function\uff09": [[222, "c.i8_unsorted_segment_sum_s", false]], "i8_where_p\uff08c function\uff09": [[225, "c.i8_where_p", false]], "i8_where_s\uff08c function\uff09": [[225, "c.i8_where_s", false]], "i8_zerolike_p\uff08c function\uff09": [[226, "c.i8_zerolike_p", false]], "i8_zerolike_s\uff08c function\uff09": [[226, "c.i8_zerolike_s", false]], "mindradar.complexabs\uff08\u5185\u7f6e\u7c7b\uff09": [[6, "mindradar.ComplexAbs", false]], "mindradar.fft\uff08\u5185\u7f6e\u7c7b\uff09": [[7, "mindradar.FFT", false]], "mindradar.ifft\uff08\u5185\u7f6e\u7c7b\uff09": [[8, "mindradar.IFFT", false]], "rank_p\uff08c function\uff09": [[151, "c.rank_p", false]], "rank_s\uff08c function\uff09": [[151, "c.rank_s", false]], "shape\uff08c function\uff09": [[172, "c.shape", false]], "size_p\uff08c function\uff09": [[176, "c.size_p", false]], "size_s\uff08c function\uff09": [[176, "c.size_s", false]], "skipgram_p\uff08c function\uff09": [[177, "c.skipgram_p", false]], "skipgram_s\uff08c function\uff09": [[177, "c.skipgram_s", false]], "sparsereshape_p\uff08c function\uff09": [[188, "c.sparsereshape_p", false]], "sparsereshape_s\uff08c function\uff09": [[188, "c.sparsereshape_s", false]], "stridedslice\uff08c function\uff09": [[200, "c.stridedslice", false]], "switch_p\uff08c function\uff09": [[204, "c.switch_p", false]], "switch_s\uff08c function\uff09": [[204, "c.switch_s", false]], "switchlayer_p\uff08c function\uff09": [[205, "c.switchlayer_p", false]], "switchlayer_s\uff08c function\uff09": [[205, "c.switchlayer_s", false]], "tensorarrayread_p\uff08c function\uff09": [[207, "c.tensorarrayread_p", false]], "tensorarrayread_s\uff08c function\uff09": [[207, "c.tensorarrayread_s", false]], "tensorarraywrite_p\uff08c function\uff09": [[207, "c.tensorarraywrite_p", false]], "tensorarraywrite_s\uff08c function\uff09": [[207, "c.tensorarraywrite_s", false]], "tensorlistgetitem_p\uff08c function\uff09": [[211, "c.tensorlistgetitem_p", false]], "tensorlistgetitem_s\uff08c function\uff09": [[211, "c.tensorlistgetitem_s", false]], "tensorlistreserve\uff08c function\uff09": [[212, "c.Tensorlistreserve", false]], "tensorlistsetitem_p\uff08c function\uff09": [[213, "c.tensorlistsetitem_p", false]], "tensorlistsetitem_s\uff08c function\uff09": [[213, "c.tensorlistsetitem_s", false]], "tensorliststack_p\uff08c function\uff09": [[214, "c.tensorliststack_p", false]], "tensorliststack_s\uff08c function\uff09": [[214, "c.tensorliststack_s", false]], "unstack_p\uff08c function\uff09": [[224, "c.unstack_p", false]], "unstack_s\uff08c function\uff09": [[224, "c.unstack_s", false]]}, "objects": {"": [[80, 0, 1, "c.Flatten", "Flatten"], [212, 0, 1, "c.Tensorlistreserve", "Tensorlistreserve"], [52, 0, 1, "c.anytype_crop_anycore", "anytype_crop_anycore"], [72, 0, 1, "c.anytype_expand_dims_anycore", "anytype_expand_dims_anycore"], [79, 0, 1, "c.anytype_fillv2_p", "anytype_fillv2_p"], [79, 0, 1, "c.anytype_fillv2_s", "anytype_fillv2_s"], [159, 0, 1, "c.anytype_reverse_sequence_anycore", "anytype_reverse_sequence_anycore"], [160, 0, 1, "c.anytype_reversev2_anycore", "anytype_reversev2_anycore"], [198, 0, 1, "c.anytype_squeeze_anycore", "anytype_squeeze_anycore"], [223, 0, 1, "c.anytype_unsqueeze_anycore", "anytype_unsqueeze_anycore"], [26, 0, 1, "c.assert", "assert"], [221, 0, 1, "c.c128_Unique_p", "c128_Unique_p"], [221, 0, 1, "c.c128_Unique_s", "c128_Unique_s"], [10, 0, 1, "c.c128_abs_p", "c128_abs_p"], [10, 0, 1, "c.c128_abs_s", "c128_abs_s"], [17, 0, 1, "c.c128_addext_p", "c128_addext_p"], [17, 0, 1, "c.c128_addext_s", "c128_addext_s"], [19, 0, 1, "c.c128_addn_p", "c128_addn_p"], [19, 0, 1, "c.c128_addn_s", "c128_addn_s"], [17, 0, 1, "c.c128_addrelu6_p", "c128_addrelu6_p"], [17, 0, 1, "c.c128_addrelu6_s", "c128_addrelu6_s"], [17, 0, 1, "c.c128_addrelu_p", "c128_addrelu_p"], [17, 0, 1, "c.c128_addrelu_s", "c128_addrelu_s"], [22, 0, 1, "c.c128_allgather_p", "c128_allgather_p"], [22, 0, 1, "c.c128_allgather_s", "c128_allgather_s"], [27, 0, 1, "c.c128_assign_p", "c128_assign_p"], [27, 0, 1, "c.c128_assign_s", "c128_assign_s"], [28, 0, 1, "c.c128_assignadd_p", "c128_assignadd_p"], [28, 0, 1, "c.c128_assignadd_s", "c128_assignadd_s"], [35, 0, 1, "c.c128_batchtospace_p", "c128_batchtospace_p"], [35, 0, 1, "c.c128_batchtospace_s", "c128_batchtospace_s"], [36, 0, 1, "c.c128_batchtospacend_p", "c128_batchtospacend_p"], [36, 0, 1, "c.c128_batchtospacend_s", "c128_batchtospacend_s"], [37, 0, 1, "c.c128_biasadd_p", "c128_biasadd_p"], [37, 0, 1, "c.c128_biasadd_s", "c128_biasadd_s"], [41, 0, 1, "c.c128_broadcastto_p", "c128_broadcastto_p"], [41, 0, 1, "c.c128_broadcastto_s", "c128_broadcastto_s"], [45, 0, 1, "c.c128_concat_p", "c128_concat_p"], [45, 0, 1, "c.c128_concat_s", "c128_concat_s"], [46, 0, 1, "c.c128_constant_of_shape_p", "c128_constant_of_shape_p"], [46, 0, 1, "c.c128_constant_of_shape_s", "c128_constant_of_shape_s"], [54, 0, 1, "c.c128_cumsum_p", "c128_cumsum_p"], [54, 0, 1, "c.c128_cumsum_s", "c128_cumsum_s"], [59, 0, 1, "c.c128_depthtospace_p", "c128_depthtospace_p"], [59, 0, 1, "c.c128_depthtospace_s", "c128_depthtospace_s"], [61, 0, 1, "c.c128_div_fusion_p", "c128_div_fusion_p"], [61, 0, 1, "c.c128_div_fusion_s", "c128_div_fusion_s"], [67, 0, 1, "c.c128_eltwise_p", "c128_eltwise_p"], [67, 0, 1, "c.c128_eltwise_s", "c128_eltwise_s"], [70, 0, 1, "c.c128_equal_p", "c128_equal_p"], [70, 0, 1, "c.c128_equal_s", "c128_equal_s"], [73, 0, 1, "c.c128_expfusion_p", "c128_expfusion_p"], [73, 0, 1, "c.c128_expfusion_s", "c128_expfusion_s"], [55, 0, 1, "c.c128_extract_features_p", "c128_extract_features_p"], [55, 0, 1, "c.c128_extract_features_s", "c128_extract_features_s"], [78, 0, 1, "c.c128_fill_p", "c128_fill_p"], [78, 0, 1, "c.c128_fill_s", "c128_fill_s"], [85, 0, 1, "c.c128_formattranspose_p", "c128_formattranspose_p"], [85, 0, 1, "c.c128_formattranspose_s", "c128_formattranspose_s"], [89, 0, 1, "c.c128_gather_nd_p", "c128_gather_nd_p"], [89, 0, 1, "c.c128_gather_nd_s", "c128_gather_nd_s"], [88, 0, 1, "c.c128_gather_p", "c128_gather_p"], [88, 0, 1, "c.c128_gather_s", "c128_gather_s"], [90, 0, 1, "c.c128_gatherd_p", "c128_gatherd_p"], [90, 0, 1, "c.c128_gatherd_s", "c128_gatherd_s"], [99, 0, 1, "c.c128_isfinite_p", "c128_isfinite_p"], [99, 0, 1, "c.c128_isfinite_s", "c128_isfinite_s"], [121, 0, 1, "c.c128_matmulfusion_p", "c128_matmulfusion_p"], [121, 0, 1, "c.c128_matmulfusion_s", "c128_matmulfusion_s"], [130, 0, 1, "c.c128_mul_p", "c128_mul_p"], [130, 0, 1, "c.c128_mul_s", "c128_mul_s"], [133, 0, 1, "c.c128_neg_grad_p", "c128_neg_grad_p"], [133, 0, 1, "c.c128_neg_grad_s", "c128_neg_grad_s"], [132, 0, 1, "c.c128_neg_p", "c128_neg_p"], [132, 0, 1, "c.c128_neg_s", "c128_neg_s"], [137, 0, 1, "c.c128_nonzero_p", "c128_nonzero_p"], [137, 0, 1, "c.c128_nonzero_s", "c128_nonzero_s"], [138, 0, 1, "c.c128_not_equal_p", "c128_not_equal_p"], [138, 0, 1, "c.c128_not_equal_s", "c128_not_equal_s"], [139, 0, 1, "c.c128_onehot_p", "c128_onehot_p"], [139, 0, 1, "c.c128_onehot_s", "c128_onehot_s"], [140, 0, 1, "c.c128_ones_like_p", "c128_ones_like_p"], [140, 0, 1, "c.c128_ones_like_s", "c128_ones_like_s"], [141, 0, 1, "c.c128_padfusion_p", "c128_padfusion_p"], [141, 0, 1, "c.c128_padfusion_s", "c128_padfusion_s"], [152, 0, 1, "c.c128_real_div_p", "c128_real_div_p"], [152, 0, 1, "c.c128_real_div_s", "c128_real_div_s"], [153, 0, 1, "c.c128_reciprocal_p", "c128_reciprocal_p"], [153, 0, 1, "c.c128_reciprocal_s", "c128_reciprocal_s"], [21, 0, 1, "c.c128_reduceall_p", "c128_reduceall_p"], [21, 0, 1, "c.c128_reduceall_s", "c128_reduceall_s"], [156, 0, 1, "c.c128_reshape_p", "c128_reshape_p"], [156, 0, 1, "c.c128_reshape_s", "c128_reshape_s"], [161, 0, 1, "c.c128_rfft_p", "c128_rfft_p"], [161, 0, 1, "c.c128_rfft_s", "c128_rfft_s"], [164, 0, 1, "c.c128_rsqrt_p", "c128_rsqrt_p"], [164, 0, 1, "c.c128_rsqrt_s", "c128_rsqrt_s"], [167, 0, 1, "c.c128_scatter_elements_p", "c128_scatter_elements_p"], [167, 0, 1, "c.c128_scatter_elements_s", "c128_scatter_elements_s"], [168, 0, 1, "c.c128_scatter_nd_p", "c128_scatter_nd_p"], [168, 0, 1, "c.c128_scatter_nd_s", "c128_scatter_nd_s"], [169, 0, 1, "c.c128_scatter_nd_update_p", "c128_scatter_nd_update_p"], [169, 0, 1, "c.c128_scatter_nd_update_s", "c128_scatter_nd_update_s"], [170, 0, 1, "c.c128_select_p", "c128_select_p"], [170, 0, 1, "c.c128_select_s", "c128_select_s"], [178, 0, 1, "c.c128_slice_p", "c128_slice_p"], [178, 0, 1, "c.c128_slice_s", "c128_slice_s"], [183, 0, 1, "c.c128_spacetobatch_p", "c128_spacetobatch_p"], [183, 0, 1, "c.c128_spacetobatch_s", "c128_spacetobatch_s"], [184, 0, 1, "c.c128_spacetobatchnd_p", "c128_spacetobatchnd_p"], [184, 0, 1, "c.c128_spacetobatchnd_s", "c128_spacetobatchnd_s"], [185, 0, 1, "c.c128_spacetodepth_p", "c128_spacetodepth_p"], [185, 0, 1, "c.c128_spacetodepth_s", "c128_spacetodepth_s"], [187, 0, 1, "c.c128_sparsefillemptyrows_p", "c128_sparsefillemptyrows_p"], [187, 0, 1, "c.c128_sparsefillemptyrows_s", "c128_sparsefillemptyrows_s"], [189, 0, 1, "c.c128_sparsesegmentsum_p", "c128_sparsesegmentsum_p"], [189, 0, 1, "c.c128_sparsesegmentsum_s", "c128_sparsesegmentsum_s"], [190, 0, 1, "c.c128_sparsetodense_p", "c128_sparsetodense_p"], [190, 0, 1, "c.c128_sparsetodense_s", "c128_sparsetodense_s"], [191, 0, 1, "c.c128_splice_p", "c128_splice_p"], [191, 0, 1, "c.c128_splice_s", "c128_splice_s"], [192, 0, 1, "c.c128_split_p", "c128_split_p"], [192, 0, 1, "c.c128_split_s", "c128_split_s"], [193, 0, 1, "c.c128_split_with_overlap_p", "c128_split_with_overlap_p"], [193, 0, 1, "c.c128_split_with_overlap_s", "c128_split_with_overlap_s"], [194, 0, 1, "c.c128_sqrt_p", "c128_sqrt_p"], [194, 0, 1, "c.c128_sqrt_s", "c128_sqrt_s"], [195, 0, 1, "c.c128_sqrtgrad_p", "c128_sqrtgrad_p"], [195, 0, 1, "c.c128_sqrtgrad_s", "c128_sqrtgrad_s"], [196, 0, 1, "c.c128_square_p", "c128_square_p"], [196, 0, 1, "c.c128_square_s", "c128_square_s"], [197, 0, 1, "c.c128_squaredifference_p", "c128_squaredifference_p"], [197, 0, 1, "c.c128_squaredifference_s", "c128_squaredifference_s"], [199, 0, 1, "c.c128_stack_p", "c128_stack_p"], [199, 0, 1, "c.c128_stack_s", "c128_stack_s"], [202, 0, 1, "c.c128_subext_p", "c128_subext_p"], [202, 0, 1, "c.c128_subext_s", "c128_subext_s"], [202, 0, 1, "c.c128_subrelu6_p", "c128_subrelu6_p"], [202, 0, 1, "c.c128_subrelu6_s", "c128_subrelu6_s"], [202, 0, 1, "c.c128_subrelu_p", "c128_subrelu_p"], [202, 0, 1, "c.c128_subrelu_s", "c128_subrelu_s"], [206, 0, 1, "c.c128_tensor_scatter_add_p", "c128_tensor_scatter_add_p"], [206, 0, 1, "c.c128_tensor_scatter_add_s", "c128_tensor_scatter_add_s"], [208, 0, 1, "c.c128_tensorarrayread_p", "c128_tensorarrayread_p"], [208, 0, 1, "c.c128_tensorarrayread_s", "c128_tensorarrayread_s"], [210, 0, 1, "c.c128_tensorlistfromtensor_p", "c128_tensorlistfromtensor_p"], [210, 0, 1, "c.c128_tensorlistfromtensor_s", "c128_tensorlistfromtensor_s"], [215, 0, 1, "c.c128_tile_p", "c128_tile_p"], [215, 0, 1, "c.c128_tile_s", "c128_tile_s"], [217, 0, 1, "c.c128_transpose_p", "c128_transpose_p"], [217, 0, 1, "c.c128_transpose_s", "c128_transpose_s"], [218, 0, 1, "c.c128_tril_p", "c128_tril_p"], [218, 0, 1, "c.c128_tril_s", "c128_tril_s"], [219, 0, 1, "c.c128_triu_p", "c128_triu_p"], [219, 0, 1, "c.c128_triu_s", "c128_triu_s"], [222, 0, 1, "c.c128_unsorted_segment_sum_p", "c128_unsorted_segment_sum_p"], [222, 0, 1, "c.c128_unsorted_segment_sum_s", "c128_unsorted_segment_sum_s"], [225, 0, 1, "c.c128_where_p", "c128_where_p"], [225, 0, 1, "c.c128_where_s", "c128_where_s"], [226, 0, 1, "c.c128_zerolike_p", "c128_zerolike_p"], [226, 0, 1, "c.c128_zerolike_s", "c128_zerolike_s"], [221, 0, 1, "c.c64_Unique_p", "c64_Unique_p"], [221, 0, 1, "c.c64_Unique_s", "c64_Unique_s"], [10, 0, 1, "c.c64_abs_p", "c64_abs_p"], [10, 0, 1, "c.c64_abs_s", "c64_abs_s"], [17, 0, 1, "c.c64_addext_p", "c64_addext_p"], [17, 0, 1, "c.c64_addext_s", "c64_addext_s"], [19, 0, 1, "c.c64_addn_p", "c64_addn_p"], [19, 0, 1, "c.c64_addn_s", "c64_addn_s"], [17, 0, 1, "c.c64_addrelu6_p", "c64_addrelu6_p"], [17, 0, 1, "c.c64_addrelu6_s", "c64_addrelu6_s"], [17, 0, 1, "c.c64_addrelu_p", "c64_addrelu_p"], [17, 0, 1, "c.c64_addrelu_s", "c64_addrelu_s"], [22, 0, 1, "c.c64_allgather_p", "c64_allgather_p"], [22, 0, 1, "c.c64_allgather_s", "c64_allgather_s"], [27, 0, 1, "c.c64_assign_p", "c64_assign_p"], [27, 0, 1, "c.c64_assign_s", "c64_assign_s"], [28, 0, 1, "c.c64_assignadd_p", "c64_assignadd_p"], [28, 0, 1, "c.c64_assignadd_s", "c64_assignadd_s"], [35, 0, 1, "c.c64_batchtospace_p", "c64_batchtospace_p"], [35, 0, 1, "c.c64_batchtospace_s", "c64_batchtospace_s"], [36, 0, 1, "c.c64_batchtospacend_p", "c64_batchtospacend_p"], [36, 0, 1, "c.c64_batchtospacend_s", "c64_batchtospacend_s"], [37, 0, 1, "c.c64_biasadd_p", "c64_biasadd_p"], [37, 0, 1, "c.c64_biasadd_s", "c64_biasadd_s"], [41, 0, 1, "c.c64_broadcastto_p", "c64_broadcastto_p"], [41, 0, 1, "c.c64_broadcastto_s", "c64_broadcastto_s"], [45, 0, 1, "c.c64_concat_p", "c64_concat_p"], [45, 0, 1, "c.c64_concat_s", "c64_concat_s"], [46, 0, 1, "c.c64_constant_of_shape_p", "c64_constant_of_shape_p"], [46, 0, 1, "c.c64_constant_of_shape_s", "c64_constant_of_shape_s"], [54, 0, 1, "c.c64_cumsum_p", "c64_cumsum_p"], [54, 0, 1, "c.c64_cumsum_s", "c64_cumsum_s"], [59, 0, 1, "c.c64_depthtospace_p", "c64_depthtospace_p"], [59, 0, 1, "c.c64_depthtospace_s", "c64_depthtospace_s"], [61, 0, 1, "c.c64_div_fusion_p", "c64_div_fusion_p"], [61, 0, 1, "c.c64_div_fusion_s", "c64_div_fusion_s"], [67, 0, 1, "c.c64_eltwise_p", "c64_eltwise_p"], [67, 0, 1, "c.c64_eltwise_s", "c64_eltwise_s"], [70, 0, 1, "c.c64_equal_p", "c64_equal_p"], [70, 0, 1, "c.c64_equal_s", "c64_equal_s"], [73, 0, 1, "c.c64_expfusion_p", "c64_expfusion_p"], [73, 0, 1, "c.c64_expfusion_s", "c64_expfusion_s"], [55, 0, 1, "c.c64_extract_features_p", "c64_extract_features_p"], [55, 0, 1, "c.c64_extract_features_s", "c64_extract_features_s"], [78, 0, 1, "c.c64_fill_p", "c64_fill_p"], [78, 0, 1, "c.c64_fill_s", "c64_fill_s"], [85, 0, 1, "c.c64_formattranspose_p", "c64_formattranspose_p"], [85, 0, 1, "c.c64_formattranspose_s", "c64_formattranspose_s"], [89, 0, 1, "c.c64_gather_nd_p", "c64_gather_nd_p"], [89, 0, 1, "c.c64_gather_nd_s", "c64_gather_nd_s"], [88, 0, 1, "c.c64_gather_p", "c64_gather_p"], [88, 0, 1, "c.c64_gather_s", "c64_gather_s"], [90, 0, 1, "c.c64_gatherd_p", "c64_gatherd_p"], [90, 0, 1, "c.c64_gatherd_s", "c64_gatherd_s"], [99, 0, 1, "c.c64_isfinite_p", "c64_isfinite_p"], [99, 0, 1, "c.c64_isfinite_s", "c64_isfinite_s"], [121, 0, 1, "c.c64_matmulfusion_p", "c64_matmulfusion_p"], [121, 0, 1, "c.c64_matmulfusion_s", "c64_matmulfusion_s"], [130, 0, 1, "c.c64_mul_p", "c64_mul_p"], [130, 0, 1, "c.c64_mul_s", "c64_mul_s"], [133, 0, 1, "c.c64_neg_grad_p", "c64_neg_grad_p"], [133, 0, 1, "c.c64_neg_grad_s", "c64_neg_grad_s"], [132, 0, 1, "c.c64_neg_p", "c64_neg_p"], [132, 0, 1, "c.c64_neg_s", "c64_neg_s"], [137, 0, 1, "c.c64_nonzero_p", "c64_nonzero_p"], [137, 0, 1, "c.c64_nonzero_s", "c64_nonzero_s"], [138, 0, 1, "c.c64_not_equal_p", "c64_not_equal_p"], [138, 0, 1, "c.c64_not_equal_s", "c64_not_equal_s"], [139, 0, 1, "c.c64_onehot_p", "c64_onehot_p"], [139, 0, 1, "c.c64_onehot_s", "c64_onehot_s"], [140, 0, 1, "c.c64_ones_like_p", "c64_ones_like_p"], [140, 0, 1, "c.c64_ones_like_s", "c64_ones_like_s"], [141, 0, 1, "c.c64_padfusion_p", "c64_padfusion_p"], [141, 0, 1, "c.c64_padfusion_s", "c64_padfusion_s"], [152, 0, 1, "c.c64_real_div_p", "c64_real_div_p"], [152, 0, 1, "c.c64_real_div_s", "c64_real_div_s"], [153, 0, 1, "c.c64_reciprocal_p", "c64_reciprocal_p"], [153, 0, 1, "c.c64_reciprocal_s", "c64_reciprocal_s"], [21, 0, 1, "c.c64_reduceall_p", "c64_reduceall_p"], [21, 0, 1, "c.c64_reduceall_s", "c64_reduceall_s"], [156, 0, 1, "c.c64_reshape_p", "c64_reshape_p"], [156, 0, 1, "c.c64_reshape_s", "c64_reshape_s"], [161, 0, 1, "c.c64_rfft_p", "c64_rfft_p"], [161, 0, 1, "c.c64_rfft_s", "c64_rfft_s"], [164, 0, 1, "c.c64_rsqrt_p", "c64_rsqrt_p"], [164, 0, 1, "c.c64_rsqrt_s", "c64_rsqrt_s"], [167, 0, 1, "c.c64_scatter_elements_p", "c64_scatter_elements_p"], [167, 0, 1, "c.c64_scatter_elements_s", "c64_scatter_elements_s"], [168, 0, 1, "c.c64_scatter_nd_p", "c64_scatter_nd_p"], [168, 0, 1, "c.c64_scatter_nd_s", "c64_scatter_nd_s"], [169, 0, 1, "c.c64_scatter_nd_update_p", "c64_scatter_nd_update_p"], [169, 0, 1, "c.c64_scatter_nd_update_s", "c64_scatter_nd_update_s"], [170, 0, 1, "c.c64_select_p", "c64_select_p"], [170, 0, 1, "c.c64_select_s", "c64_select_s"], [178, 0, 1, "c.c64_slice_p", "c64_slice_p"], [178, 0, 1, "c.c64_slice_s", "c64_slice_s"], [183, 0, 1, "c.c64_spacetobatch_p", "c64_spacetobatch_p"], [183, 0, 1, "c.c64_spacetobatch_s", "c64_spacetobatch_s"], [184, 0, 1, "c.c64_spacetobatchnd_p", "c64_spacetobatchnd_p"], [184, 0, 1, "c.c64_spacetobatchnd_s", "c64_spacetobatchnd_s"], [185, 0, 1, "c.c64_spacetodepth_p", "c64_spacetodepth_p"], [185, 0, 1, "c.c64_spacetodepth_s", "c64_spacetodepth_s"], [187, 0, 1, "c.c64_sparsefillemptyrows_p", "c64_sparsefillemptyrows_p"], [187, 0, 1, "c.c64_sparsefillemptyrows_s", "c64_sparsefillemptyrows_s"], [189, 0, 1, "c.c64_sparsesegmentsum_p", "c64_sparsesegmentsum_p"], [189, 0, 1, "c.c64_sparsesegmentsum_s", "c64_sparsesegmentsum_s"], [190, 0, 1, "c.c64_sparsetodense_p", "c64_sparsetodense_p"], [190, 0, 1, "c.c64_sparsetodense_s", "c64_sparsetodense_s"], [191, 0, 1, "c.c64_splice_p", "c64_splice_p"], [191, 0, 1, "c.c64_splice_s", "c64_splice_s"], [192, 0, 1, "c.c64_split_p", "c64_split_p"], [192, 0, 1, "c.c64_split_s", "c64_split_s"], [193, 0, 1, "c.c64_split_with_overlap_p", "c64_split_with_overlap_p"], [193, 0, 1, "c.c64_split_with_overlap_s", "c64_split_with_overlap_s"], [194, 0, 1, "c.c64_sqrt_p", "c64_sqrt_p"], [194, 0, 1, "c.c64_sqrt_s", "c64_sqrt_s"], [195, 0, 1, "c.c64_sqrtgrad_p", "c64_sqrtgrad_p"], [195, 0, 1, "c.c64_sqrtgrad_s", "c64_sqrtgrad_s"], [196, 0, 1, "c.c64_square_p", "c64_square_p"], [196, 0, 1, "c.c64_square_s", "c64_square_s"], [197, 0, 1, "c.c64_squaredifference_p", "c64_squaredifference_p"], [197, 0, 1, "c.c64_squaredifference_s", "c64_squaredifference_s"], [199, 0, 1, "c.c64_stack_p", "c64_stack_p"], [199, 0, 1, "c.c64_stack_s", "c64_stack_s"], [202, 0, 1, "c.c64_subext_p", "c64_subext_p"], [202, 0, 1, "c.c64_subext_s", "c64_subext_s"], [202, 0, 1, "c.c64_subrelu6_p", "c64_subrelu6_p"], [202, 0, 1, "c.c64_subrelu6_s", "c64_subrelu6_s"], [202, 0, 1, "c.c64_subrelu_p", "c64_subrelu_p"], [202, 0, 1, "c.c64_subrelu_s", "c64_subrelu_s"], [206, 0, 1, "c.c64_tensor_scatter_add_p", "c64_tensor_scatter_add_p"], [206, 0, 1, "c.c64_tensor_scatter_add_s", "c64_tensor_scatter_add_s"], [208, 0, 1, "c.c64_tensorarrayread_p", "c64_tensorarrayread_p"], [208, 0, 1, "c.c64_tensorarrayread_s", "c64_tensorarrayread_s"], [210, 0, 1, "c.c64_tensorlistfromtensor_p", "c64_tensorlistfromtensor_p"], [210, 0, 1, "c.c64_tensorlistfromtensor_s", "c64_tensorlistfromtensor_s"], [215, 0, 1, "c.c64_tile_p", "c64_tile_p"], [215, 0, 1, "c.c64_tile_s", "c64_tile_s"], [217, 0, 1, "c.c64_transpose_p", "c64_transpose_p"], [217, 0, 1, "c.c64_transpose_s", "c64_transpose_s"], [218, 0, 1, "c.c64_tril_p", "c64_tril_p"], [218, 0, 1, "c.c64_tril_s", "c64_tril_s"], [219, 0, 1, "c.c64_triu_p", "c64_triu_p"], [219, 0, 1, "c.c64_triu_s", "c64_triu_s"], [222, 0, 1, "c.c64_unsorted_segment_sum_p", "c64_unsorted_segment_sum_p"], [222, 0, 1, "c.c64_unsorted_segment_sum_s", "c64_unsorted_segment_sum_s"], [225, 0, 1, "c.c64_where_p", "c64_where_p"], [225, 0, 1, "c.c64_where_s", "c64_where_s"], [226, 0, 1, "c.c64_zerolike_p", "c64_zerolike_p"], [226, 0, 1, "c.c64_zerolike_s", "c64_zerolike_s"], [42, 0, 1, "c.castc128Toc128_p", "castc128Toc128_p"], [42, 0, 1, "c.castc128Toc128_s", "castc128Toc128_s"], [42, 0, 1, "c.castc64Toc64_p", "castc64Toc64_p"], [42, 0, 1, "c.castc64Toc64_s", "castc64Toc64_s"], [42, 0, 1, "c.castdpTodp_p", "castdpTodp_p"], [42, 0, 1, "c.castdpTodp_s", "castdpTodp_s"], [42, 0, 1, "c.castdpTofp_p", "castdpTofp_p"], [42, 0, 1, "c.castdpTofp_s", "castdpTofp_s"], [42, 0, 1, "c.castfpTofp_p", "castfpTofp_p"], [42, 0, 1, "c.castfpTofp_s", "castfpTofp_s"], [42, 0, 1, "c.castfpToint16_p", "castfpToint16_p"], [42, 0, 1, "c.castfpToint16_s", "castfpToint16_s"], [42, 0, 1, "c.castfpToint32_p", "castfpToint32_p"], [42, 0, 1, "c.castfpToint32_s", "castfpToint32_s"], [42, 0, 1, "c.castfpToint8_p", "castfpToint8_p"], [42, 0, 1, "c.castfpToint8_s", "castfpToint8_s"], [42, 0, 1, "c.casti16Tofp_p", "casti16Tofp_p"], [42, 0, 1, "c.casti16Tofp_s", "casti16Tofp_s"], [42, 0, 1, "c.casti16Tointi16_p", "casti16Tointi16_p"], [42, 0, 1, "c.casti16Tointi16_s", "casti16Tointi16_s"], [42, 0, 1, "c.casti16Tointi32_p", "casti16Tointi32_p"], [42, 0, 1, "c.casti16Tointi32_s", "casti16Tointi32_s"], [42, 0, 1, "c.casti16Tointi8_p", "casti16Tointi8_p"], [42, 0, 1, "c.casti16Tointi8_s", "casti16Tointi8_s"], [42, 0, 1, "c.casti32Tofp_p", "casti32Tofp_p"], [42, 0, 1, "c.casti32Tofp_s", "casti32Tofp_s"], [42, 0, 1, "c.casti32Tointi16_p", "casti32Tointi16_p"], [42, 0, 1, "c.casti32Tointi16_s", "casti32Tointi16_s"], [42, 0, 1, "c.casti32Tointi32_p", "casti32Tointi32_p"], [42, 0, 1, "c.casti32Tointi32_s", "casti32Tointi32_s"], [42, 0, 1, "c.casti32Tointi8_p", "casti32Tointi8_p"], [42, 0, 1, "c.casti32Tointi8_s", "casti32Tointi8_s"], [42, 0, 1, "c.casti8Tofp_p", "casti8Tofp_p"], [42, 0, 1, "c.casti8Tofp_s", "casti8Tofp_s"], [42, 0, 1, "c.casti8Tointi16_p", "casti8Tointi16_p"], [42, 0, 1, "c.casti8Tointi16_s", "casti8Tointi16_s"], [42, 0, 1, "c.casti8Tointi32_p", "casti8Tointi32_p"], [42, 0, 1, "c.casti8Tointi32_s", "casti8Tointi32_s"], [42, 0, 1, "c.casti8Tointi8_p", "casti8Tointi8_p"], [42, 0, 1, "c.casti8Tointi8_s", "casti8Tointi8_s"], [56, 0, 1, "c.customnormalize_p", "customnormalize_p"], [56, 0, 1, "c.customnormalize_s", "customnormalize_s"], [57, 0, 1, "c.custompredict_p", "custompredict_p"], [57, 0, 1, "c.custompredict_s", "custompredict_s"], [221, 0, 1, "c.dp_Unique_p", "dp_Unique_p"], [221, 0, 1, "c.dp_Unique_s", "dp_Unique_s"], [10, 0, 1, "c.dp_abs_p", "dp_abs_p"], [10, 0, 1, "c.dp_abs_s", "dp_abs_s"], [17, 0, 1, "c.dp_addext_p", "dp_addext_p"], [17, 0, 1, "c.dp_addext_s", "dp_addext_s"], [19, 0, 1, "c.dp_addn_p", "dp_addn_p"], [19, 0, 1, "c.dp_addn_s", "dp_addn_s"], [17, 0, 1, "c.dp_addrelu6_p", "dp_addrelu6_p"], [17, 0, 1, "c.dp_addrelu6_s", "dp_addrelu6_s"], [17, 0, 1, "c.dp_addrelu_p", "dp_addrelu_p"], [17, 0, 1, "c.dp_addrelu_s", "dp_addrelu_s"], [22, 0, 1, "c.dp_allgather_p", "dp_allgather_p"], [22, 0, 1, "c.dp_allgather_s", "dp_allgather_s"], [112, 0, 1, "c.dp_and_p", "dp_and_p"], [112, 0, 1, "c.dp_and_s", "dp_and_s"], [27, 0, 1, "c.dp_assign_p", "dp_assign_p"], [27, 0, 1, "c.dp_assign_s", "dp_assign_s"], [28, 0, 1, "c.dp_assignadd_p", "dp_assignadd_p"], [28, 0, 1, "c.dp_assignadd_s", "dp_assignadd_s"], [35, 0, 1, "c.dp_batchtospace_p", "dp_batchtospace_p"], [35, 0, 1, "c.dp_batchtospace_s", "dp_batchtospace_s"], [36, 0, 1, "c.dp_batchtospacend_p", "dp_batchtospacend_p"], [36, 0, 1, "c.dp_batchtospacend_s", "dp_batchtospacend_s"], [37, 0, 1, "c.dp_biasadd_p", "dp_biasadd_p"], [37, 0, 1, "c.dp_biasadd_s", "dp_biasadd_s"], [41, 0, 1, "c.dp_broadcastto_p", "dp_broadcastto_p"], [41, 0, 1, "c.dp_broadcastto_s", "dp_broadcastto_s"], [43, 0, 1, "c.dp_ceil_p", "dp_ceil_p"], [43, 0, 1, "c.dp_ceil_s", "dp_ceil_s"], [44, 0, 1, "c.dp_clip_p", "dp_clip_p"], [44, 0, 1, "c.dp_clip_s", "dp_clip_s"], [45, 0, 1, "c.dp_concat_p", "dp_concat_p"], [45, 0, 1, "c.dp_concat_s", "dp_concat_s"], [46, 0, 1, "c.dp_constant_of_shape_p", "dp_constant_of_shape_p"], [46, 0, 1, "c.dp_constant_of_shape_s", "dp_constant_of_shape_s"], [51, 0, 1, "c.dp_cos_p", "dp_cos_p"], [51, 0, 1, "c.dp_cos_s", "dp_cos_s"], [54, 0, 1, "c.dp_cumsum_p", "dp_cumsum_p"], [54, 0, 1, "c.dp_cumsum_s", "dp_cumsum_s"], [59, 0, 1, "c.dp_depthtospace_p", "dp_depthtospace_p"], [59, 0, 1, "c.dp_depthtospace_s", "dp_depthtospace_s"], [61, 0, 1, "c.dp_div_fusion_p", "dp_div_fusion_p"], [61, 0, 1, "c.dp_div_fusion_s", "dp_div_fusion_s"], [67, 0, 1, "c.dp_eltwise_p", "dp_eltwise_p"], [67, 0, 1, "c.dp_eltwise_s", "dp_eltwise_s"], [70, 0, 1, "c.dp_equal_p", "dp_equal_p"], [70, 0, 1, "c.dp_equal_s", "dp_equal_s"], [71, 0, 1, "c.dp_erf_p", "dp_erf_p"], [71, 0, 1, "c.dp_erf_s", "dp_erf_s"], [73, 0, 1, "c.dp_expfusion_p", "dp_expfusion_p"], [73, 0, 1, "c.dp_expfusion_s", "dp_expfusion_s"], [55, 0, 1, "c.dp_extract_features_p", "dp_extract_features_p"], [55, 0, 1, "c.dp_extract_features_s", "dp_extract_features_s"], [78, 0, 1, "c.dp_fill_p", "dp_fill_p"], [78, 0, 1, "c.dp_fill_s", "dp_fill_s"], [82, 0, 1, "c.dp_floor_p", "dp_floor_p"], [82, 0, 1, "c.dp_floor_s", "dp_floor_s"], [83, 0, 1, "c.dp_floordiv_p", "dp_floordiv_p"], [83, 0, 1, "c.dp_floordiv_s", "dp_floordiv_s"], [84, 0, 1, "c.dp_floormod_p", "dp_floormod_p"], [84, 0, 1, "c.dp_floormod_s", "dp_floormod_s"], [85, 0, 1, "c.dp_formattranspose_p", "dp_formattranspose_p"], [85, 0, 1, "c.dp_formattranspose_s", "dp_formattranspose_s"], [89, 0, 1, "c.dp_gather_nd_p", "dp_gather_nd_p"], [89, 0, 1, "c.dp_gather_nd_s", "dp_gather_nd_s"], [88, 0, 1, "c.dp_gather_p", "dp_gather_p"], [88, 0, 1, "c.dp_gather_s", "dp_gather_s"], [90, 0, 1, "c.dp_gatherd_p", "dp_gatherd_p"], [90, 0, 1, "c.dp_gatherd_s", "dp_gatherd_s"], [92, 0, 1, "c.dp_greater_s", "dp_greater_s"], [93, 0, 1, "c.dp_greaterequal_s", "dp_greaterequal_s"], [99, 0, 1, "c.dp_isfinite_p", "dp_isfinite_p"], [99, 0, 1, "c.dp_isfinite_s", "dp_isfinite_s"], [104, 0, 1, "c.dp_less_p", "dp_less_p"], [104, 0, 1, "c.dp_less_s", "dp_less_s"], [105, 0, 1, "c.dp_lessequal_p", "dp_lessequal_p"], [105, 0, 1, "c.dp_lessequal_s", "dp_lessequal_s"], [108, 0, 1, "c.dp_log1p_p", "dp_log1p_p"], [108, 0, 1, "c.dp_log1p_s", "dp_log1p_s"], [110, 0, 1, "c.dp_logical_not_p", "dp_logical_not_p"], [110, 0, 1, "c.dp_logical_not_s", "dp_logical_not_s"], [111, 0, 1, "c.dp_logical_or_p", "dp_logical_or_p"], [111, 0, 1, "c.dp_logical_or_s", "dp_logical_or_s"], [116, 0, 1, "c.dp_lsh_projection_p", "dp_lsh_projection_p"], [116, 0, 1, "c.dp_lsh_projection_s", "dp_lsh_projection_s"], [121, 0, 1, "c.dp_matmulfusion_p", "dp_matmulfusion_p"], [121, 0, 1, "c.dp_matmulfusion_s", "dp_matmulfusion_s"], [122, 0, 1, "c.dp_maximum_p", "dp_maximum_p"], [122, 0, 1, "c.dp_maximum_s", "dp_maximum_s"], [124, 0, 1, "c.dp_maxpool_fusion_p", "dp_maxpool_fusion_p"], [124, 0, 1, "c.dp_maxpool_fusion_s", "dp_maxpool_fusion_s"], [125, 0, 1, "c.dp_maxpool_grad_p", "dp_maxpool_grad_p"], [125, 0, 1, "c.dp_maxpool_grad_s", "dp_maxpool_grad_s"], [127, 0, 1, "c.dp_minimum_p", "dp_minimum_p"], [127, 0, 1, "c.dp_minimum_s", "dp_minimum_s"], [129, 0, 1, "c.dp_mod_p", "dp_mod_p"], [129, 0, 1, "c.dp_mod_s", "dp_mod_s"], [130, 0, 1, "c.dp_mul_p", "dp_mul_p"], [130, 0, 1, "c.dp_mul_s", "dp_mul_s"], [133, 0, 1, "c.dp_neg_grad_p", "dp_neg_grad_p"], [133, 0, 1, "c.dp_neg_grad_s", "dp_neg_grad_s"], [132, 0, 1, "c.dp_neg_p", "dp_neg_p"], [132, 0, 1, "c.dp_neg_s", "dp_neg_s"], [137, 0, 1, "c.dp_nonzero_p", "dp_nonzero_p"], [137, 0, 1, "c.dp_nonzero_s", "dp_nonzero_s"], [138, 0, 1, "c.dp_not_equal_p", "dp_not_equal_p"], [138, 0, 1, "c.dp_not_equal_s", "dp_not_equal_s"], [139, 0, 1, "c.dp_onehot_p", "dp_onehot_p"], [139, 0, 1, "c.dp_onehot_s", "dp_onehot_s"], [140, 0, 1, "c.dp_ones_like_p", "dp_ones_like_p"], [140, 0, 1, "c.dp_ones_like_s", "dp_ones_like_s"], [141, 0, 1, "c.dp_padfusion_p", "dp_padfusion_p"], [141, 0, 1, "c.dp_padfusion_s", "dp_padfusion_s"], [142, 0, 1, "c.dp_pow_fusion_p", "dp_pow_fusion_p"], [142, 0, 1, "c.dp_pow_fusion_s", "dp_pow_fusion_s"], [147, 0, 1, "c.dp_raggedrange_p", "dp_raggedrange_p"], [147, 0, 1, "c.dp_raggedrange_s", "dp_raggedrange_s"], [150, 0, 1, "c.dp_range_p", "dp_range_p"], [150, 0, 1, "c.dp_range_s", "dp_range_s"], [152, 0, 1, "c.dp_real_div_p", "dp_real_div_p"], [152, 0, 1, "c.dp_real_div_s", "dp_real_div_s"], [153, 0, 1, "c.dp_reciprocal_p", "dp_reciprocal_p"], [153, 0, 1, "c.dp_reciprocal_s", "dp_reciprocal_s"], [154, 0, 1, "c.dp_reduce_p", "dp_reduce_p"], [154, 0, 1, "c.dp_reduce_s", "dp_reduce_s"], [21, 0, 1, "c.dp_reduceall_p", "dp_reduceall_p"], [21, 0, 1, "c.dp_reduceall_s", "dp_reduceall_s"], [155, 0, 1, "c.dp_reducescatter_p", "dp_reducescatter_p"], [155, 0, 1, "c.dp_reducescatter_s", "dp_reducescatter_s"], [156, 0, 1, "c.dp_reshape_p", "dp_reshape_p"], [156, 0, 1, "c.dp_reshape_s", "dp_reshape_s"], [163, 0, 1, "c.dp_round_p", "dp_round_p"], [163, 0, 1, "c.dp_round_s", "dp_round_s"], [164, 0, 1, "c.dp_rsqrt_p", "dp_rsqrt_p"], [164, 0, 1, "c.dp_rsqrt_s", "dp_rsqrt_s"], [166, 0, 1, "c.dp_scalefusion_p", "dp_scalefusion_p"], [166, 0, 1, "c.dp_scalefusion_s", "dp_scalefusion_s"], [167, 0, 1, "c.dp_scatter_elements_p", "dp_scatter_elements_p"], [167, 0, 1, "c.dp_scatter_elements_s", "dp_scatter_elements_s"], [168, 0, 1, "c.dp_scatter_nd_p", "dp_scatter_nd_p"], [168, 0, 1, "c.dp_scatter_nd_s", "dp_scatter_nd_s"], [169, 0, 1, "c.dp_scatter_nd_update_p", "dp_scatter_nd_update_p"], [169, 0, 1, "c.dp_scatter_nd_update_s", "dp_scatter_nd_update_s"], [170, 0, 1, "c.dp_select_p", "dp_select_p"], [170, 0, 1, "c.dp_select_s", "dp_select_s"], [175, 0, 1, "c.dp_sin_p", "dp_sin_p"], [175, 0, 1, "c.dp_sin_s", "dp_sin_s"], [178, 0, 1, "c.dp_slice_p", "dp_slice_p"], [178, 0, 1, "c.dp_slice_s", "dp_slice_s"], [183, 0, 1, "c.dp_spacetobatch_p", "dp_spacetobatch_p"], [183, 0, 1, "c.dp_spacetobatch_s", "dp_spacetobatch_s"], [184, 0, 1, "c.dp_spacetobatchnd_p", "dp_spacetobatchnd_p"], [184, 0, 1, "c.dp_spacetobatchnd_s", "dp_spacetobatchnd_s"], [185, 0, 1, "c.dp_spacetodepth_p", "dp_spacetodepth_p"], [185, 0, 1, "c.dp_spacetodepth_s", "dp_spacetodepth_s"], [187, 0, 1, "c.dp_sparsefillemptyrows_p", "dp_sparsefillemptyrows_p"], [187, 0, 1, "c.dp_sparsefillemptyrows_s", "dp_sparsefillemptyrows_s"], [189, 0, 1, "c.dp_sparsesegmentsum_p", "dp_sparsesegmentsum_p"], [189, 0, 1, "c.dp_sparsesegmentsum_s", "dp_sparsesegmentsum_s"], [190, 0, 1, "c.dp_sparsetodense_p", "dp_sparsetodense_p"], [190, 0, 1, "c.dp_sparsetodense_s", "dp_sparsetodense_s"], [191, 0, 1, "c.dp_splice_p", "dp_splice_p"], [191, 0, 1, "c.dp_splice_s", "dp_splice_s"], [192, 0, 1, "c.dp_split_p", "dp_split_p"], [192, 0, 1, "c.dp_split_s", "dp_split_s"], [193, 0, 1, "c.dp_split_with_overlap_p", "dp_split_with_overlap_p"], [193, 0, 1, "c.dp_split_with_overlap_s", "dp_split_with_overlap_s"], [194, 0, 1, "c.dp_sqrt_p", "dp_sqrt_p"], [194, 0, 1, "c.dp_sqrt_s", "dp_sqrt_s"], [195, 0, 1, "c.dp_sqrtgrad_p", "dp_sqrtgrad_p"], [195, 0, 1, "c.dp_sqrtgrad_s", "dp_sqrtgrad_s"], [196, 0, 1, "c.dp_square_p", "dp_square_p"], [196, 0, 1, "c.dp_square_s", "dp_square_s"], [197, 0, 1, "c.dp_squaredifference_p", "dp_squaredifference_p"], [197, 0, 1, "c.dp_squaredifference_s", "dp_squaredifference_s"], [199, 0, 1, "c.dp_stack_p", "dp_stack_p"], [199, 0, 1, "c.dp_stack_s", "dp_stack_s"], [202, 0, 1, "c.dp_subext_p", "dp_subext_p"], [202, 0, 1, "c.dp_subext_s", "dp_subext_s"], [202, 0, 1, "c.dp_subrelu6_p", "dp_subrelu6_p"], [202, 0, 1, "c.dp_subrelu6_s", "dp_subrelu6_s"], [202, 0, 1, "c.dp_subrelu_p", "dp_subrelu_p"], [202, 0, 1, "c.dp_subrelu_s", "dp_subrelu_s"], [206, 0, 1, "c.dp_tensor_scatter_add_p", "dp_tensor_scatter_add_p"], [206, 0, 1, "c.dp_tensor_scatter_add_s", "dp_tensor_scatter_add_s"], [208, 0, 1, "c.dp_tensorarrayread_p", "dp_tensorarrayread_p"], [208, 0, 1, "c.dp_tensorarrayread_s", "dp_tensorarrayread_s"], [210, 0, 1, "c.dp_tensorlistfromtensor_p", "dp_tensorlistfromtensor_p"], [210, 0, 1, "c.dp_tensorlistfromtensor_s", "dp_tensorlistfromtensor_s"], [215, 0, 1, "c.dp_tile_p", "dp_tile_p"], [215, 0, 1, "c.dp_tile_s", "dp_tile_s"], [217, 0, 1, "c.dp_transpose_p", "dp_transpose_p"], [217, 0, 1, "c.dp_transpose_s", "dp_transpose_s"], [218, 0, 1, "c.dp_tril_p", "dp_tril_p"], [218, 0, 1, "c.dp_tril_s", "dp_tril_s"], [219, 0, 1, "c.dp_triu_p", "dp_triu_p"], [219, 0, 1, "c.dp_triu_s", "dp_triu_s"], [222, 0, 1, "c.dp_unsorted_segment_sum_p", "dp_unsorted_segment_sum_p"], [222, 0, 1, "c.dp_unsorted_segment_sum_s", "dp_unsorted_segment_sum_s"], [225, 0, 1, "c.dp_where_p", "dp_where_p"], [225, 0, 1, "c.dp_where_s", "dp_where_s"], [226, 0, 1, "c.dp_zerolike_p", "dp_zerolike_p"], [226, 0, 1, "c.dp_zerolike_s", "dp_zerolike_s"], [95, 0, 1, "c.fp_Gru_p", "fp_Gru_p"], [95, 0, 1, "c.fp_Gru_s", "fp_Gru_s"], [117, 0, 1, "c.fp_Lstm_p", "fp_Lstm_p"], [117, 0, 1, "c.fp_Lstm_s", "fp_Lstm_s"], [66, 0, 1, "c.fp_QuantData_p", "fp_QuantData_p"], [66, 0, 1, "c.fp_QuantData_s", "fp_QuantData_s"], [221, 0, 1, "c.fp_Unique_p", "fp_Unique_p"], [221, 0, 1, "c.fp_Unique_s", "fp_Unique_s"], [10, 0, 1, "c.fp_abs_p", "fp_abs_p"], [10, 0, 1, "c.fp_abs_s", "fp_abs_s"], [11, 0, 1, "c.fp_absgrad_p", "fp_absgrad_p"], [11, 0, 1, "c.fp_absgrad_s", "fp_absgrad_s"], [14, 0, 1, "c.fp_adam_p", "fp_adam_p"], [14, 0, 1, "c.fp_adam_s", "fp_adam_s"], [15, 0, 1, "c.fp_adamweightdecay_p", "fp_adamweightdecay_p"], [15, 0, 1, "c.fp_adamweightdecay_s", "fp_adamweightdecay_s"], [16, 0, 1, "c.fp_adder_p", "fp_adder_p"], [16, 0, 1, "c.fp_adder_s", "fp_adder_s"], [17, 0, 1, "c.fp_addext_p", "fp_addext_p"], [17, 0, 1, "c.fp_addext_s", "fp_addext_s"], [18, 0, 1, "c.fp_addgrad_p", "fp_addgrad_p"], [18, 0, 1, "c.fp_addgrad_s", "fp_addgrad_s"], [19, 0, 1, "c.fp_addn_p", "fp_addn_p"], [19, 0, 1, "c.fp_addn_s", "fp_addn_s"], [17, 0, 1, "c.fp_addrelu6_p", "fp_addrelu6_p"], [17, 0, 1, "c.fp_addrelu6_s", "fp_addrelu6_s"], [17, 0, 1, "c.fp_addrelu_p", "fp_addrelu_p"], [17, 0, 1, "c.fp_addrelu_s", "fp_addrelu_s"], [20, 0, 1, "c.fp_affine_p", "fp_affine_p"], [20, 0, 1, "c.fp_affine_s", "fp_affine_s"], [22, 0, 1, "c.fp_allgather_p", "fp_allgather_p"], [22, 0, 1, "c.fp_allgather_s", "fp_allgather_s"], [112, 0, 1, "c.fp_and_p", "fp_and_p"], [112, 0, 1, "c.fp_and_s", "fp_and_s"], [23, 0, 1, "c.fp_applymomentum_p", "fp_applymomentum_p"], [23, 0, 1, "c.fp_applymomentum_s", "fp_applymomentum_s"], [24, 0, 1, "c.fp_argmax_p", "fp_argmax_p"], [24, 0, 1, "c.fp_argmax_s", "fp_argmax_s"], [25, 0, 1, "c.fp_argmin_p", "fp_argmin_p"], [25, 0, 1, "c.fp_argmin_s", "fp_argmin_s"], [27, 0, 1, "c.fp_assign_p", "fp_assign_p"], [27, 0, 1, "c.fp_assign_s", "fp_assign_s"], [28, 0, 1, "c.fp_assignadd_p", "fp_assignadd_p"], [28, 0, 1, "c.fp_assignadd_s", "fp_assignadd_s"], [29, 0, 1, "c.fp_attention_p", "fp_attention_p"], [29, 0, 1, "c.fp_attention_s", "fp_attention_s"], [30, 0, 1, "c.fp_audio_spectrogram_p", "fp_audio_spectrogram_p"], [30, 0, 1, "c.fp_audio_spectrogram_s", "fp_audio_spectrogram_s"], [31, 0, 1, "c.fp_avgpool_fusion_p", "fp_avgpool_fusion_p"], [31, 0, 1, "c.fp_avgpool_fusion_s", "fp_avgpool_fusion_s"], [32, 0, 1, "c.fp_avgpoolinggrad_p", "fp_avgpoolinggrad_p"], [32, 0, 1, "c.fp_avgpoolinggrad_s", "fp_avgpoolinggrad_s"], [33, 0, 1, "c.fp_batchnorm_p", "fp_batchnorm_p"], [33, 0, 1, "c.fp_batchnorm_s", "fp_batchnorm_s"], [34, 0, 1, "c.fp_batchnormgrad_p", "fp_batchnormgrad_p"], [34, 0, 1, "c.fp_batchnormgrad_s", "fp_batchnormgrad_s"], [35, 0, 1, "c.fp_batchtospace_p", "fp_batchtospace_p"], [35, 0, 1, "c.fp_batchtospace_s", "fp_batchtospace_s"], [36, 0, 1, "c.fp_batchtospacend_p", "fp_batchtospacend_p"], [36, 0, 1, "c.fp_batchtospacend_s", "fp_batchtospacend_s"], [37, 0, 1, "c.fp_biasadd_p", "fp_biasadd_p"], [37, 0, 1, "c.fp_biasadd_s", "fp_biasadd_s"], [38, 0, 1, "c.fp_biasaddgrad_p", "fp_biasaddgrad_p"], [38, 0, 1, "c.fp_biasaddgrad_s", "fp_biasaddgrad_s"], [39, 0, 1, "c.fp_binarycrossentropy_p", "fp_binarycrossentropy_p"], [39, 0, 1, "c.fp_binarycrossentropy_s", "fp_binarycrossentropy_s"], [40, 0, 1, "c.fp_binarycrossentropygrad_p", "fp_binarycrossentropygrad_p"], [40, 0, 1, "c.fp_binarycrossentropygrad_s", "fp_binarycrossentropygrad_s"], [41, 0, 1, "c.fp_broadcastto_p", "fp_broadcastto_p"], [41, 0, 1, "c.fp_broadcastto_s", "fp_broadcastto_s"], [43, 0, 1, "c.fp_ceil_p", "fp_ceil_p"], [43, 0, 1, "c.fp_ceil_s", "fp_ceil_s"], [12, 0, 1, "c.fp_celu_p", "fp_celu_p"], [12, 0, 1, "c.fp_celu_s", "fp_celu_s"], [12, 0, 1, "c.fp_clip_p", "fp_clip_p"], [12, 0, 1, "c.fp_clip_s", "fp_clip_s"], [45, 0, 1, "c.fp_concat_p", "fp_concat_p"], [45, 0, 1, "c.fp_concat_s", "fp_concat_s"], [46, 0, 1, "c.fp_constant_of_shape_p", "fp_constant_of_shape_p"], [46, 0, 1, "c.fp_constant_of_shape_s", "fp_constant_of_shape_s"], [47, 0, 1, "c.fp_conv2d_p", "fp_conv2d_p"], [47, 0, 1, "c.fp_conv2d_s", "fp_conv2d_s"], [49, 0, 1, "c.fp_conv2dbackpropfilterfusion_p", "fp_conv2dbackpropfilterfusion_p"], [49, 0, 1, "c.fp_conv2dbackpropfilterfusion_s", "fp_conv2dbackpropfilterfusion_s"], [50, 0, 1, "c.fp_conv2dbackpropinputfusion_p", "fp_conv2dbackpropinputfusion_p"], [50, 0, 1, "c.fp_conv2dbackpropinputfusion_s", "fp_conv2dbackpropinputfusion_s"], [48, 0, 1, "c.fp_convtranspose_p", "fp_convtranspose_p"], [48, 0, 1, "c.fp_convtranspose_s", "fp_convtranspose_s"], [51, 0, 1, "c.fp_cos_p", "fp_cos_p"], [51, 0, 1, "c.fp_cos_s", "fp_cos_s"], [53, 0, 1, "c.fp_crop_and_resize_anycore", "fp_crop_and_resize_anycore"], [54, 0, 1, "c.fp_cumsum_p", "fp_cumsum_p"], [54, 0, 1, "c.fp_cumsum_s", "fp_cumsum_s"], [58, 0, 1, "c.fp_deconvgradfilter_p", "fp_deconvgradfilter_p"], [58, 0, 1, "c.fp_deconvgradfilter_s", "fp_deconvgradfilter_s"], [59, 0, 1, "c.fp_depthtospace_p", "fp_depthtospace_p"], [59, 0, 1, "c.fp_depthtospace_s", "fp_depthtospace_s"], [60, 0, 1, "c.fp_detection_post_process_p", "fp_detection_post_process_p"], [60, 0, 1, "c.fp_detection_post_process_s", "fp_detection_post_process_s"], [61, 0, 1, "c.fp_div_fusion_p", "fp_div_fusion_p"], [61, 0, 1, "c.fp_div_fusion_s", "fp_div_fusion_s"], [63, 0, 1, "c.fp_dropout_p", "fp_dropout_p"], [63, 0, 1, "c.fp_dropout_s", "fp_dropout_s"], [64, 0, 1, "c.fp_dropoutgrad_p", "fp_dropoutgrad_p"], [64, 0, 1, "c.fp_dropoutgrad_s", "fp_dropoutgrad_s"], [67, 0, 1, "c.fp_eltwise_p", "fp_eltwise_p"], [67, 0, 1, "c.fp_eltwise_s", "fp_eltwise_s"], [13, 0, 1, "c.fp_elu_grad_p", "fp_elu_grad_p"], [13, 0, 1, "c.fp_elu_grad_s", "fp_elu_grad_s"], [12, 0, 1, "c.fp_elu_p", "fp_elu_p"], [12, 0, 1, "c.fp_elu_s", "fp_elu_s"], [69, 0, 1, "c.fp_embeddinglookup_p", "fp_embeddinglookup_p"], [69, 0, 1, "c.fp_embeddinglookup_s", "fp_embeddinglookup_s"], [70, 0, 1, "c.fp_equal_p", "fp_equal_p"], [70, 0, 1, "c.fp_equal_s", "fp_equal_s"], [71, 0, 1, "c.fp_erf_p", "fp_erf_p"], [71, 0, 1, "c.fp_erf_s", "fp_erf_s"], [73, 0, 1, "c.fp_expfusion_p", "fp_expfusion_p"], [73, 0, 1, "c.fp_expfusion_s", "fp_expfusion_s"], [55, 0, 1, "c.fp_extract_features_p", "fp_extract_features_p"], [55, 0, 1, "c.fp_extract_features_s", "fp_extract_features_s"], [74, 0, 1, "c.fp_fake_quant_with_min_max_vars_p", "fp_fake_quant_with_min_max_vars_p"], [75, 0, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "fp_fake_quant_with_min_max_vars_per_channel_p"], [75, 0, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "fp_fake_quant_with_min_max_vars_per_channel_s"], [74, 0, 1, "c.fp_fake_quant_with_min_max_vars_s", "fp_fake_quant_with_min_max_vars_s"], [78, 0, 1, "c.fp_fill_p", "fp_fill_p"], [78, 0, 1, "c.fp_fill_s", "fp_fill_s"], [81, 0, 1, "c.fp_flattengrad_p", "fp_flattengrad_p"], [81, 0, 1, "c.fp_flattengrad_s", "fp_flattengrad_s"], [82, 0, 1, "c.fp_floor_p", "fp_floor_p"], [82, 0, 1, "c.fp_floor_s", "fp_floor_s"], [83, 0, 1, "c.fp_floordiv_p", "fp_floordiv_p"], [83, 0, 1, "c.fp_floordiv_s", "fp_floordiv_s"], [84, 0, 1, "c.fp_floormod_p", "fp_floormod_p"], [84, 0, 1, "c.fp_floormod_s", "fp_floormod_s"], [85, 0, 1, "c.fp_formattranspose_p", "fp_formattranspose_p"], [85, 0, 1, "c.fp_formattranspose_s", "fp_formattranspose_s"], [86, 0, 1, "c.fp_fullconnection_p", "fp_fullconnection_p"], [86, 0, 1, "c.fp_fullconnection_s", "fp_fullconnection_s"], [87, 0, 1, "c.fp_fusedbatchnorm_p", "fp_fusedbatchnorm_p"], [87, 0, 1, "c.fp_fusedbatchnorm_s", "fp_fusedbatchnorm_s"], [89, 0, 1, "c.fp_gather_nd_p", "fp_gather_nd_p"], [89, 0, 1, "c.fp_gather_nd_s", "fp_gather_nd_s"], [88, 0, 1, "c.fp_gather_p", "fp_gather_p"], [88, 0, 1, "c.fp_gather_s", "fp_gather_s"], [90, 0, 1, "c.fp_gatherd_p", "fp_gatherd_p"], [90, 0, 1, "c.fp_gatherd_s", "fp_gatherd_s"], [13, 0, 1, "c.fp_gelu_grad_p", "fp_gelu_grad_p"], [13, 0, 1, "c.fp_gelu_grad_s", "fp_gelu_grad_s"], [12, 0, 1, "c.fp_gelu_p", "fp_gelu_p"], [12, 0, 1, "c.fp_gelu_s", "fp_gelu_s"], [91, 0, 1, "c.fp_glu_p", "fp_glu_p"], [91, 0, 1, "c.fp_glu_s", "fp_glu_s"], [62, 0, 1, "c.fp_graddiv1l_p", "fp_graddiv1l_p"], [62, 0, 1, "c.fp_graddiv1l_s", "fp_graddiv1l_s"], [62, 0, 1, "c.fp_graddiv2l_p", "fp_graddiv2l_p"], [62, 0, 1, "c.fp_graddiv2l_s", "fp_graddiv2l_s"], [62, 0, 1, "c.fp_graddiv_p", "fp_graddiv_p"], [62, 0, 1, "c.fp_graddiv_s", "fp_graddiv_s"], [131, 0, 1, "c.fp_gradmul1l_p", "fp_gradmul1l_p"], [131, 0, 1, "c.fp_gradmul1l_s", "fp_gradmul1l_s"], [131, 0, 1, "c.fp_gradmul2l_p", "fp_gradmul2l_p"], [131, 0, 1, "c.fp_gradmul2l_s", "fp_gradmul2l_s"], [131, 0, 1, "c.fp_gradmul_p", "fp_gradmul_p"], [131, 0, 1, "c.fp_gradmul_s", "fp_gradmul_s"], [92, 0, 1, "c.fp_greater_p", "fp_greater_p"], [92, 0, 1, "c.fp_greater_s", "fp_greater_s"], [93, 0, 1, "c.fp_greaterequal_p", "fp_greaterequal_p"], [93, 0, 1, "c.fp_greaterequal_s", "fp_greaterequal_s"], [94, 0, 1, "c.fp_groupnormfusion_p", "fp_groupnormfusion_p"], [94, 0, 1, "c.fp_groupnormfusion_s", "fp_groupnormfusion_s"], [13, 0, 1, "c.fp_h_sigmoid_grad_p", "fp_h_sigmoid_grad_p"], [13, 0, 1, "c.fp_h_sigmoid_grad_s", "fp_h_sigmoid_grad_s"], [13, 0, 1, "c.fp_h_swish_grad_p", "fp_h_swish_grad_p"], [13, 0, 1, "c.fp_h_swish_grad_s", "fp_h_swish_grad_s"], [13, 0, 1, "c.fp_hard_shrink_grad_p", "fp_hard_shrink_grad_p"], [13, 0, 1, "c.fp_hard_shrink_grad_s", "fp_hard_shrink_grad_s"], [12, 0, 1, "c.fp_hardshrink_p", "fp_hardshrink_p"], [12, 0, 1, "c.fp_hardshrink_s", "fp_hardshrink_s"], [12, 0, 1, "c.fp_hardtanh_p", "fp_hardtanh_p"], [12, 0, 1, "c.fp_hardtanh_s", "fp_hardtanh_s"], [12, 0, 1, "c.fp_hsigmoid_p", "fp_hsigmoid_p"], [12, 0, 1, "c.fp_hsigmoid_s", "fp_hsigmoid_s"], [12, 0, 1, "c.fp_hswish_p", "fp_hswish_p"], [12, 0, 1, "c.fp_hswish_s", "fp_hswish_s"], [97, 0, 1, "c.fp_instancenorm_p", "fp_instancenorm_p"], [97, 0, 1, "c.fp_instancenorm_s", "fp_instancenorm_s"], [99, 0, 1, "c.fp_isfinite_p", "fp_isfinite_p"], [99, 0, 1, "c.fp_isfinite_s", "fp_isfinite_s"], [100, 0, 1, "c.fp_l2norm_p", "fp_l2norm_p"], [100, 0, 1, "c.fp_l2norm_s", "fp_l2norm_s"], [13, 0, 1, "c.fp_l_relu_grad_p", "fp_l_relu_grad_p"], [13, 0, 1, "c.fp_l_relu_grad_s", "fp_l_relu_grad_s"], [101, 0, 1, "c.fp_layernormfusion_p", "fp_layernormfusion_p"], [101, 0, 1, "c.fp_layernormfusion_s", "fp_layernormfusion_s"], [102, 0, 1, "c.fp_layernormgrad_p", "fp_layernormgrad_p"], [102, 0, 1, "c.fp_layernormgrad_s", "fp_layernormgrad_s"], [103, 0, 1, "c.fp_leaky_relu_p", "fp_leaky_relu_p"], [103, 0, 1, "c.fp_leaky_relu_s", "fp_leaky_relu_s"], [104, 0, 1, "c.fp_less_p", "fp_less_p"], [104, 0, 1, "c.fp_less_s", "fp_less_s"], [105, 0, 1, "c.fp_lessequal_p", "fp_lessequal_p"], [105, 0, 1, "c.fp_lessequal_s", "fp_lessequal_s"], [106, 0, 1, "c.fp_linspace_p", "fp_linspace_p"], [106, 0, 1, "c.fp_linspace_s", "fp_linspace_s"], [108, 0, 1, "c.fp_log1p_p", "fp_log1p_p"], [108, 0, 1, "c.fp_log1p_s", "fp_log1p_s"], [109, 0, 1, "c.fp_log_grad_p", "fp_log_grad_p"], [109, 0, 1, "c.fp_log_grad_s", "fp_log_grad_s"], [107, 0, 1, "c.fp_log_p", "fp_log_p"], [107, 0, 1, "c.fp_log_s", "fp_log_s"], [110, 0, 1, "c.fp_logical_not_p", "fp_logical_not_p"], [110, 0, 1, "c.fp_logical_not_s", "fp_logical_not_s"], [111, 0, 1, "c.fp_logical_or_p", "fp_logical_or_p"], [111, 0, 1, "c.fp_logical_or_s", "fp_logical_or_s"], [113, 0, 1, "c.fp_logsoftmax_p", "fp_logsoftmax_p"], [113, 0, 1, "c.fp_logsoftmax_s", "fp_logsoftmax_s"], [114, 0, 1, "c.fp_lpnorm_p", "fp_lpnorm_p"], [114, 0, 1, "c.fp_lpnorm_s", "fp_lpnorm_s"], [12, 0, 1, "c.fp_lrelu_p", "fp_lrelu_p"], [12, 0, 1, "c.fp_lrelu_s", "fp_lrelu_s"], [115, 0, 1, "c.fp_lrn_p", "fp_lrn_p"], [115, 0, 1, "c.fp_lrn_s", "fp_lrn_s"], [116, 0, 1, "c.fp_lsh_projection_p", "fp_lsh_projection_p"], [116, 0, 1, "c.fp_lsh_projection_s", "fp_lsh_projection_s"], [118, 0, 1, "c.fp_lstmgrad_p", "fp_lstmgrad_p"], [118, 0, 1, "c.fp_lstmgrad_s", "fp_lstmgrad_s"], [119, 0, 1, "c.fp_lstmgraddata_p", "fp_lstmgraddata_p"], [119, 0, 1, "c.fp_lstmgraddata_s", "fp_lstmgraddata_s"], [120, 0, 1, "c.fp_lstmgradweight_p", "fp_lstmgradweight_p"], [120, 0, 1, "c.fp_lstmgradweight_s", "fp_lstmgradweight_s"], [121, 0, 1, "c.fp_matmulfusion_p", "fp_matmulfusion_p"], [121, 0, 1, "c.fp_matmulfusion_s", "fp_matmulfusion_s"], [122, 0, 1, "c.fp_maximum_p", "fp_maximum_p"], [122, 0, 1, "c.fp_maximum_s", "fp_maximum_s"], [123, 0, 1, "c.fp_maximumgrad_p", "fp_maximumgrad_p"], [123, 0, 1, "c.fp_maximumgrad_s", "fp_maximumgrad_s"], [124, 0, 1, "c.fp_maxpool_fusion_p", "fp_maxpool_fusion_p"], [124, 0, 1, "c.fp_maxpool_fusion_s", "fp_maxpool_fusion_s"], [125, 0, 1, "c.fp_maxpool_grad_p", "fp_maxpool_grad_p"], [125, 0, 1, "c.fp_maxpool_grad_s", "fp_maxpool_grad_s"], [126, 0, 1, "c.fp_mfcc_p", "fp_mfcc_p"], [126, 0, 1, "c.fp_mfcc_s", "fp_mfcc_s"], [127, 0, 1, "c.fp_minimum_p", "fp_minimum_p"], [127, 0, 1, "c.fp_minimum_s", "fp_minimum_s"], [128, 0, 1, "c.fp_minimumgrad_p", "fp_minimumgrad_p"], [128, 0, 1, "c.fp_minimumgrad_s", "fp_minimumgrad_s"], [129, 0, 1, "c.fp_mod_p", "fp_mod_p"], [129, 0, 1, "c.fp_mod_s", "fp_mod_s"], [130, 0, 1, "c.fp_mul_p", "fp_mul_p"], [130, 0, 1, "c.fp_mul_s", "fp_mul_s"], [133, 0, 1, "c.fp_neg_grad_p", "fp_neg_grad_p"], [133, 0, 1, "c.fp_neg_grad_s", "fp_neg_grad_s"], [132, 0, 1, "c.fp_neg_p", "fp_neg_p"], [132, 0, 1, "c.fp_neg_s", "fp_neg_s"], [134, 0, 1, "c.fp_nllloss_p", "fp_nllloss_p"], [134, 0, 1, "c.fp_nllloss_s", "fp_nllloss_s"], [135, 0, 1, "c.fp_nlllossgrad_p", "fp_nlllossgrad_p"], [135, 0, 1, "c.fp_nlllossgrad_s", "fp_nlllossgrad_s"], [136, 0, 1, "c.fp_non_max_suppression_p", "fp_non_max_suppression_p"], [136, 0, 1, "c.fp_non_max_suppression_s", "fp_non_max_suppression_s"], [137, 0, 1, "c.fp_nonzero_p", "fp_nonzero_p"], [137, 0, 1, "c.fp_nonzero_s", "fp_nonzero_s"], [138, 0, 1, "c.fp_not_equal_p", "fp_not_equal_p"], [138, 0, 1, "c.fp_not_equal_s", "fp_not_equal_s"], [139, 0, 1, "c.fp_onehot_p", "fp_onehot_p"], [139, 0, 1, "c.fp_onehot_s", "fp_onehot_s"], [140, 0, 1, "c.fp_ones_like_p", "fp_ones_like_p"], [140, 0, 1, "c.fp_ones_like_s", "fp_ones_like_s"], [141, 0, 1, "c.fp_padfusion_p", "fp_padfusion_p"], [141, 0, 1, "c.fp_padfusion_s", "fp_padfusion_s"], [142, 0, 1, "c.fp_pow_fusion_p", "fp_pow_fusion_p"], [142, 0, 1, "c.fp_pow_fusion_s", "fp_pow_fusion_s"], [143, 0, 1, "c.fp_power_grad_p", "fp_power_grad_p"], [143, 0, 1, "c.fp_power_grad_s", "fp_power_grad_s"], [144, 0, 1, "c.fp_prelufusion_p", "fp_prelufusion_p"], [144, 0, 1, "c.fp_prelufusion_s", "fp_prelufusion_s"], [145, 0, 1, "c.fp_priorbox_p", "fp_priorbox_p"], [145, 0, 1, "c.fp_priorbox_s", "fp_priorbox_s"], [147, 0, 1, "c.fp_raggedrange_p", "fp_raggedrange_p"], [147, 0, 1, "c.fp_raggedrange_s", "fp_raggedrange_s"], [148, 0, 1, "c.fp_random_normal_p", "fp_random_normal_p"], [148, 0, 1, "c.fp_random_normal_s", "fp_random_normal_s"], [149, 0, 1, "c.fp_random_standard_normal_p", "fp_random_standard_normal_p"], [149, 0, 1, "c.fp_random_standard_normal_s", "fp_random_standard_normal_s"], [150, 0, 1, "c.fp_range_p", "fp_range_p"], [150, 0, 1, "c.fp_range_s", "fp_range_s"], [152, 0, 1, "c.fp_real_div_p", "fp_real_div_p"], [152, 0, 1, "c.fp_real_div_s", "fp_real_div_s"], [153, 0, 1, "c.fp_reciprocal_p", "fp_reciprocal_p"], [153, 0, 1, "c.fp_reciprocal_s", "fp_reciprocal_s"], [154, 0, 1, "c.fp_reduce_p", "fp_reduce_p"], [154, 0, 1, "c.fp_reduce_s", "fp_reduce_s"], [21, 0, 1, "c.fp_reduceall_p", "fp_reduceall_p"], [21, 0, 1, "c.fp_reduceall_s", "fp_reduceall_s"], [155, 0, 1, "c.fp_reducescatter_p", "fp_reducescatter_p"], [155, 0, 1, "c.fp_reducescatter_s", "fp_reducescatter_s"], [13, 0, 1, "c.fp_relu6_grad_p", "fp_relu6_grad_p"], [13, 0, 1, "c.fp_relu6_grad_s", "fp_relu6_grad_s"], [12, 0, 1, "c.fp_relu6_p", "fp_relu6_p"], [12, 0, 1, "c.fp_relu6_s", "fp_relu6_s"], [13, 0, 1, "c.fp_relu_grad_p", "fp_relu_grad_p"], [13, 0, 1, "c.fp_relu_grad_s", "fp_relu_grad_s"], [12, 0, 1, "c.fp_relu_p", "fp_relu_p"], [12, 0, 1, "c.fp_relu_s", "fp_relu_s"], [156, 0, 1, "c.fp_reshape_p", "fp_reshape_p"], [156, 0, 1, "c.fp_reshape_s", "fp_reshape_s"], [157, 0, 1, "c.fp_resize_anycore", "fp_resize_anycore"], [158, 0, 1, "c.fp_resizebilineargrad_p", "fp_resizebilineargrad_p"], [158, 0, 1, "c.fp_resizebilineargrad_s", "fp_resizebilineargrad_s"], [158, 0, 1, "c.fp_resizenearestneighborgrad_p", "fp_resizenearestneighborgrad_p"], [158, 0, 1, "c.fp_resizenearestneighborgrad_s", "fp_resizenearestneighborgrad_s"], [162, 0, 1, "c.fp_roipooling_p", "fp_roipooling_p"], [162, 0, 1, "c.fp_roipooling_s", "fp_roipooling_s"], [163, 0, 1, "c.fp_round_p", "fp_round_p"], [163, 0, 1, "c.fp_round_s", "fp_round_s"], [164, 0, 1, "c.fp_rsqrt_p", "fp_rsqrt_p"], [164, 0, 1, "c.fp_rsqrt_s", "fp_rsqrt_s"], [165, 0, 1, "c.fp_rsqrtgrad_p", "fp_rsqrtgrad_p"], [165, 0, 1, "c.fp_rsqrtgrad_s", "fp_rsqrtgrad_s"], [166, 0, 1, "c.fp_scalefusion_p", "fp_scalefusion_p"], [166, 0, 1, "c.fp_scalefusion_s", "fp_scalefusion_s"], [167, 0, 1, "c.fp_scatter_elements_p", "fp_scatter_elements_p"], [167, 0, 1, "c.fp_scatter_elements_s", "fp_scatter_elements_s"], [168, 0, 1, "c.fp_scatter_nd_p", "fp_scatter_nd_p"], [168, 0, 1, "c.fp_scatter_nd_s", "fp_scatter_nd_s"], [169, 0, 1, "c.fp_scatter_nd_update_p", "fp_scatter_nd_update_p"], [169, 0, 1, "c.fp_scatter_nd_update_s", "fp_scatter_nd_update_s"], [170, 0, 1, "c.fp_select_p", "fp_select_p"], [170, 0, 1, "c.fp_select_s", "fp_select_s"], [171, 0, 1, "c.fp_sgd_p", "fp_sgd_p"], [171, 0, 1, "c.fp_sgd_s", "fp_sgd_s"], [13, 0, 1, "c.fp_sigmoid_grad_p", "fp_sigmoid_grad_p"], [13, 0, 1, "c.fp_sigmoid_grad_s", "fp_sigmoid_grad_s"], [12, 0, 1, "c.fp_sigmoid_p", "fp_sigmoid_p"], [12, 0, 1, "c.fp_sigmoid_s", "fp_sigmoid_s"], [174, 0, 1, "c.fp_sigmoidcrossentropywithlogits_p", "fp_sigmoidcrossentropywithlogits_p"], [174, 0, 1, "c.fp_sigmoidcrossentropywithlogits_s", "fp_sigmoidcrossentropywithlogits_s"], [173, 0, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_p", "fp_sigmoidcrossentropywithlogitsgrad_p"], [173, 0, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_s", "fp_sigmoidcrossentropywithlogitsgrad_s"], [175, 0, 1, "c.fp_sin_p", "fp_sin_p"], [175, 0, 1, "c.fp_sin_s", "fp_sin_s"], [178, 0, 1, "c.fp_slice_p", "fp_slice_p"], [178, 0, 1, "c.fp_slice_s", "fp_slice_s"], [179, 0, 1, "c.fp_smoothl1loss_p", "fp_smoothl1loss_p"], [179, 0, 1, "c.fp_smoothl1loss_s", "fp_smoothl1loss_s"], [180, 0, 1, "c.fp_smoothl1lossgrad_p", "fp_smoothl1lossgrad_p"], [180, 0, 1, "c.fp_smoothl1lossgrad_s", "fp_smoothl1lossgrad_s"], [182, 0, 1, "c.fp_softmax_cross_entropy_with_logits_p", "fp_softmax_cross_entropy_with_logits_p"], [182, 0, 1, "c.fp_softmax_cross_entropy_with_logits_s", "fp_softmax_cross_entropy_with_logits_s"], [181, 0, 1, "c.fp_softmax_p", "fp_softmax_p"], [181, 0, 1, "c.fp_softmax_s", "fp_softmax_s"], [13, 0, 1, "c.fp_softplus_grad_p", "fp_softplus_grad_p"], [13, 0, 1, "c.fp_softplus_grad_s", "fp_softplus_grad_s"], [12, 0, 1, "c.fp_softplus_p", "fp_softplus_p"], [12, 0, 1, "c.fp_softplus_s", "fp_softplus_s"], [13, 0, 1, "c.fp_softshrink_grad_p", "fp_softshrink_grad_p"], [13, 0, 1, "c.fp_softshrink_grad_s", "fp_softshrink_grad_s"], [12, 0, 1, "c.fp_softshrink_p", "fp_softshrink_p"], [12, 0, 1, "c.fp_softshrink_s", "fp_softshrink_s"], [12, 0, 1, "c.fp_softsignopt_p", "fp_softsignopt_p"], [12, 0, 1, "c.fp_softsignopt_s", "fp_softsignopt_s"], [183, 0, 1, "c.fp_spacetobatch_p", "fp_spacetobatch_p"], [183, 0, 1, "c.fp_spacetobatch_s", "fp_spacetobatch_s"], [184, 0, 1, "c.fp_spacetobatchnd_p", "fp_spacetobatchnd_p"], [184, 0, 1, "c.fp_spacetobatchnd_s", "fp_spacetobatchnd_s"], [185, 0, 1, "c.fp_spacetodepth_p", "fp_spacetodepth_p"], [185, 0, 1, "c.fp_spacetodepth_s", "fp_spacetodepth_s"], [186, 0, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "fp_sparse_softmax_cross_entropy_with_logits_p"], [186, 0, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "fp_sparse_softmax_cross_entropy_with_logits_s"], [187, 0, 1, "c.fp_sparsefillemptyrows_p", "fp_sparsefillemptyrows_p"], [187, 0, 1, "c.fp_sparsefillemptyrows_s", "fp_sparsefillemptyrows_s"], [189, 0, 1, "c.fp_sparsesegmentsum_p", "fp_sparsesegmentsum_p"], [189, 0, 1, "c.fp_sparsesegmentsum_s", "fp_sparsesegmentsum_s"], [190, 0, 1, "c.fp_sparsetodense_p", "fp_sparsetodense_p"], [190, 0, 1, "c.fp_sparsetodense_s", "fp_sparsetodense_s"], [191, 0, 1, "c.fp_splice_p", "fp_splice_p"], [191, 0, 1, "c.fp_splice_s", "fp_splice_s"], [192, 0, 1, "c.fp_split_p", "fp_split_p"], [192, 0, 1, "c.fp_split_s", "fp_split_s"], [193, 0, 1, "c.fp_split_with_overlap_p", "fp_split_with_overlap_p"], [193, 0, 1, "c.fp_split_with_overlap_s", "fp_split_with_overlap_s"], [194, 0, 1, "c.fp_sqrt_p", "fp_sqrt_p"], [194, 0, 1, "c.fp_sqrt_s", "fp_sqrt_s"], [195, 0, 1, "c.fp_sqrtgrad_p", "fp_sqrtgrad_p"], [195, 0, 1, "c.fp_sqrtgrad_s", "fp_sqrtgrad_s"], [196, 0, 1, "c.fp_square_p", "fp_square_p"], [196, 0, 1, "c.fp_square_s", "fp_square_s"], [197, 0, 1, "c.fp_squaredifference_p", "fp_squaredifference_p"], [197, 0, 1, "c.fp_squaredifference_s", "fp_squaredifference_s"], [199, 0, 1, "c.fp_stack_p", "fp_stack_p"], [199, 0, 1, "c.fp_stack_s", "fp_stack_s"], [201, 0, 1, "c.fp_stridedslicegrad_p", "fp_stridedslicegrad_p"], [201, 0, 1, "c.fp_stridedslicegrad_s", "fp_stridedslicegrad_s"], [202, 0, 1, "c.fp_subext_p", "fp_subext_p"], [202, 0, 1, "c.fp_subext_s", "fp_subext_s"], [203, 0, 1, "c.fp_subgrad_p", "fp_subgrad_p"], [203, 0, 1, "c.fp_subgrad_s", "fp_subgrad_s"], [202, 0, 1, "c.fp_subrelu6_p", "fp_subrelu6_p"], [202, 0, 1, "c.fp_subrelu6_s", "fp_subrelu6_s"], [202, 0, 1, "c.fp_subrelu_p", "fp_subrelu_p"], [202, 0, 1, "c.fp_subrelu_s", "fp_subrelu_s"], [12, 0, 1, "c.fp_swish_p", "fp_swish_p"], [12, 0, 1, "c.fp_swish_s", "fp_swish_s"], [13, 0, 1, "c.fp_tanh_grad_p", "fp_tanh_grad_p"], [13, 0, 1, "c.fp_tanh_grad_s", "fp_tanh_grad_s"], [12, 0, 1, "c.fp_tanh_p", "fp_tanh_p"], [12, 0, 1, "c.fp_tanh_s", "fp_tanh_s"], [206, 0, 1, "c.fp_tensor_scatter_add_p", "fp_tensor_scatter_add_p"], [206, 0, 1, "c.fp_tensor_scatter_add_s", "fp_tensor_scatter_add_s"], [208, 0, 1, "c.fp_tensorarrayread_p", "fp_tensorarrayread_p"], [208, 0, 1, "c.fp_tensorarrayread_s", "fp_tensorarrayread_s"], [210, 0, 1, "c.fp_tensorlistfromtensor_p", "fp_tensorlistfromtensor_p"], [210, 0, 1, "c.fp_tensorlistfromtensor_s", "fp_tensorlistfromtensor_s"], [215, 0, 1, "c.fp_tile_p", "fp_tile_p"], [215, 0, 1, "c.fp_tile_s", "fp_tile_s"], [146, 0, 1, "c.fp_to_i8_quant_p", "fp_to_i8_quant_p"], [146, 0, 1, "c.fp_to_i8_quant_s", "fp_to_i8_quant_s"], [216, 0, 1, "c.fp_topk_fusion_p", "fp_topk_fusion_p"], [216, 0, 1, "c.fp_topk_fusion_s", "fp_topk_fusion_s"], [217, 0, 1, "c.fp_transpose_p", "fp_transpose_p"], [217, 0, 1, "c.fp_transpose_s", "fp_transpose_s"], [218, 0, 1, "c.fp_tril_p", "fp_tril_p"], [218, 0, 1, "c.fp_tril_s", "fp_tril_s"], [219, 0, 1, "c.fp_triu_p", "fp_triu_p"], [219, 0, 1, "c.fp_triu_s", "fp_triu_s"], [220, 0, 1, "c.fp_uniform_real_p", "fp_uniform_real_p"], [220, 0, 1, "c.fp_uniform_real_s", "fp_uniform_real_s"], [222, 0, 1, "c.fp_unsorted_segment_sum_p", "fp_unsorted_segment_sum_p"], [222, 0, 1, "c.fp_unsorted_segment_sum_s", "fp_unsorted_segment_sum_s"], [225, 0, 1, "c.fp_where_p", "fp_where_p"], [225, 0, 1, "c.fp_where_s", "fp_where_s"], [226, 0, 1, "c.fp_zerolike_p", "fp_zerolike_p"], [226, 0, 1, "c.fp_zerolike_s", "fp_zerolike_s"], [95, 0, 1, "c.hp_Gru_p", "hp_Gru_p"], [95, 0, 1, "c.hp_Gru_s", "hp_Gru_s"], [117, 0, 1, "c.hp_Lstm_p", "hp_Lstm_p"], [117, 0, 1, "c.hp_Lstm_s", "hp_Lstm_s"], [66, 0, 1, "c.hp_QuantData_p", "hp_QuantData_p"], [66, 0, 1, "c.hp_QuantData_s", "hp_QuantData_s"], [221, 0, 1, "c.hp_Unique_p", "hp_Unique_p"], [221, 0, 1, "c.hp_Unique_s", "hp_Unique_s"], [10, 0, 1, "c.hp_abs_p", "hp_abs_p"], [10, 0, 1, "c.hp_abs_s", "hp_abs_s"], [11, 0, 1, "c.hp_absgrad_p", "hp_absgrad_p"], [11, 0, 1, "c.hp_absgrad_s", "hp_absgrad_s"], [15, 0, 1, "c.hp_adamweightdecay_p", "hp_adamweightdecay_p"], [15, 0, 1, "c.hp_adamweightdecay_s", "hp_adamweightdecay_s"], [16, 0, 1, "c.hp_adder_p", "hp_adder_p"], [16, 0, 1, "c.hp_adder_s", "hp_adder_s"], [17, 0, 1, "c.hp_addext_p", "hp_addext_p"], [17, 0, 1, "c.hp_addext_s", "hp_addext_s"], [18, 0, 1, "c.hp_addgrad_p", "hp_addgrad_p"], [18, 0, 1, "c.hp_addgrad_s", "hp_addgrad_s"], [17, 0, 1, "c.hp_addrelu6_p", "hp_addrelu6_p"], [17, 0, 1, "c.hp_addrelu6_s", "hp_addrelu6_s"], [17, 0, 1, "c.hp_addrelu_p", "hp_addrelu_p"], [17, 0, 1, "c.hp_addrelu_s", "hp_addrelu_s"], [20, 0, 1, "c.hp_affine_p", "hp_affine_p"], [20, 0, 1, "c.hp_affine_s", "hp_affine_s"], [112, 0, 1, "c.hp_and_p", "hp_and_p"], [112, 0, 1, "c.hp_and_s", "hp_and_s"], [23, 0, 1, "c.hp_applymomentum_p", "hp_applymomentum_p"], [23, 0, 1, "c.hp_applymomentum_s", "hp_applymomentum_s"], [24, 0, 1, "c.hp_argmax_p", "hp_argmax_p"], [24, 0, 1, "c.hp_argmax_s", "hp_argmax_s"], [25, 0, 1, "c.hp_argmin_p", "hp_argmin_p"], [25, 0, 1, "c.hp_argmin_s", "hp_argmin_s"], [27, 0, 1, "c.hp_assign_p", "hp_assign_p"], [27, 0, 1, "c.hp_assign_s", "hp_assign_s"], [28, 0, 1, "c.hp_assignadd_p", "hp_assignadd_p"], [28, 0, 1, "c.hp_assignadd_s", "hp_assignadd_s"], [31, 0, 1, "c.hp_avgpool_fusion_p", "hp_avgpool_fusion_p"], [31, 0, 1, "c.hp_avgpool_fusion_s", "hp_avgpool_fusion_s"], [32, 0, 1, "c.hp_avgpoolinggrad_p", "hp_avgpoolinggrad_p"], [32, 0, 1, "c.hp_avgpoolinggrad_s", "hp_avgpoolinggrad_s"], [33, 0, 1, "c.hp_batchnorm_p", "hp_batchnorm_p"], [33, 0, 1, "c.hp_batchnorm_s", "hp_batchnorm_s"], [34, 0, 1, "c.hp_batchnormgrad_p", "hp_batchnormgrad_p"], [34, 0, 1, "c.hp_batchnormgrad_s", "hp_batchnormgrad_s"], [35, 0, 1, "c.hp_batchtospace_p", "hp_batchtospace_p"], [35, 0, 1, "c.hp_batchtospace_s", "hp_batchtospace_s"], [36, 0, 1, "c.hp_batchtospacend_p", "hp_batchtospacend_p"], [36, 0, 1, "c.hp_batchtospacend_s", "hp_batchtospacend_s"], [37, 0, 1, "c.hp_biasadd_p", "hp_biasadd_p"], [37, 0, 1, "c.hp_biasadd_s", "hp_biasadd_s"], [38, 0, 1, "c.hp_biasaddgrad_p", "hp_biasaddgrad_p"], [38, 0, 1, "c.hp_biasaddgrad_s", "hp_biasaddgrad_s"], [39, 0, 1, "c.hp_binarycrossentropy_p", "hp_binarycrossentropy_p"], [39, 0, 1, "c.hp_binarycrossentropy_s", "hp_binarycrossentropy_s"], [40, 0, 1, "c.hp_binarycrossentropygrad_p", "hp_binarycrossentropygrad_p"], [40, 0, 1, "c.hp_binarycrossentropygrad_s", "hp_binarycrossentropygrad_s"], [41, 0, 1, "c.hp_broadcastto_p", "hp_broadcastto_p"], [41, 0, 1, "c.hp_broadcastto_s", "hp_broadcastto_s"], [43, 0, 1, "c.hp_ceil_p", "hp_ceil_p"], [43, 0, 1, "c.hp_ceil_s", "hp_ceil_s"], [12, 0, 1, "c.hp_celu_p", "hp_celu_p"], [12, 0, 1, "c.hp_celu_s", "hp_celu_s"], [12, 0, 1, "c.hp_clip_p", "hp_clip_p"], [12, 0, 1, "c.hp_clip_s", "hp_clip_s"], [45, 0, 1, "c.hp_concat_p", "hp_concat_p"], [45, 0, 1, "c.hp_concat_s", "hp_concat_s"], [47, 0, 1, "c.hp_conv2d_p", "hp_conv2d_p"], [47, 0, 1, "c.hp_conv2d_s", "hp_conv2d_s"], [49, 0, 1, "c.hp_conv2dbackpropfilterfusion_p", "hp_conv2dbackpropfilterfusion_p"], [49, 0, 1, "c.hp_conv2dbackpropfilterfusion_s", "hp_conv2dbackpropfilterfusion_s"], [50, 0, 1, "c.hp_conv2dbackpropinputfusion_p", "hp_conv2dbackpropinputfusion_p"], [50, 0, 1, "c.hp_conv2dbackpropinputfusion_s", "hp_conv2dbackpropinputfusion_s"], [48, 0, 1, "c.hp_convtranspose_p", "hp_convtranspose_p"], [48, 0, 1, "c.hp_convtranspose_s", "hp_convtranspose_s"], [51, 0, 1, "c.hp_cos_p", "hp_cos_p"], [51, 0, 1, "c.hp_cos_s", "hp_cos_s"], [53, 0, 1, "c.hp_crop_and_resize_anycore", "hp_crop_and_resize_anycore"], [54, 0, 1, "c.hp_cumsum_p", "hp_cumsum_p"], [54, 0, 1, "c.hp_cumsum_s", "hp_cumsum_s"], [58, 0, 1, "c.hp_deconvgradfilter_p", "hp_deconvgradfilter_p"], [58, 0, 1, "c.hp_deconvgradfilter_s", "hp_deconvgradfilter_s"], [59, 0, 1, "c.hp_depthtospace_p", "hp_depthtospace_p"], [59, 0, 1, "c.hp_depthtospace_s", "hp_depthtospace_s"], [60, 0, 1, "c.hp_detection_post_process_p", "hp_detection_post_process_p"], [60, 0, 1, "c.hp_detection_post_process_s", "hp_detection_post_process_s"], [61, 0, 1, "c.hp_div_fusion_p", "hp_div_fusion_p"], [61, 0, 1, "c.hp_div_fusion_s", "hp_div_fusion_s"], [63, 0, 1, "c.hp_dropout_p", "hp_dropout_p"], [63, 0, 1, "c.hp_dropout_s", "hp_dropout_s"], [64, 0, 1, "c.hp_dropoutgrad_p", "hp_dropoutgrad_p"], [64, 0, 1, "c.hp_dropoutgrad_s", "hp_dropoutgrad_s"], [67, 0, 1, "c.hp_eltwise_p", "hp_eltwise_p"], [67, 0, 1, "c.hp_eltwise_s", "hp_eltwise_s"], [12, 0, 1, "c.hp_elu_p", "hp_elu_p"], [12, 0, 1, "c.hp_elu_s", "hp_elu_s"], [69, 0, 1, "c.hp_embeddinglookup_p", "hp_embeddinglookup_p"], [69, 0, 1, "c.hp_embeddinglookup_s", "hp_embeddinglookup_s"], [70, 0, 1, "c.hp_equal_p", "hp_equal_p"], [70, 0, 1, "c.hp_equal_s", "hp_equal_s"], [71, 0, 1, "c.hp_erf_p", "hp_erf_p"], [71, 0, 1, "c.hp_erf_s", "hp_erf_s"], [73, 0, 1, "c.hp_expfusion_p", "hp_expfusion_p"], [73, 0, 1, "c.hp_expfusion_s", "hp_expfusion_s"], [74, 0, 1, "c.hp_fake_quant_with_min_max_vars_p", "hp_fake_quant_with_min_max_vars_p"], [75, 0, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "hp_fake_quant_with_min_max_vars_per_channel_p"], [75, 0, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "hp_fake_quant_with_min_max_vars_per_channel_s"], [74, 0, 1, "c.hp_fake_quant_with_min_max_vars_s", "hp_fake_quant_with_min_max_vars_s"], [78, 0, 1, "c.hp_fill_p", "hp_fill_p"], [78, 0, 1, "c.hp_fill_s", "hp_fill_s"], [81, 0, 1, "c.hp_flattengrad_p", "hp_flattengrad_p"], [81, 0, 1, "c.hp_flattengrad_s", "hp_flattengrad_s"], [82, 0, 1, "c.hp_floor_p", "hp_floor_p"], [82, 0, 1, "c.hp_floor_s", "hp_floor_s"], [83, 0, 1, "c.hp_floordiv_p", "hp_floordiv_p"], [83, 0, 1, "c.hp_floordiv_s", "hp_floordiv_s"], [84, 0, 1, "c.hp_floormod_p", "hp_floormod_p"], [84, 0, 1, "c.hp_floormod_s", "hp_floormod_s"], [85, 0, 1, "c.hp_formattranspose_p", "hp_formattranspose_p"], [85, 0, 1, "c.hp_formattranspose_s", "hp_formattranspose_s"], [86, 0, 1, "c.hp_fullconnection_p", "hp_fullconnection_p"], [86, 0, 1, "c.hp_fullconnection_s", "hp_fullconnection_s"], [87, 0, 1, "c.hp_fusedbatchnorm_p", "hp_fusedbatchnorm_p"], [87, 0, 1, "c.hp_fusedbatchnorm_s", "hp_fusedbatchnorm_s"], [89, 0, 1, "c.hp_gather_nd_p", "hp_gather_nd_p"], [89, 0, 1, "c.hp_gather_nd_s", "hp_gather_nd_s"], [88, 0, 1, "c.hp_gather_p", "hp_gather_p"], [88, 0, 1, "c.hp_gather_s", "hp_gather_s"], [12, 0, 1, "c.hp_gelu_p", "hp_gelu_p"], [12, 0, 1, "c.hp_gelu_s", "hp_gelu_s"], [91, 0, 1, "c.hp_glu_p", "hp_glu_p"], [91, 0, 1, "c.hp_glu_s", "hp_glu_s"], [62, 0, 1, "c.hp_graddiv1l_p", "hp_graddiv1l_p"], [62, 0, 1, "c.hp_graddiv1l_s", "hp_graddiv1l_s"], [62, 0, 1, "c.hp_graddiv2l_p", "hp_graddiv2l_p"], [62, 0, 1, "c.hp_graddiv2l_s", "hp_graddiv2l_s"], [62, 0, 1, "c.hp_graddiv_p", "hp_graddiv_p"], [62, 0, 1, "c.hp_graddiv_s", "hp_graddiv_s"], [131, 0, 1, "c.hp_gradmul1l_p", "hp_gradmul1l_p"], [131, 0, 1, "c.hp_gradmul1l_s", "hp_gradmul1l_s"], [131, 0, 1, "c.hp_gradmul2l_p", "hp_gradmul2l_p"], [131, 0, 1, "c.hp_gradmul2l_s", "hp_gradmul2l_s"], [131, 0, 1, "c.hp_gradmul_p", "hp_gradmul_p"], [131, 0, 1, "c.hp_gradmul_s", "hp_gradmul_s"], [92, 0, 1, "c.hp_greater_p", "hp_greater_p"], [92, 0, 1, "c.hp_greater_s", "hp_greater_s"], [93, 0, 1, "c.hp_greaterequal_p", "hp_greaterequal_p"], [93, 0, 1, "c.hp_greaterequal_s", "hp_greaterequal_s"], [94, 0, 1, "c.hp_groupnormfusion_p", "hp_groupnormfusion_p"], [94, 0, 1, "c.hp_groupnormfusion_s", "hp_groupnormfusion_s"], [12, 0, 1, "c.hp_hardshrink_p", "hp_hardshrink_p"], [12, 0, 1, "c.hp_hardshrink_s", "hp_hardshrink_s"], [12, 0, 1, "c.hp_hardtanh_p", "hp_hardtanh_p"], [12, 0, 1, "c.hp_hardtanh_s", "hp_hardtanh_s"], [12, 0, 1, "c.hp_hsigmoid_p", "hp_hsigmoid_p"], [12, 0, 1, "c.hp_hsigmoid_s", "hp_hsigmoid_s"], [12, 0, 1, "c.hp_hswish_p", "hp_hswish_p"], [12, 0, 1, "c.hp_hswish_s", "hp_hswish_s"], [97, 0, 1, "c.hp_instancenorm_p", "hp_instancenorm_p"], [97, 0, 1, "c.hp_instancenorm_s", "hp_instancenorm_s"], [99, 0, 1, "c.hp_isfinite_p", "hp_isfinite_p"], [99, 0, 1, "c.hp_isfinite_s", "hp_isfinite_s"], [100, 0, 1, "c.hp_l2norm_p", "hp_l2norm_p"], [100, 0, 1, "c.hp_l2norm_s", "hp_l2norm_s"], [101, 0, 1, "c.hp_layernormfusion_p", "hp_layernormfusion_p"], [101, 0, 1, "c.hp_layernormfusion_s", "hp_layernormfusion_s"], [102, 0, 1, "c.hp_layernormgrad_p", "hp_layernormgrad_p"], [102, 0, 1, "c.hp_layernormgrad_s", "hp_layernormgrad_s"], [103, 0, 1, "c.hp_leaky_relu_p", "hp_leaky_relu_p"], [103, 0, 1, "c.hp_leaky_relu_s", "hp_leaky_relu_s"], [104, 0, 1, "c.hp_less_p", "hp_less_p"], [104, 0, 1, "c.hp_less_s", "hp_less_s"], [105, 0, 1, "c.hp_lessequal_p", "hp_lessequal_p"], [105, 0, 1, "c.hp_lessequal_s", "hp_lessequal_s"], [108, 0, 1, "c.hp_log1p_p", "hp_log1p_p"], [108, 0, 1, "c.hp_log1p_s", "hp_log1p_s"], [109, 0, 1, "c.hp_log_grad_p", "hp_log_grad_p"], [109, 0, 1, "c.hp_log_grad_s", "hp_log_grad_s"], [107, 0, 1, "c.hp_log_p", "hp_log_p"], [107, 0, 1, "c.hp_log_s", "hp_log_s"], [110, 0, 1, "c.hp_logical_not_p", "hp_logical_not_p"], [110, 0, 1, "c.hp_logical_not_s", "hp_logical_not_s"], [111, 0, 1, "c.hp_logical_or_p", "hp_logical_or_p"], [111, 0, 1, "c.hp_logical_or_s", "hp_logical_or_s"], [113, 0, 1, "c.hp_logsoftmax_p", "hp_logsoftmax_p"], [113, 0, 1, "c.hp_logsoftmax_s", "hp_logsoftmax_s"], [114, 0, 1, "c.hp_lpnorm_p", "hp_lpnorm_p"], [114, 0, 1, "c.hp_lpnorm_s", "hp_lpnorm_s"], [12, 0, 1, "c.hp_lrelu_p", "hp_lrelu_p"], [12, 0, 1, "c.hp_lrelu_s", "hp_lrelu_s"], [115, 0, 1, "c.hp_lrn_p", "hp_lrn_p"], [115, 0, 1, "c.hp_lrn_s", "hp_lrn_s"], [116, 0, 1, "c.hp_lsh_projection_p", "hp_lsh_projection_p"], [116, 0, 1, "c.hp_lsh_projection_s", "hp_lsh_projection_s"], [118, 0, 1, "c.hp_lstmgrad_p", "hp_lstmgrad_p"], [118, 0, 1, "c.hp_lstmgrad_s", "hp_lstmgrad_s"], [119, 0, 1, "c.hp_lstmgraddata_p", "hp_lstmgraddata_p"], [119, 0, 1, "c.hp_lstmgraddata_s", "hp_lstmgraddata_s"], [120, 0, 1, "c.hp_lstmgradweight_p", "hp_lstmgradweight_p"], [120, 0, 1, "c.hp_lstmgradweight_s", "hp_lstmgradweight_s"], [122, 0, 1, "c.hp_maximum_p", "hp_maximum_p"], [122, 0, 1, "c.hp_maximum_s", "hp_maximum_s"], [123, 0, 1, "c.hp_maximumgrad_p", "hp_maximumgrad_p"], [123, 0, 1, "c.hp_maximumgrad_s", "hp_maximumgrad_s"], [124, 0, 1, "c.hp_maxpool_fusion_p", "hp_maxpool_fusion_p"], [124, 0, 1, "c.hp_maxpool_fusion_s", "hp_maxpool_fusion_s"], [125, 0, 1, "c.hp_maxpool_grad_p", "hp_maxpool_grad_p"], [125, 0, 1, "c.hp_maxpool_grad_s", "hp_maxpool_grad_s"], [126, 0, 1, "c.hp_mfcc_p", "hp_mfcc_p"], [126, 0, 1, "c.hp_mfcc_s", "hp_mfcc_s"], [127, 0, 1, "c.hp_minimum_p", "hp_minimum_p"], [127, 0, 1, "c.hp_minimum_s", "hp_minimum_s"], [128, 0, 1, "c.hp_minimumgrad_p", "hp_minimumgrad_p"], [128, 0, 1, "c.hp_minimumgrad_s", "hp_minimumgrad_s"], [129, 0, 1, "c.hp_mod_p", "hp_mod_p"], [129, 0, 1, "c.hp_mod_s", "hp_mod_s"], [130, 0, 1, "c.hp_mul_p", "hp_mul_p"], [130, 0, 1, "c.hp_mul_s", "hp_mul_s"], [133, 0, 1, "c.hp_neg_grad_p", "hp_neg_grad_p"], [133, 0, 1, "c.hp_neg_grad_s", "hp_neg_grad_s"], [132, 0, 1, "c.hp_neg_p", "hp_neg_p"], [132, 0, 1, "c.hp_neg_s", "hp_neg_s"], [134, 0, 1, "c.hp_nllloss_p", "hp_nllloss_p"], [134, 0, 1, "c.hp_nllloss_s", "hp_nllloss_s"], [135, 0, 1, "c.hp_nlllossgrad_p", "hp_nlllossgrad_p"], [135, 0, 1, "c.hp_nlllossgrad_s", "hp_nlllossgrad_s"], [136, 0, 1, "c.hp_non_max_suppression_p", "hp_non_max_suppression_p"], [136, 0, 1, "c.hp_non_max_suppression_s", "hp_non_max_suppression_s"], [138, 0, 1, "c.hp_not_equal_p", "hp_not_equal_p"], [138, 0, 1, "c.hp_not_equal_s", "hp_not_equal_s"], [139, 0, 1, "c.hp_onehot_p", "hp_onehot_p"], [139, 0, 1, "c.hp_onehot_s", "hp_onehot_s"], [140, 0, 1, "c.hp_ones_like_p", "hp_ones_like_p"], [140, 0, 1, "c.hp_ones_like_s", "hp_ones_like_s"], [141, 0, 1, "c.hp_padfusion_p", "hp_padfusion_p"], [141, 0, 1, "c.hp_padfusion_s", "hp_padfusion_s"], [142, 0, 1, "c.hp_pow_fusion_p", "hp_pow_fusion_p"], [142, 0, 1, "c.hp_pow_fusion_s", "hp_pow_fusion_s"], [143, 0, 1, "c.hp_power_grad_p", "hp_power_grad_p"], [143, 0, 1, "c.hp_power_grad_s", "hp_power_grad_s"], [144, 0, 1, "c.hp_prelufusion_p", "hp_prelufusion_p"], [144, 0, 1, "c.hp_prelufusion_s", "hp_prelufusion_s"], [145, 0, 1, "c.hp_priorbox_p", "hp_priorbox_p"], [145, 0, 1, "c.hp_priorbox_s", "hp_priorbox_s"], [148, 0, 1, "c.hp_random_normal_p", "hp_random_normal_p"], [148, 0, 1, "c.hp_random_normal_s", "hp_random_normal_s"], [149, 0, 1, "c.hp_random_standard_normal_p", "hp_random_standard_normal_p"], [149, 0, 1, "c.hp_random_standard_normal_s", "hp_random_standard_normal_s"], [152, 0, 1, "c.hp_real_div_p", "hp_real_div_p"], [152, 0, 1, "c.hp_real_div_s", "hp_real_div_s"], [153, 0, 1, "c.hp_reciprocal_p", "hp_reciprocal_p"], [153, 0, 1, "c.hp_reciprocal_s", "hp_reciprocal_s"], [154, 0, 1, "c.hp_reduce_p", "hp_reduce_p"], [154, 0, 1, "c.hp_reduce_s", "hp_reduce_s"], [21, 0, 1, "c.hp_reduceall_p", "hp_reduceall_p"], [21, 0, 1, "c.hp_reduceall_s", "hp_reduceall_s"], [155, 0, 1, "c.hp_reducescatter_p", "hp_reducescatter_p"], [155, 0, 1, "c.hp_reducescatter_s", "hp_reducescatter_s"], [12, 0, 1, "c.hp_relu6_p", "hp_relu6_p"], [12, 0, 1, "c.hp_relu6_s", "hp_relu6_s"], [13, 0, 1, "c.hp_relu_grad_p", "hp_relu_grad_p"], [13, 0, 1, "c.hp_relu_grad_s", "hp_relu_grad_s"], [12, 0, 1, "c.hp_relu_p", "hp_relu_p"], [12, 0, 1, "c.hp_relu_s", "hp_relu_s"], [156, 0, 1, "c.hp_reshape_p", "hp_reshape_p"], [156, 0, 1, "c.hp_reshape_s", "hp_reshape_s"], [157, 0, 1, "c.hp_resize_anycore", "hp_resize_anycore"], [158, 0, 1, "c.hp_resizebilineargrad_p", "hp_resizebilineargrad_p"], [158, 0, 1, "c.hp_resizebilineargrad_s", "hp_resizebilineargrad_s"], [158, 0, 1, "c.hp_resizenearestneighborgrad_p", "hp_resizenearestneighborgrad_p"], [158, 0, 1, "c.hp_resizenearestneighborgrad_s", "hp_resizenearestneighborgrad_s"], [162, 0, 1, "c.hp_roipooling_p", "hp_roipooling_p"], [162, 0, 1, "c.hp_roipooling_s", "hp_roipooling_s"], [163, 0, 1, "c.hp_round_p", "hp_round_p"], [163, 0, 1, "c.hp_round_s", "hp_round_s"], [164, 0, 1, "c.hp_rsqrt_p", "hp_rsqrt_p"], [164, 0, 1, "c.hp_rsqrt_s", "hp_rsqrt_s"], [165, 0, 1, "c.hp_rsqrtgrad_p", "hp_rsqrtgrad_p"], [165, 0, 1, "c.hp_rsqrtgrad_s", "hp_rsqrtgrad_s"], [166, 0, 1, "c.hp_scalefusion_p", "hp_scalefusion_p"], [166, 0, 1, "c.hp_scalefusion_s", "hp_scalefusion_s"], [167, 0, 1, "c.hp_scatter_elements_p", "hp_scatter_elements_p"], [167, 0, 1, "c.hp_scatter_elements_s", "hp_scatter_elements_s"], [168, 0, 1, "c.hp_scatter_nd_p", "hp_scatter_nd_p"], [168, 0, 1, "c.hp_scatter_nd_s", "hp_scatter_nd_s"], [169, 0, 1, "c.hp_scatter_nd_update_p", "hp_scatter_nd_update_p"], [169, 0, 1, "c.hp_scatter_nd_update_s", "hp_scatter_nd_update_s"], [170, 0, 1, "c.hp_select_p", "hp_select_p"], [170, 0, 1, "c.hp_select_s", "hp_select_s"], [171, 0, 1, "c.hp_sgd_p", "hp_sgd_p"], [171, 0, 1, "c.hp_sgd_s", "hp_sgd_s"], [12, 0, 1, "c.hp_sigmoid_p", "hp_sigmoid_p"], [12, 0, 1, "c.hp_sigmoid_s", "hp_sigmoid_s"], [174, 0, 1, "c.hp_sigmoidcrossentropywithlogits_p", "hp_sigmoidcrossentropywithlogits_p"], [174, 0, 1, "c.hp_sigmoidcrossentropywithlogits_s", "hp_sigmoidcrossentropywithlogits_s"], [173, 0, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_p", "hp_sigmoidcrossentropywithlogitsgrad_p"], [173, 0, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_s", "hp_sigmoidcrossentropywithlogitsgrad_s"], [175, 0, 1, "c.hp_sin_p", "hp_sin_p"], [175, 0, 1, "c.hp_sin_s", "hp_sin_s"], [178, 0, 1, "c.hp_slice_p", "hp_slice_p"], [178, 0, 1, "c.hp_slice_s", "hp_slice_s"], [179, 0, 1, "c.hp_smoothl1loss_p", "hp_smoothl1loss_p"], [179, 0, 1, "c.hp_smoothl1loss_s", "hp_smoothl1loss_s"], [180, 0, 1, "c.hp_smoothl1lossgrad_p", "hp_smoothl1lossgrad_p"], [180, 0, 1, "c.hp_smoothl1lossgrad_s", "hp_smoothl1lossgrad_s"], [182, 0, 1, "c.hp_softmax_cross_entropy_with_logits_p", "hp_softmax_cross_entropy_with_logits_p"], [182, 0, 1, "c.hp_softmax_cross_entropy_with_logits_s", "hp_softmax_cross_entropy_with_logits_s"], [181, 0, 1, "c.hp_softmax_p", "hp_softmax_p"], [181, 0, 1, "c.hp_softmax_s", "hp_softmax_s"], [12, 0, 1, "c.hp_softplus_p", "hp_softplus_p"], [12, 0, 1, "c.hp_softplus_s", "hp_softplus_s"], [12, 0, 1, "c.hp_softshrink_p", "hp_softshrink_p"], [12, 0, 1, "c.hp_softshrink_s", "hp_softshrink_s"], [12, 0, 1, "c.hp_softsignopt_p", "hp_softsignopt_p"], [12, 0, 1, "c.hp_softsignopt_s", "hp_softsignopt_s"], [183, 0, 1, "c.hp_spacetobatch_p", "hp_spacetobatch_p"], [183, 0, 1, "c.hp_spacetobatch_s", "hp_spacetobatch_s"], [184, 0, 1, "c.hp_spacetobatchnd_p", "hp_spacetobatchnd_p"], [184, 0, 1, "c.hp_spacetobatchnd_s", "hp_spacetobatchnd_s"], [185, 0, 1, "c.hp_spacetodepth_p", "hp_spacetodepth_p"], [185, 0, 1, "c.hp_spacetodepth_s", "hp_spacetodepth_s"], [186, 0, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "hp_sparse_softmax_cross_entropy_with_logits_p"], [186, 0, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "hp_sparse_softmax_cross_entropy_with_logits_s"], [187, 0, 1, "c.hp_sparsefillemptyrows_p", "hp_sparsefillemptyrows_p"], [187, 0, 1, "c.hp_sparsefillemptyrows_s", "hp_sparsefillemptyrows_s"], [189, 0, 1, "c.hp_sparsesegmentsum_p", "hp_sparsesegmentsum_p"], [189, 0, 1, "c.hp_sparsesegmentsum_s", "hp_sparsesegmentsum_s"], [190, 0, 1, "c.hp_sparsetodense_p", "hp_sparsetodense_p"], [190, 0, 1, "c.hp_sparsetodense_s", "hp_sparsetodense_s"], [191, 0, 1, "c.hp_splice_p", "hp_splice_p"], [191, 0, 1, "c.hp_splice_s", "hp_splice_s"], [192, 0, 1, "c.hp_split_p", "hp_split_p"], [192, 0, 1, "c.hp_split_s", "hp_split_s"], [193, 0, 1, "c.hp_split_with_overlap_p", "hp_split_with_overlap_p"], [193, 0, 1, "c.hp_split_with_overlap_s", "hp_split_with_overlap_s"], [194, 0, 1, "c.hp_sqrt_p", "hp_sqrt_p"], [194, 0, 1, "c.hp_sqrt_s", "hp_sqrt_s"], [195, 0, 1, "c.hp_sqrtgrad_p", "hp_sqrtgrad_p"], [195, 0, 1, "c.hp_sqrtgrad_s", "hp_sqrtgrad_s"], [196, 0, 1, "c.hp_square_p", "hp_square_p"], [196, 0, 1, "c.hp_square_s", "hp_square_s"], [197, 0, 1, "c.hp_squaredifference_p", "hp_squaredifference_p"], [197, 0, 1, "c.hp_squaredifference_s", "hp_squaredifference_s"], [199, 0, 1, "c.hp_stack_p", "hp_stack_p"], [199, 0, 1, "c.hp_stack_s", "hp_stack_s"], [201, 0, 1, "c.hp_stridedslicegrad_p", "hp_stridedslicegrad_p"], [201, 0, 1, "c.hp_stridedslicegrad_s", "hp_stridedslicegrad_s"], [202, 0, 1, "c.hp_subext_p", "hp_subext_p"], [202, 0, 1, "c.hp_subext_s", "hp_subext_s"], [203, 0, 1, "c.hp_subgrad_p", "hp_subgrad_p"], [203, 0, 1, "c.hp_subgrad_s", "hp_subgrad_s"], [202, 0, 1, "c.hp_subrelu6_p", "hp_subrelu6_p"], [202, 0, 1, "c.hp_subrelu6_s", "hp_subrelu6_s"], [202, 0, 1, "c.hp_subrelu_p", "hp_subrelu_p"], [202, 0, 1, "c.hp_subrelu_s", "hp_subrelu_s"], [12, 0, 1, "c.hp_swish_p", "hp_swish_p"], [12, 0, 1, "c.hp_swish_s", "hp_swish_s"], [12, 0, 1, "c.hp_tanh_p", "hp_tanh_p"], [12, 0, 1, "c.hp_tanh_s", "hp_tanh_s"], [206, 0, 1, "c.hp_tensor_scatter_add_p", "hp_tensor_scatter_add_p"], [206, 0, 1, "c.hp_tensor_scatter_add_s", "hp_tensor_scatter_add_s"], [208, 0, 1, "c.hp_tensorarrayread_p", "hp_tensorarrayread_p"], [208, 0, 1, "c.hp_tensorarrayread_s", "hp_tensorarrayread_s"], [210, 0, 1, "c.hp_tensorlistfromtensor_p", "hp_tensorlistfromtensor_p"], [210, 0, 1, "c.hp_tensorlistfromtensor_s", "hp_tensorlistfromtensor_s"], [215, 0, 1, "c.hp_tile_p", "hp_tile_p"], [215, 0, 1, "c.hp_tile_s", "hp_tile_s"], [146, 0, 1, "c.hp_to_i8_quant_p", "hp_to_i8_quant_p"], [146, 0, 1, "c.hp_to_i8_quant_s", "hp_to_i8_quant_s"], [216, 0, 1, "c.hp_topk_fusion_p", "hp_topk_fusion_p"], [216, 0, 1, "c.hp_topk_fusion_s", "hp_topk_fusion_s"], [217, 0, 1, "c.hp_transpose_p", "hp_transpose_p"], [217, 0, 1, "c.hp_transpose_s", "hp_transpose_s"], [218, 0, 1, "c.hp_tril_p", "hp_tril_p"], [218, 0, 1, "c.hp_tril_s", "hp_tril_s"], [219, 0, 1, "c.hp_triu_p", "hp_triu_p"], [219, 0, 1, "c.hp_triu_s", "hp_triu_s"], [220, 0, 1, "c.hp_uniform_real_p", "hp_uniform_real_p"], [220, 0, 1, "c.hp_uniform_real_s", "hp_uniform_real_s"], [225, 0, 1, "c.hp_where_p", "hp_where_p"], [225, 0, 1, "c.hp_where_s", "hp_where_s"], [226, 0, 1, "c.hp_zerolike_p", "hp_zerolike_p"], [226, 0, 1, "c.hp_zerolike_s", "hp_zerolike_s"], [221, 0, 1, "c.i16_Unique_p", "i16_Unique_p"], [221, 0, 1, "c.i16_Unique_s", "i16_Unique_s"], [10, 0, 1, "c.i16_abs_p", "i16_abs_p"], [10, 0, 1, "c.i16_abs_s", "i16_abs_s"], [17, 0, 1, "c.i16_addext_p", "i16_addext_p"], [17, 0, 1, "c.i16_addext_s", "i16_addext_s"], [19, 0, 1, "c.i16_addn_p", "i16_addn_p"], [19, 0, 1, "c.i16_addn_s", "i16_addn_s"], [17, 0, 1, "c.i16_addrelu6_p", "i16_addrelu6_p"], [17, 0, 1, "c.i16_addrelu6_s", "i16_addrelu6_s"], [17, 0, 1, "c.i16_addrelu_p", "i16_addrelu_p"], [17, 0, 1, "c.i16_addrelu_s", "i16_addrelu_s"], [22, 0, 1, "c.i16_allgather_p", "i16_allgather_p"], [22, 0, 1, "c.i16_allgather_s", "i16_allgather_s"], [112, 0, 1, "c.i16_and_p", "i16_and_p"], [112, 0, 1, "c.i16_and_s", "i16_and_s"], [27, 0, 1, "c.i16_assign_p", "i16_assign_p"], [27, 0, 1, "c.i16_assign_s", "i16_assign_s"], [28, 0, 1, "c.i16_assignadd_p", "i16_assignadd_p"], [28, 0, 1, "c.i16_assignadd_s", "i16_assignadd_s"], [35, 0, 1, "c.i16_batchtospace_p", "i16_batchtospace_p"], [35, 0, 1, "c.i16_batchtospace_s", "i16_batchtospace_s"], [36, 0, 1, "c.i16_batchtospacend_p", "i16_batchtospacend_p"], [36, 0, 1, "c.i16_batchtospacend_s", "i16_batchtospacend_s"], [37, 0, 1, "c.i16_biasadd_p", "i16_biasadd_p"], [37, 0, 1, "c.i16_biasadd_s", "i16_biasadd_s"], [41, 0, 1, "c.i16_broadcastto_p", "i16_broadcastto_p"], [41, 0, 1, "c.i16_broadcastto_s", "i16_broadcastto_s"], [44, 0, 1, "c.i16_clip_p", "i16_clip_p"], [44, 0, 1, "c.i16_clip_s", "i16_clip_s"], [45, 0, 1, "c.i16_concat_p", "i16_concat_p"], [45, 0, 1, "c.i16_concat_s", "i16_concat_s"], [46, 0, 1, "c.i16_constant_of_shape_p", "i16_constant_of_shape_p"], [46, 0, 1, "c.i16_constant_of_shape_s", "i16_constant_of_shape_s"], [51, 0, 1, "c.i16_cos_p", "i16_cos_p"], [51, 0, 1, "c.i16_cos_s", "i16_cos_s"], [54, 0, 1, "c.i16_cumsum_p", "i16_cumsum_p"], [54, 0, 1, "c.i16_cumsum_s", "i16_cumsum_s"], [59, 0, 1, "c.i16_depthtospace_p", "i16_depthtospace_p"], [59, 0, 1, "c.i16_depthtospace_s", "i16_depthtospace_s"], [61, 0, 1, "c.i16_div_fusion_p", "i16_div_fusion_p"], [61, 0, 1, "c.i16_div_fusion_s", "i16_div_fusion_s"], [67, 0, 1, "c.i16_eltwise_p", "i16_eltwise_p"], [67, 0, 1, "c.i16_eltwise_s", "i16_eltwise_s"], [70, 0, 1, "c.i16_equal_p", "i16_equal_p"], [70, 0, 1, "c.i16_equal_s", "i16_equal_s"], [73, 0, 1, "c.i16_expfusion_p", "i16_expfusion_p"], [73, 0, 1, "c.i16_expfusion_s", "i16_expfusion_s"], [55, 0, 1, "c.i16_extract_features_p", "i16_extract_features_p"], [55, 0, 1, "c.i16_extract_features_s", "i16_extract_features_s"], [78, 0, 1, "c.i16_fill_p", "i16_fill_p"], [78, 0, 1, "c.i16_fill_s", "i16_fill_s"], [85, 0, 1, "c.i16_formattranspose_p", "i16_formattranspose_p"], [85, 0, 1, "c.i16_formattranspose_s", "i16_formattranspose_s"], [89, 0, 1, "c.i16_gather_nd_p", "i16_gather_nd_p"], [89, 0, 1, "c.i16_gather_nd_s", "i16_gather_nd_s"], [88, 0, 1, "c.i16_gather_p", "i16_gather_p"], [88, 0, 1, "c.i16_gather_s", "i16_gather_s"], [90, 0, 1, "c.i16_gatherd_p", "i16_gatherd_p"], [90, 0, 1, "c.i16_gatherd_s", "i16_gatherd_s"], [92, 0, 1, "c.i16_greater_p", "i16_greater_p"], [92, 0, 1, "c.i16_greater_s", "i16_greater_s"], [93, 0, 1, "c.i16_greaterequal_p", "i16_greaterequal_p"], [93, 0, 1, "c.i16_greaterequal_s", "i16_greaterequal_s"], [98, 0, 1, "c.i16_invertpermutation_p", "i16_invertpermutation_p"], [98, 0, 1, "c.i16_invertpermutation_s", "i16_invertpermutation_s"], [99, 0, 1, "c.i16_isfinite_p", "i16_isfinite_p"], [99, 0, 1, "c.i16_isfinite_s", "i16_isfinite_s"], [104, 0, 1, "c.i16_less_p", "i16_less_p"], [104, 0, 1, "c.i16_less_s", "i16_less_s"], [105, 0, 1, "c.i16_lessequal_p", "i16_lessequal_p"], [105, 0, 1, "c.i16_lessequal_s", "i16_lessequal_s"], [108, 0, 1, "c.i16_log1p_p", "i16_log1p_p"], [108, 0, 1, "c.i16_log1p_s", "i16_log1p_s"], [107, 0, 1, "c.i16_log_p", "i16_log_p"], [107, 0, 1, "c.i16_log_s", "i16_log_s"], [110, 0, 1, "c.i16_logical_not_p", "i16_logical_not_p"], [110, 0, 1, "c.i16_logical_not_s", "i16_logical_not_s"], [111, 0, 1, "c.i16_logical_or_p", "i16_logical_or_p"], [111, 0, 1, "c.i16_logical_or_s", "i16_logical_or_s"], [116, 0, 1, "c.i16_lsh_projection_p", "i16_lsh_projection_p"], [116, 0, 1, "c.i16_lsh_projection_s", "i16_lsh_projection_s"], [121, 0, 1, "c.i16_matmulfusion_p", "i16_matmulfusion_p"], [121, 0, 1, "c.i16_matmulfusion_s", "i16_matmulfusion_s"], [122, 0, 1, "c.i16_maximum_p", "i16_maximum_p"], [122, 0, 1, "c.i16_maximum_s", "i16_maximum_s"], [127, 0, 1, "c.i16_minimum_p", "i16_minimum_p"], [127, 0, 1, "c.i16_minimum_s", "i16_minimum_s"], [129, 0, 1, "c.i16_mod_p", "i16_mod_p"], [129, 0, 1, "c.i16_mod_s", "i16_mod_s"], [130, 0, 1, "c.i16_mul_p", "i16_mul_p"], [130, 0, 1, "c.i16_mul_s", "i16_mul_s"], [133, 0, 1, "c.i16_neg_grad_p", "i16_neg_grad_p"], [133, 0, 1, "c.i16_neg_grad_s", "i16_neg_grad_s"], [132, 0, 1, "c.i16_neg_p", "i16_neg_p"], [132, 0, 1, "c.i16_neg_s", "i16_neg_s"], [137, 0, 1, "c.i16_nonzero_p", "i16_nonzero_p"], [137, 0, 1, "c.i16_nonzero_s", "i16_nonzero_s"], [138, 0, 1, "c.i16_not_equal_p", "i16_not_equal_p"], [138, 0, 1, "c.i16_not_equal_s", "i16_not_equal_s"], [139, 0, 1, "c.i16_onehot_p", "i16_onehot_p"], [139, 0, 1, "c.i16_onehot_s", "i16_onehot_s"], [140, 0, 1, "c.i16_ones_like_p", "i16_ones_like_p"], [140, 0, 1, "c.i16_ones_like_s", "i16_ones_like_s"], [141, 0, 1, "c.i16_padfusion_p", "i16_padfusion_p"], [141, 0, 1, "c.i16_padfusion_s", "i16_padfusion_s"], [142, 0, 1, "c.i16_pow_fusion_p", "i16_pow_fusion_p"], [142, 0, 1, "c.i16_pow_fusion_s", "i16_pow_fusion_s"], [147, 0, 1, "c.i16_raggedrange_p", "i16_raggedrange_p"], [147, 0, 1, "c.i16_raggedrange_s", "i16_raggedrange_s"], [150, 0, 1, "c.i16_range_p", "i16_range_p"], [150, 0, 1, "c.i16_range_s", "i16_range_s"], [152, 0, 1, "c.i16_real_div_p", "i16_real_div_p"], [152, 0, 1, "c.i16_real_div_s", "i16_real_div_s"], [153, 0, 1, "c.i16_reciprocal_p", "i16_reciprocal_p"], [153, 0, 1, "c.i16_reciprocal_s", "i16_reciprocal_s"], [154, 0, 1, "c.i16_reduce_p", "i16_reduce_p"], [154, 0, 1, "c.i16_reduce_s", "i16_reduce_s"], [21, 0, 1, "c.i16_reduceall_p", "i16_reduceall_p"], [21, 0, 1, "c.i16_reduceall_s", "i16_reduceall_s"], [155, 0, 1, "c.i16_reducescatter_p", "i16_reducescatter_p"], [155, 0, 1, "c.i16_reducescatter_s", "i16_reducescatter_s"], [156, 0, 1, "c.i16_reshape_p", "i16_reshape_p"], [156, 0, 1, "c.i16_reshape_s", "i16_reshape_s"], [164, 0, 1, "c.i16_rsqrt_p", "i16_rsqrt_p"], [164, 0, 1, "c.i16_rsqrt_s", "i16_rsqrt_s"], [166, 0, 1, "c.i16_scalefusion_p", "i16_scalefusion_p"], [166, 0, 1, "c.i16_scalefusion_s", "i16_scalefusion_s"], [167, 0, 1, "c.i16_scatter_elements_p", "i16_scatter_elements_p"], [167, 0, 1, "c.i16_scatter_elements_s", "i16_scatter_elements_s"], [168, 0, 1, "c.i16_scatter_nd_p", "i16_scatter_nd_p"], [168, 0, 1, "c.i16_scatter_nd_s", "i16_scatter_nd_s"], [169, 0, 1, "c.i16_scatter_nd_update_p", "i16_scatter_nd_update_p"], [169, 0, 1, "c.i16_scatter_nd_update_s", "i16_scatter_nd_update_s"], [170, 0, 1, "c.i16_select_p", "i16_select_p"], [170, 0, 1, "c.i16_select_s", "i16_select_s"], [175, 0, 1, "c.i16_sin_p", "i16_sin_p"], [175, 0, 1, "c.i16_sin_s", "i16_sin_s"], [178, 0, 1, "c.i16_slice_p", "i16_slice_p"], [178, 0, 1, "c.i16_slice_s", "i16_slice_s"], [183, 0, 1, "c.i16_spacetobatch_p", "i16_spacetobatch_p"], [183, 0, 1, "c.i16_spacetobatch_s", "i16_spacetobatch_s"], [184, 0, 1, "c.i16_spacetobatchnd_p", "i16_spacetobatchnd_p"], [184, 0, 1, "c.i16_spacetobatchnd_s", "i16_spacetobatchnd_s"], [185, 0, 1, "c.i16_spacetodepth_p", "i16_spacetodepth_p"], [185, 0, 1, "c.i16_spacetodepth_s", "i16_spacetodepth_s"], [187, 0, 1, "c.i16_sparsefillemptyrows_p", "i16_sparsefillemptyrows_p"], [187, 0, 1, "c.i16_sparsefillemptyrows_s", "i16_sparsefillemptyrows_s"], [189, 0, 1, "c.i16_sparsesegmentsum_p", "i16_sparsesegmentsum_p"], [189, 0, 1, "c.i16_sparsesegmentsum_s", "i16_sparsesegmentsum_s"], [190, 0, 1, "c.i16_sparsetodense_p", "i16_sparsetodense_p"], [190, 0, 1, "c.i16_sparsetodense_s", "i16_sparsetodense_s"], [191, 0, 1, "c.i16_splice_p", "i16_splice_p"], [191, 0, 1, "c.i16_splice_s", "i16_splice_s"], [192, 0, 1, "c.i16_split_p", "i16_split_p"], [192, 0, 1, "c.i16_split_s", "i16_split_s"], [193, 0, 1, "c.i16_split_with_overlap_p", "i16_split_with_overlap_p"], [193, 0, 1, "c.i16_split_with_overlap_s", "i16_split_with_overlap_s"], [194, 0, 1, "c.i16_sqrt_p", "i16_sqrt_p"], [194, 0, 1, "c.i16_sqrt_s", "i16_sqrt_s"], [195, 0, 1, "c.i16_sqrtgrad_p", "i16_sqrtgrad_p"], [195, 0, 1, "c.i16_sqrtgrad_s", "i16_sqrtgrad_s"], [196, 0, 1, "c.i16_square_p", "i16_square_p"], [196, 0, 1, "c.i16_square_s", "i16_square_s"], [197, 0, 1, "c.i16_squaredifference_p", "i16_squaredifference_p"], [197, 0, 1, "c.i16_squaredifference_s", "i16_squaredifference_s"], [199, 0, 1, "c.i16_stack_p", "i16_stack_p"], [199, 0, 1, "c.i16_stack_s", "i16_stack_s"], [202, 0, 1, "c.i16_subrelu6_p", "i16_subrelu6_p"], [202, 0, 1, "c.i16_subrelu6_s", "i16_subrelu6_s"], [202, 0, 1, "c.i16_subrelu_p", "i16_subrelu_p"], [202, 0, 1, "c.i16_subrelu_s", "i16_subrelu_s"], [206, 0, 1, "c.i16_tensor_scatter_add_p", "i16_tensor_scatter_add_p"], [206, 0, 1, "c.i16_tensor_scatter_add_s", "i16_tensor_scatter_add_s"], [208, 0, 1, "c.i16_tensorarrayread_p", "i16_tensorarrayread_p"], [208, 0, 1, "c.i16_tensorarrayread_s", "i16_tensorarrayread_s"], [210, 0, 1, "c.i16_tensorlistfromtensor_p", "i16_tensorlistfromtensor_p"], [210, 0, 1, "c.i16_tensorlistfromtensor_s", "i16_tensorlistfromtensor_s"], [215, 0, 1, "c.i16_tile_p", "i16_tile_p"], [215, 0, 1, "c.i16_tile_s", "i16_tile_s"], [216, 0, 1, "c.i16_topk_fusion_p", "i16_topk_fusion_p"], [216, 0, 1, "c.i16_topk_fusion_s", "i16_topk_fusion_s"], [217, 0, 1, "c.i16_transpose_p", "i16_transpose_p"], [217, 0, 1, "c.i16_transpose_s", "i16_transpose_s"], [218, 0, 1, "c.i16_tril_p", "i16_tril_p"], [218, 0, 1, "c.i16_tril_s", "i16_tril_s"], [219, 0, 1, "c.i16_triu_p", "i16_triu_p"], [219, 0, 1, "c.i16_triu_s", "i16_triu_s"], [222, 0, 1, "c.i16_unsorted_segment_sum_p", "i16_unsorted_segment_sum_p"], [222, 0, 1, "c.i16_unsorted_segment_sum_s", "i16_unsorted_segment_sum_s"], [225, 0, 1, "c.i16_where_p", "i16_where_p"], [225, 0, 1, "c.i16_where_s", "i16_where_s"], [226, 0, 1, "c.i16_zerolike_p", "i16_zerolike_p"], [226, 0, 1, "c.i16_zerolike_s", "i16_zerolike_s"], [221, 0, 1, "c.i32_Unique_p", "i32_Unique_p"], [221, 0, 1, "c.i32_Unique_s", "i32_Unique_s"], [10, 0, 1, "c.i32_abs_p", "i32_abs_p"], [10, 0, 1, "c.i32_abs_s", "i32_abs_s"], [17, 0, 1, "c.i32_addext_p", "i32_addext_p"], [17, 0, 1, "c.i32_addext_s", "i32_addext_s"], [19, 0, 1, "c.i32_addn_p", "i32_addn_p"], [19, 0, 1, "c.i32_addn_s", "i32_addn_s"], [17, 0, 1, "c.i32_addrelu6_p", "i32_addrelu6_p"], [17, 0, 1, "c.i32_addrelu6_s", "i32_addrelu6_s"], [17, 0, 1, "c.i32_addrelu_p", "i32_addrelu_p"], [17, 0, 1, "c.i32_addrelu_s", "i32_addrelu_s"], [22, 0, 1, "c.i32_allgather_p", "i32_allgather_p"], [22, 0, 1, "c.i32_allgather_s", "i32_allgather_s"], [112, 0, 1, "c.i32_and_p", "i32_and_p"], [112, 0, 1, "c.i32_and_s", "i32_and_s"], [27, 0, 1, "c.i32_assign_p", "i32_assign_p"], [27, 0, 1, "c.i32_assign_s", "i32_assign_s"], [28, 0, 1, "c.i32_assignadd_p", "i32_assignadd_p"], [28, 0, 1, "c.i32_assignadd_s", "i32_assignadd_s"], [35, 0, 1, "c.i32_batchtospace_p", "i32_batchtospace_p"], [35, 0, 1, "c.i32_batchtospace_s", "i32_batchtospace_s"], [36, 0, 1, "c.i32_batchtospacend_p", "i32_batchtospacend_p"], [36, 0, 1, "c.i32_batchtospacend_s", "i32_batchtospacend_s"], [37, 0, 1, "c.i32_biasadd_p", "i32_biasadd_p"], [37, 0, 1, "c.i32_biasadd_s", "i32_biasadd_s"], [41, 0, 1, "c.i32_broadcastto_p", "i32_broadcastto_p"], [41, 0, 1, "c.i32_broadcastto_s", "i32_broadcastto_s"], [44, 0, 1, "c.i32_clip_p", "i32_clip_p"], [44, 0, 1, "c.i32_clip_s", "i32_clip_s"], [45, 0, 1, "c.i32_concat_p", "i32_concat_p"], [45, 0, 1, "c.i32_concat_s", "i32_concat_s"], [46, 0, 1, "c.i32_constant_of_shape_p", "i32_constant_of_shape_p"], [46, 0, 1, "c.i32_constant_of_shape_s", "i32_constant_of_shape_s"], [51, 0, 1, "c.i32_cos_p", "i32_cos_p"], [51, 0, 1, "c.i32_cos_s", "i32_cos_s"], [54, 0, 1, "c.i32_cumsum_p", "i32_cumsum_p"], [54, 0, 1, "c.i32_cumsum_s", "i32_cumsum_s"], [59, 0, 1, "c.i32_depthtospace_p", "i32_depthtospace_p"], [59, 0, 1, "c.i32_depthtospace_s", "i32_depthtospace_s"], [61, 0, 1, "c.i32_div_fusion_p", "i32_div_fusion_p"], [61, 0, 1, "c.i32_div_fusion_s", "i32_div_fusion_s"], [67, 0, 1, "c.i32_eltwise_p", "i32_eltwise_p"], [67, 0, 1, "c.i32_eltwise_s", "i32_eltwise_s"], [70, 0, 1, "c.i32_equal_p", "i32_equal_p"], [70, 0, 1, "c.i32_equal_s", "i32_equal_s"], [73, 0, 1, "c.i32_expfusion_p", "i32_expfusion_p"], [73, 0, 1, "c.i32_expfusion_s", "i32_expfusion_s"], [55, 0, 1, "c.i32_extract_features_p", "i32_extract_features_p"], [55, 0, 1, "c.i32_extract_features_s", "i32_extract_features_s"], [78, 0, 1, "c.i32_fill_p", "i32_fill_p"], [78, 0, 1, "c.i32_fill_s", "i32_fill_s"], [85, 0, 1, "c.i32_formattranspose_p", "i32_formattranspose_p"], [85, 0, 1, "c.i32_formattranspose_s", "i32_formattranspose_s"], [89, 0, 1, "c.i32_gather_nd_p", "i32_gather_nd_p"], [89, 0, 1, "c.i32_gather_nd_s", "i32_gather_nd_s"], [88, 0, 1, "c.i32_gather_p", "i32_gather_p"], [88, 0, 1, "c.i32_gather_s", "i32_gather_s"], [90, 0, 1, "c.i32_gatherd_p", "i32_gatherd_p"], [90, 0, 1, "c.i32_gatherd_s", "i32_gatherd_s"], [92, 0, 1, "c.i32_greater_p", "i32_greater_p"], [92, 0, 1, "c.i32_greater_s", "i32_greater_s"], [93, 0, 1, "c.i32_greaterequal_p", "i32_greaterequal_p"], [93, 0, 1, "c.i32_greaterequal_s", "i32_greaterequal_s"], [96, 0, 1, "c.i32_hashtablelookup_p", "i32_hashtablelookup_p"], [96, 0, 1, "c.i32_hashtablelookup_s", "i32_hashtablelookup_s"], [98, 0, 1, "c.i32_invertpermutation_p", "i32_invertpermutation_p"], [98, 0, 1, "c.i32_invertpermutation_s", "i32_invertpermutation_s"], [99, 0, 1, "c.i32_isfinite_p", "i32_isfinite_p"], [99, 0, 1, "c.i32_isfinite_s", "i32_isfinite_s"], [104, 0, 1, "c.i32_less_p", "i32_less_p"], [104, 0, 1, "c.i32_less_s", "i32_less_s"], [105, 0, 1, "c.i32_lessequal_p", "i32_lessequal_p"], [105, 0, 1, "c.i32_lessequal_s", "i32_lessequal_s"], [108, 0, 1, "c.i32_log1p_p", "i32_log1p_p"], [108, 0, 1, "c.i32_log1p_s", "i32_log1p_s"], [107, 0, 1, "c.i32_log_p", "i32_log_p"], [107, 0, 1, "c.i32_log_s", "i32_log_s"], [110, 0, 1, "c.i32_logical_not_p", "i32_logical_not_p"], [110, 0, 1, "c.i32_logical_not_s", "i32_logical_not_s"], [111, 0, 1, "c.i32_logical_or_p", "i32_logical_or_p"], [111, 0, 1, "c.i32_logical_or_s", "i32_logical_or_s"], [116, 0, 1, "c.i32_lsh_projection_p", "i32_lsh_projection_p"], [116, 0, 1, "c.i32_lsh_projection_s", "i32_lsh_projection_s"], [121, 0, 1, "c.i32_matmulfusion_p", "i32_matmulfusion_p"], [121, 0, 1, "c.i32_matmulfusion_s", "i32_matmulfusion_s"], [122, 0, 1, "c.i32_maximum_p", "i32_maximum_p"], [122, 0, 1, "c.i32_maximum_s", "i32_maximum_s"], [127, 0, 1, "c.i32_minimum_p", "i32_minimum_p"], [127, 0, 1, "c.i32_minimum_s", "i32_minimum_s"], [129, 0, 1, "c.i32_mod_p", "i32_mod_p"], [129, 0, 1, "c.i32_mod_s", "i32_mod_s"], [130, 0, 1, "c.i32_mul_p", "i32_mul_p"], [130, 0, 1, "c.i32_mul_s", "i32_mul_s"], [133, 0, 1, "c.i32_neg_grad_p", "i32_neg_grad_p"], [133, 0, 1, "c.i32_neg_grad_s", "i32_neg_grad_s"], [132, 0, 1, "c.i32_neg_p", "i32_neg_p"], [132, 0, 1, "c.i32_neg_s", "i32_neg_s"], [137, 0, 1, "c.i32_nonzero_p", "i32_nonzero_p"], [137, 0, 1, "c.i32_nonzero_s", "i32_nonzero_s"], [138, 0, 1, "c.i32_not_equal_p", "i32_not_equal_p"], [138, 0, 1, "c.i32_not_equal_s", "i32_not_equal_s"], [139, 0, 1, "c.i32_onehot_p", "i32_onehot_p"], [139, 0, 1, "c.i32_onehot_s", "i32_onehot_s"], [140, 0, 1, "c.i32_ones_like_p", "i32_ones_like_p"], [140, 0, 1, "c.i32_ones_like_s", "i32_ones_like_s"], [141, 0, 1, "c.i32_padfusion_p", "i32_padfusion_p"], [141, 0, 1, "c.i32_padfusion_s", "i32_padfusion_s"], [142, 0, 1, "c.i32_pow_fusion_p", "i32_pow_fusion_p"], [142, 0, 1, "c.i32_pow_fusion_s", "i32_pow_fusion_s"], [147, 0, 1, "c.i32_raggedrange_p", "i32_raggedrange_p"], [147, 0, 1, "c.i32_raggedrange_s", "i32_raggedrange_s"], [150, 0, 1, "c.i32_range_p", "i32_range_p"], [150, 0, 1, "c.i32_range_s", "i32_range_s"], [152, 0, 1, "c.i32_real_div_p", "i32_real_div_p"], [152, 0, 1, "c.i32_real_div_s", "i32_real_div_s"], [153, 0, 1, "c.i32_reciprocal_p", "i32_reciprocal_p"], [153, 0, 1, "c.i32_reciprocal_s", "i32_reciprocal_s"], [154, 0, 1, "c.i32_reduce_p", "i32_reduce_p"], [154, 0, 1, "c.i32_reduce_s", "i32_reduce_s"], [21, 0, 1, "c.i32_reduceall_p", "i32_reduceall_p"], [21, 0, 1, "c.i32_reduceall_s", "i32_reduceall_s"], [155, 0, 1, "c.i32_reducescatter_p", "i32_reducescatter_p"], [155, 0, 1, "c.i32_reducescatter_s", "i32_reducescatter_s"], [156, 0, 1, "c.i32_reshape_p", "i32_reshape_p"], [156, 0, 1, "c.i32_reshape_s", "i32_reshape_s"], [164, 0, 1, "c.i32_rsqrt_p", "i32_rsqrt_p"], [164, 0, 1, "c.i32_rsqrt_s", "i32_rsqrt_s"], [166, 0, 1, "c.i32_scalefusion_p", "i32_scalefusion_p"], [166, 0, 1, "c.i32_scalefusion_s", "i32_scalefusion_s"], [167, 0, 1, "c.i32_scatter_elements_p", "i32_scatter_elements_p"], [167, 0, 1, "c.i32_scatter_elements_s", "i32_scatter_elements_s"], [168, 0, 1, "c.i32_scatter_nd_p", "i32_scatter_nd_p"], [168, 0, 1, "c.i32_scatter_nd_s", "i32_scatter_nd_s"], [169, 0, 1, "c.i32_scatter_nd_update_p", "i32_scatter_nd_update_p"], [169, 0, 1, "c.i32_scatter_nd_update_s", "i32_scatter_nd_update_s"], [170, 0, 1, "c.i32_select_p", "i32_select_p"], [170, 0, 1, "c.i32_select_s", "i32_select_s"], [175, 0, 1, "c.i32_sin_p", "i32_sin_p"], [175, 0, 1, "c.i32_sin_s", "i32_sin_s"], [178, 0, 1, "c.i32_slice_p", "i32_slice_p"], [178, 0, 1, "c.i32_slice_s", "i32_slice_s"], [183, 0, 1, "c.i32_spacetobatch_p", "i32_spacetobatch_p"], [183, 0, 1, "c.i32_spacetobatch_s", "i32_spacetobatch_s"], [184, 0, 1, "c.i32_spacetobatchnd_p", "i32_spacetobatchnd_p"], [184, 0, 1, "c.i32_spacetobatchnd_s", "i32_spacetobatchnd_s"], [185, 0, 1, "c.i32_spacetodepth_p", "i32_spacetodepth_p"], [185, 0, 1, "c.i32_spacetodepth_s", "i32_spacetodepth_s"], [187, 0, 1, "c.i32_sparsefillemptyrows_p", "i32_sparsefillemptyrows_p"], [187, 0, 1, "c.i32_sparsefillemptyrows_s", "i32_sparsefillemptyrows_s"], [189, 0, 1, "c.i32_sparsesegmentsum_p", "i32_sparsesegmentsum_p"], [189, 0, 1, "c.i32_sparsesegmentsum_s", "i32_sparsesegmentsum_s"], [190, 0, 1, "c.i32_sparsetodense_p", "i32_sparsetodense_p"], [190, 0, 1, "c.i32_sparsetodense_s", "i32_sparsetodense_s"], [191, 0, 1, "c.i32_splice_p", "i32_splice_p"], [191, 0, 1, "c.i32_splice_s", "i32_splice_s"], [192, 0, 1, "c.i32_split_p", "i32_split_p"], [192, 0, 1, "c.i32_split_s", "i32_split_s"], [193, 0, 1, "c.i32_split_with_overlap_p", "i32_split_with_overlap_p"], [193, 0, 1, "c.i32_split_with_overlap_s", "i32_split_with_overlap_s"], [194, 0, 1, "c.i32_sqrt_p", "i32_sqrt_p"], [194, 0, 1, "c.i32_sqrt_s", "i32_sqrt_s"], [195, 0, 1, "c.i32_sqrtgrad_p", "i32_sqrtgrad_p"], [195, 0, 1, "c.i32_sqrtgrad_s", "i32_sqrtgrad_s"], [196, 0, 1, "c.i32_square_p", "i32_square_p"], [196, 0, 1, "c.i32_square_s", "i32_square_s"], [197, 0, 1, "c.i32_squaredifference_p", "i32_squaredifference_p"], [197, 0, 1, "c.i32_squaredifference_s", "i32_squaredifference_s"], [199, 0, 1, "c.i32_stack_p", "i32_stack_p"], [199, 0, 1, "c.i32_stack_s", "i32_stack_s"], [202, 0, 1, "c.i32_subrelu6_p", "i32_subrelu6_p"], [202, 0, 1, "c.i32_subrelu6_s", "i32_subrelu6_s"], [202, 0, 1, "c.i32_subrelu_p", "i32_subrelu_p"], [202, 0, 1, "c.i32_subrelu_s", "i32_subrelu_s"], [206, 0, 1, "c.i32_tensor_scatter_add_p", "i32_tensor_scatter_add_p"], [206, 0, 1, "c.i32_tensor_scatter_add_s", "i32_tensor_scatter_add_s"], [208, 0, 1, "c.i32_tensorarrayread_p", "i32_tensorarrayread_p"], [208, 0, 1, "c.i32_tensorarrayread_s", "i32_tensorarrayread_s"], [210, 0, 1, "c.i32_tensorlistfromtensor_p", "i32_tensorlistfromtensor_p"], [210, 0, 1, "c.i32_tensorlistfromtensor_s", "i32_tensorlistfromtensor_s"], [215, 0, 1, "c.i32_tile_p", "i32_tile_p"], [215, 0, 1, "c.i32_tile_s", "i32_tile_s"], [216, 0, 1, "c.i32_topk_fusion_p", "i32_topk_fusion_p"], [216, 0, 1, "c.i32_topk_fusion_s", "i32_topk_fusion_s"], [217, 0, 1, "c.i32_transpose_p", "i32_transpose_p"], [217, 0, 1, "c.i32_transpose_s", "i32_transpose_s"], [218, 0, 1, "c.i32_tril_p", "i32_tril_p"], [218, 0, 1, "c.i32_tril_s", "i32_tril_s"], [219, 0, 1, "c.i32_triu_p", "i32_triu_p"], [219, 0, 1, "c.i32_triu_s", "i32_triu_s"], [222, 0, 1, "c.i32_unsorted_segment_sum_p", "i32_unsorted_segment_sum_p"], [222, 0, 1, "c.i32_unsorted_segment_sum_s", "i32_unsorted_segment_sum_s"], [225, 0, 1, "c.i32_where_p", "i32_where_p"], [225, 0, 1, "c.i32_where_s", "i32_where_s"], [226, 0, 1, "c.i32_zerolike_p", "i32_zerolike_p"], [226, 0, 1, "c.i32_zerolike_s", "i32_zerolike_s"], [95, 0, 1, "c.i8_Gru_p", "i8_Gru_p"], [95, 0, 1, "c.i8_Gru_s", "i8_Gru_s"], [221, 0, 1, "c.i8_Unique_p", "i8_Unique_p"], [221, 0, 1, "c.i8_Unique_s", "i8_Unique_s"], [10, 0, 1, "c.i8_abs_p", "i8_abs_p"], [10, 0, 1, "c.i8_abs_s", "i8_abs_s"], [16, 0, 1, "c.i8_adder_p", "i8_adder_p"], [16, 0, 1, "c.i8_adder_s", "i8_adder_s"], [17, 0, 1, "c.i8_addext_p", "i8_addext_p"], [17, 0, 1, "c.i8_addext_s", "i8_addext_s"], [19, 0, 1, "c.i8_addn_p", "i8_addn_p"], [19, 0, 1, "c.i8_addn_s", "i8_addn_s"], [17, 0, 1, "c.i8_addrelu6_p", "i8_addrelu6_p"], [17, 0, 1, "c.i8_addrelu6_s", "i8_addrelu6_s"], [17, 0, 1, "c.i8_addrelu_p", "i8_addrelu_p"], [17, 0, 1, "c.i8_addrelu_s", "i8_addrelu_s"], [20, 0, 1, "c.i8_affine_p", "i8_affine_p"], [20, 0, 1, "c.i8_affine_s", "i8_affine_s"], [22, 0, 1, "c.i8_allgather_p", "i8_allgather_p"], [22, 0, 1, "c.i8_allgather_s", "i8_allgather_s"], [112, 0, 1, "c.i8_and_p", "i8_and_p"], [112, 0, 1, "c.i8_and_s", "i8_and_s"], [27, 0, 1, "c.i8_assign_p", "i8_assign_p"], [27, 0, 1, "c.i8_assign_s", "i8_assign_s"], [28, 0, 1, "c.i8_assignadd_p", "i8_assignadd_p"], [28, 0, 1, "c.i8_assignadd_s", "i8_assignadd_s"], [31, 0, 1, "c.i8_avgpool_fusion_p", "i8_avgpool_fusion_p"], [31, 0, 1, "c.i8_avgpool_fusion_s", "i8_avgpool_fusion_s"], [33, 0, 1, "c.i8_batchnorm_p", "i8_batchnorm_p"], [33, 0, 1, "c.i8_batchnorm_s", "i8_batchnorm_s"], [35, 0, 1, "c.i8_batchtospace_p", "i8_batchtospace_p"], [35, 0, 1, "c.i8_batchtospace_s", "i8_batchtospace_s"], [36, 0, 1, "c.i8_batchtospacend_p", "i8_batchtospacend_p"], [36, 0, 1, "c.i8_batchtospacend_s", "i8_batchtospacend_s"], [37, 0, 1, "c.i8_biasadd_p", "i8_biasadd_p"], [37, 0, 1, "c.i8_biasadd_s", "i8_biasadd_s"], [39, 0, 1, "c.i8_binarycrossentropy_p", "i8_binarycrossentropy_p"], [39, 0, 1, "c.i8_binarycrossentropy_s", "i8_binarycrossentropy_s"], [41, 0, 1, "c.i8_broadcastto_p", "i8_broadcastto_p"], [41, 0, 1, "c.i8_broadcastto_s", "i8_broadcastto_s"], [12, 0, 1, "c.i8_celu_p", "i8_celu_p"], [12, 0, 1, "c.i8_celu_s", "i8_celu_s"], [12, 0, 1, "c.i8_clip_p", "i8_clip_p"], [12, 0, 1, "c.i8_clip_s", "i8_clip_s"], [45, 0, 1, "c.i8_concat_p", "i8_concat_p"], [45, 0, 1, "c.i8_concat_s", "i8_concat_s"], [46, 0, 1, "c.i8_constant_of_shape_p", "i8_constant_of_shape_p"], [46, 0, 1, "c.i8_constant_of_shape_s", "i8_constant_of_shape_s"], [47, 0, 1, "c.i8_conv2d_p", "i8_conv2d_p"], [47, 0, 1, "c.i8_conv2d_s", "i8_conv2d_s"], [48, 0, 1, "c.i8_convtranspose_p", "i8_convtranspose_p"], [48, 0, 1, "c.i8_convtranspose_s", "i8_convtranspose_s"], [51, 0, 1, "c.i8_cos_p", "i8_cos_p"], [51, 0, 1, "c.i8_cos_s", "i8_cos_s"], [53, 0, 1, "c.i8_crop_and_resize_anycore", "i8_crop_and_resize_anycore"], [54, 0, 1, "c.i8_cumsum_p", "i8_cumsum_p"], [54, 0, 1, "c.i8_cumsum_s", "i8_cumsum_s"], [59, 0, 1, "c.i8_depthtospace_p", "i8_depthtospace_p"], [59, 0, 1, "c.i8_depthtospace_s", "i8_depthtospace_s"], [60, 0, 1, "c.i8_detection_post_process_p", "i8_detection_post_process_p"], [60, 0, 1, "c.i8_detection_post_process_s", "i8_detection_post_process_s"], [61, 0, 1, "c.i8_div_fusion_p", "i8_div_fusion_p"], [61, 0, 1, "c.i8_div_fusion_s", "i8_div_fusion_s"], [67, 0, 1, "c.i8_eltwise_p", "i8_eltwise_p"], [67, 0, 1, "c.i8_eltwise_s", "i8_eltwise_s"], [12, 0, 1, "c.i8_elu_p", "i8_elu_p"], [12, 0, 1, "c.i8_elu_s", "i8_elu_s"], [70, 0, 1, "c.i8_equal_p", "i8_equal_p"], [70, 0, 1, "c.i8_equal_s", "i8_equal_s"], [73, 0, 1, "c.i8_expfusion_p", "i8_expfusion_p"], [73, 0, 1, "c.i8_expfusion_s", "i8_expfusion_s"], [55, 0, 1, "c.i8_extract_features_p", "i8_extract_features_p"], [55, 0, 1, "c.i8_extract_features_s", "i8_extract_features_s"], [78, 0, 1, "c.i8_fill_p", "i8_fill_p"], [78, 0, 1, "c.i8_fill_s", "i8_fill_s"], [85, 0, 1, "c.i8_formattranspose_p", "i8_formattranspose_p"], [85, 0, 1, "c.i8_formattranspose_s", "i8_formattranspose_s"], [86, 0, 1, "c.i8_fullconnection_p", "i8_fullconnection_p"], [86, 0, 1, "c.i8_fullconnection_s", "i8_fullconnection_s"], [89, 0, 1, "c.i8_gather_nd_p", "i8_gather_nd_p"], [89, 0, 1, "c.i8_gather_nd_s", "i8_gather_nd_s"], [88, 0, 1, "c.i8_gather_p", "i8_gather_p"], [88, 0, 1, "c.i8_gather_s", "i8_gather_s"], [90, 0, 1, "c.i8_gatherd_p", "i8_gatherd_p"], [90, 0, 1, "c.i8_gatherd_s", "i8_gatherd_s"], [12, 0, 1, "c.i8_gelu_p", "i8_gelu_p"], [12, 0, 1, "c.i8_gelu_s", "i8_gelu_s"], [91, 0, 1, "c.i8_glu_p", "i8_glu_p"], [91, 0, 1, "c.i8_glu_s", "i8_glu_s"], [92, 0, 1, "c.i8_greater_p", "i8_greater_p"], [92, 0, 1, "c.i8_greater_s", "i8_greater_s"], [93, 0, 1, "c.i8_greaterequal_p", "i8_greaterequal_p"], [93, 0, 1, "c.i8_greaterequal_s", "i8_greaterequal_s"], [12, 0, 1, "c.i8_hardshrink_p", "i8_hardshrink_p"], [12, 0, 1, "c.i8_hardshrink_s", "i8_hardshrink_s"], [12, 0, 1, "c.i8_hardtanh_p", "i8_hardtanh_p"], [12, 0, 1, "c.i8_hardtanh_s", "i8_hardtanh_s"], [12, 0, 1, "c.i8_hsigmoid_p", "i8_hsigmoid_p"], [12, 0, 1, "c.i8_hsigmoid_s", "i8_hsigmoid_s"], [12, 0, 1, "c.i8_hswish_p", "i8_hswish_p"], [12, 0, 1, "c.i8_hswish_s", "i8_hswish_s"], [98, 0, 1, "c.i8_invertpermutation_p", "i8_invertpermutation_p"], [98, 0, 1, "c.i8_invertpermutation_s", "i8_invertpermutation_s"], [99, 0, 1, "c.i8_isfinite_p", "i8_isfinite_p"], [99, 0, 1, "c.i8_isfinite_s", "i8_isfinite_s"], [101, 0, 1, "c.i8_layernormfusion_p", "i8_layernormfusion_p"], [101, 0, 1, "c.i8_layernormfusion_s", "i8_layernormfusion_s"], [103, 0, 1, "c.i8_leaky_relu_p", "i8_leaky_relu_p"], [103, 0, 1, "c.i8_leaky_relu_s", "i8_leaky_relu_s"], [104, 0, 1, "c.i8_less_p", "i8_less_p"], [104, 0, 1, "c.i8_less_s", "i8_less_s"], [105, 0, 1, "c.i8_lessequal_p", "i8_lessequal_p"], [105, 0, 1, "c.i8_lessequal_s", "i8_lessequal_s"], [108, 0, 1, "c.i8_log1p_p", "i8_log1p_p"], [108, 0, 1, "c.i8_log1p_s", "i8_log1p_s"], [110, 0, 1, "c.i8_logical_not_p", "i8_logical_not_p"], [110, 0, 1, "c.i8_logical_not_s", "i8_logical_not_s"], [111, 0, 1, "c.i8_logical_or_p", "i8_logical_or_p"], [111, 0, 1, "c.i8_logical_or_s", "i8_logical_or_s"], [113, 0, 1, "c.i8_logsoftmax_p", "i8_logsoftmax_p"], [113, 0, 1, "c.i8_logsoftmax_s", "i8_logsoftmax_s"], [12, 0, 1, "c.i8_lrelu_p", "i8_lrelu_p"], [12, 0, 1, "c.i8_lrelu_s", "i8_lrelu_s"], [116, 0, 1, "c.i8_lsh_projection_p", "i8_lsh_projection_p"], [116, 0, 1, "c.i8_lsh_projection_s", "i8_lsh_projection_s"], [122, 0, 1, "c.i8_maximum_p", "i8_maximum_p"], [122, 0, 1, "c.i8_maximum_s", "i8_maximum_s"], [127, 0, 1, "c.i8_minimum_p", "i8_minimum_p"], [127, 0, 1, "c.i8_minimum_s", "i8_minimum_s"], [129, 0, 1, "c.i8_mod_p", "i8_mod_p"], [129, 0, 1, "c.i8_mod_s", "i8_mod_s"], [130, 0, 1, "c.i8_mul_p", "i8_mul_p"], [130, 0, 1, "c.i8_mul_s", "i8_mul_s"], [133, 0, 1, "c.i8_neg_grad_p", "i8_neg_grad_p"], [133, 0, 1, "c.i8_neg_grad_s", "i8_neg_grad_s"], [132, 0, 1, "c.i8_neg_p", "i8_neg_p"], [132, 0, 1, "c.i8_neg_s", "i8_neg_s"], [134, 0, 1, "c.i8_nllloss_p", "i8_nllloss_p"], [134, 0, 1, "c.i8_nllloss_s", "i8_nllloss_s"], [136, 0, 1, "c.i8_non_max_suppression_p", "i8_non_max_suppression_p"], [136, 0, 1, "c.i8_non_max_suppression_s", "i8_non_max_suppression_s"], [137, 0, 1, "c.i8_nonzero_p", "i8_nonzero_p"], [137, 0, 1, "c.i8_nonzero_s", "i8_nonzero_s"], [138, 0, 1, "c.i8_not_equal_p", "i8_not_equal_p"], [138, 0, 1, "c.i8_not_equal_s", "i8_not_equal_s"], [139, 0, 1, "c.i8_onehot_p", "i8_onehot_p"], [139, 0, 1, "c.i8_onehot_s", "i8_onehot_s"], [140, 0, 1, "c.i8_ones_like_p", "i8_ones_like_p"], [140, 0, 1, "c.i8_ones_like_s", "i8_ones_like_s"], [141, 0, 1, "c.i8_padfusion_p", "i8_padfusion_p"], [141, 0, 1, "c.i8_padfusion_s", "i8_padfusion_s"], [142, 0, 1, "c.i8_pow_fusion_p", "i8_pow_fusion_p"], [142, 0, 1, "c.i8_pow_fusion_s", "i8_pow_fusion_s"], [144, 0, 1, "c.i8_prelufusion_p", "i8_prelufusion_p"], [144, 0, 1, "c.i8_prelufusion_s", "i8_prelufusion_s"], [147, 0, 1, "c.i8_raggedrange_p", "i8_raggedrange_p"], [147, 0, 1, "c.i8_raggedrange_s", "i8_raggedrange_s"], [150, 0, 1, "c.i8_range_p", "i8_range_p"], [150, 0, 1, "c.i8_range_s", "i8_range_s"], [152, 0, 1, "c.i8_real_div_p", "i8_real_div_p"], [152, 0, 1, "c.i8_real_div_s", "i8_real_div_s"], [153, 0, 1, "c.i8_reciprocal_p", "i8_reciprocal_p"], [153, 0, 1, "c.i8_reciprocal_s", "i8_reciprocal_s"], [154, 0, 1, "c.i8_reduce_p", "i8_reduce_p"], [154, 0, 1, "c.i8_reduce_s", "i8_reduce_s"], [21, 0, 1, "c.i8_reduceall_p", "i8_reduceall_p"], [21, 0, 1, "c.i8_reduceall_s", "i8_reduceall_s"], [155, 0, 1, "c.i8_reducescatter_p", "i8_reducescatter_p"], [155, 0, 1, "c.i8_reducescatter_s", "i8_reducescatter_s"], [12, 0, 1, "c.i8_relu6_p", "i8_relu6_p"], [12, 0, 1, "c.i8_relu6_s", "i8_relu6_s"], [12, 0, 1, "c.i8_relu_p", "i8_relu_p"], [12, 0, 1, "c.i8_relu_s", "i8_relu_s"], [156, 0, 1, "c.i8_reshape_p", "i8_reshape_p"], [156, 0, 1, "c.i8_reshape_s", "i8_reshape_s"], [157, 0, 1, "c.i8_resize_anycore", "i8_resize_anycore"], [162, 0, 1, "c.i8_roipooling_p", "i8_roipooling_p"], [162, 0, 1, "c.i8_roipooling_s", "i8_roipooling_s"], [164, 0, 1, "c.i8_rsqrt_p", "i8_rsqrt_p"], [164, 0, 1, "c.i8_rsqrt_s", "i8_rsqrt_s"], [166, 0, 1, "c.i8_scalefusion_p", "i8_scalefusion_p"], [166, 0, 1, "c.i8_scalefusion_s", "i8_scalefusion_s"], [167, 0, 1, "c.i8_scatter_elements_p", "i8_scatter_elements_p"], [167, 0, 1, "c.i8_scatter_elements_s", "i8_scatter_elements_s"], [168, 0, 1, "c.i8_scatter_nd_p", "i8_scatter_nd_p"], [168, 0, 1, "c.i8_scatter_nd_s", "i8_scatter_nd_s"], [169, 0, 1, "c.i8_scatter_nd_update_p", "i8_scatter_nd_update_p"], [169, 0, 1, "c.i8_scatter_nd_update_s", "i8_scatter_nd_update_s"], [170, 0, 1, "c.i8_select_p", "i8_select_p"], [170, 0, 1, "c.i8_select_s", "i8_select_s"], [12, 0, 1, "c.i8_sigmoid_p", "i8_sigmoid_p"], [12, 0, 1, "c.i8_sigmoid_s", "i8_sigmoid_s"], [174, 0, 1, "c.i8_sigmoidcrossentropywithlogits_p", "i8_sigmoidcrossentropywithlogits_p"], [174, 0, 1, "c.i8_sigmoidcrossentropywithlogits_s", "i8_sigmoidcrossentropywithlogits_s"], [175, 0, 1, "c.i8_sin_p", "i8_sin_p"], [175, 0, 1, "c.i8_sin_s", "i8_sin_s"], [178, 0, 1, "c.i8_slice_p", "i8_slice_p"], [178, 0, 1, "c.i8_slice_s", "i8_slice_s"], [179, 0, 1, "c.i8_smoothl1loss_p", "i8_smoothl1loss_p"], [179, 0, 1, "c.i8_smoothl1loss_s", "i8_smoothl1loss_s"], [182, 0, 1, "c.i8_softmax_cross_entropy_with_logits_p", "i8_softmax_cross_entropy_with_logits_p"], [182, 0, 1, "c.i8_softmax_cross_entropy_with_logits_s", "i8_softmax_cross_entropy_with_logits_s"], [181, 0, 1, "c.i8_softmax_p", "i8_softmax_p"], [181, 0, 1, "c.i8_softmax_s", "i8_softmax_s"], [12, 0, 1, "c.i8_softplus_p", "i8_softplus_p"], [12, 0, 1, "c.i8_softplus_s", "i8_softplus_s"], [12, 0, 1, "c.i8_softshrink_p", "i8_softshrink_p"], [12, 0, 1, "c.i8_softshrink_s", "i8_softshrink_s"], [12, 0, 1, "c.i8_softsignopt_p", "i8_softsignopt_p"], [12, 0, 1, "c.i8_softsignopt_s", "i8_softsignopt_s"], [183, 0, 1, "c.i8_spacetobatch_p", "i8_spacetobatch_p"], [183, 0, 1, "c.i8_spacetobatch_s", "i8_spacetobatch_s"], [184, 0, 1, "c.i8_spacetobatchnd_p", "i8_spacetobatchnd_p"], [184, 0, 1, "c.i8_spacetobatchnd_s", "i8_spacetobatchnd_s"], [185, 0, 1, "c.i8_spacetodepth_p", "i8_spacetodepth_p"], [185, 0, 1, "c.i8_spacetodepth_s", "i8_spacetodepth_s"], [187, 0, 1, "c.i8_sparsefillemptyrows_p", "i8_sparsefillemptyrows_p"], [187, 0, 1, "c.i8_sparsefillemptyrows_s", "i8_sparsefillemptyrows_s"], [190, 0, 1, "c.i8_sparsetodense_p", "i8_sparsetodense_p"], [190, 0, 1, "c.i8_sparsetodense_s", "i8_sparsetodense_s"], [191, 0, 1, "c.i8_splice_p", "i8_splice_p"], [191, 0, 1, "c.i8_splice_s", "i8_splice_s"], [192, 0, 1, "c.i8_split_p", "i8_split_p"], [192, 0, 1, "c.i8_split_s", "i8_split_s"], [193, 0, 1, "c.i8_split_with_overlap_p", "i8_split_with_overlap_p"], [193, 0, 1, "c.i8_split_with_overlap_s", "i8_split_with_overlap_s"], [194, 0, 1, "c.i8_sqrt_p", "i8_sqrt_p"], [194, 0, 1, "c.i8_sqrt_s", "i8_sqrt_s"], [195, 0, 1, "c.i8_sqrtgrad_p", "i8_sqrtgrad_p"], [195, 0, 1, "c.i8_sqrtgrad_s", "i8_sqrtgrad_s"], [196, 0, 1, "c.i8_square_p", "i8_square_p"], [196, 0, 1, "c.i8_square_s", "i8_square_s"], [197, 0, 1, "c.i8_squaredifference_p", "i8_squaredifference_p"], [197, 0, 1, "c.i8_squaredifference_s", "i8_squaredifference_s"], [199, 0, 1, "c.i8_stack_p", "i8_stack_p"], [199, 0, 1, "c.i8_stack_s", "i8_stack_s"], [202, 0, 1, "c.i8_subrelu6_p", "i8_subrelu6_p"], [202, 0, 1, "c.i8_subrelu6_s", "i8_subrelu6_s"], [202, 0, 1, "c.i8_subrelu_p", "i8_subrelu_p"], [202, 0, 1, "c.i8_subrelu_s", "i8_subrelu_s"], [12, 0, 1, "c.i8_swish_p", "i8_swish_p"], [12, 0, 1, "c.i8_swish_s", "i8_swish_s"], [12, 0, 1, "c.i8_tanh_p", "i8_tanh_p"], [12, 0, 1, "c.i8_tanh_s", "i8_tanh_s"], [206, 0, 1, "c.i8_tensor_scatter_add_p", "i8_tensor_scatter_add_p"], [206, 0, 1, "c.i8_tensor_scatter_add_s", "i8_tensor_scatter_add_s"], [208, 0, 1, "c.i8_tensorarrayread_p", "i8_tensorarrayread_p"], [208, 0, 1, "c.i8_tensorarrayread_s", "i8_tensorarrayread_s"], [210, 0, 1, "c.i8_tensorlistfromtensor_p", "i8_tensorlistfromtensor_p"], [210, 0, 1, "c.i8_tensorlistfromtensor_s", "i8_tensorlistfromtensor_s"], [215, 0, 1, "c.i8_tile_p", "i8_tile_p"], [215, 0, 1, "c.i8_tile_s", "i8_tile_s"], [146, 0, 1, "c.i8_to_fp_dequant_p", "i8_to_fp_dequant_p"], [146, 0, 1, "c.i8_to_fp_dequant_s", "i8_to_fp_dequant_s"], [146, 0, 1, "c.i8_to_hp_dequant_p", "i8_to_hp_dequant_p"], [146, 0, 1, "c.i8_to_hp_dequant_s", "i8_to_hp_dequant_s"], [217, 0, 1, "c.i8_transpose_p", "i8_transpose_p"], [217, 0, 1, "c.i8_transpose_s", "i8_transpose_s"], [218, 0, 1, "c.i8_tril_p", "i8_tril_p"], [218, 0, 1, "c.i8_tril_s", "i8_tril_s"], [219, 0, 1, "c.i8_triu_p", "i8_triu_p"], [219, 0, 1, "c.i8_triu_s", "i8_triu_s"], [222, 0, 1, "c.i8_unsorted_segment_sum_p", "i8_unsorted_segment_sum_p"], [222, 0, 1, "c.i8_unsorted_segment_sum_s", "i8_unsorted_segment_sum_s"], [225, 0, 1, "c.i8_where_p", "i8_where_p"], [225, 0, 1, "c.i8_where_s", "i8_where_s"], [226, 0, 1, "c.i8_zerolike_p", "i8_zerolike_p"], [226, 0, 1, "c.i8_zerolike_s", "i8_zerolike_s"], [151, 0, 1, "c.rank_p", "rank_p"], [151, 0, 1, "c.rank_s", "rank_s"], [172, 0, 1, "c.shape", "shape"], [176, 0, 1, "c.size_p", "size_p"], [176, 0, 1, "c.size_s", "size_s"], [177, 0, 1, "c.skipgram_p", "skipgram_p"], [177, 0, 1, "c.skipgram_s", "skipgram_s"], [188, 0, 1, "c.sparsereshape_p", "sparsereshape_p"], [188, 0, 1, "c.sparsereshape_s", "sparsereshape_s"], [200, 0, 1, "c.stridedslice", "stridedslice"], [204, 0, 1, "c.switch_p", "switch_p"], [204, 0, 1, "c.switch_s", "switch_s"], [205, 0, 1, "c.switchlayer_p", "switchlayer_p"], [205, 0, 1, "c.switchlayer_s", "switchlayer_s"], [207, 0, 1, "c.tensorarrayread_p", "tensorarrayread_p"], [207, 0, 1, "c.tensorarrayread_s", "tensorarrayread_s"], [207, 0, 1, "c.tensorarraywrite_p", "tensorarraywrite_p"], [207, 0, 1, "c.tensorarraywrite_s", "tensorarraywrite_s"], [211, 0, 1, "c.tensorlistgetitem_p", "tensorlistgetitem_p"], [211, 0, 1, "c.tensorlistgetitem_s", "tensorlistgetitem_s"], [213, 0, 1, "c.tensorlistsetitem_p", "tensorlistsetitem_p"], [213, 0, 1, "c.tensorlistsetitem_s", "tensorlistsetitem_s"], [214, 0, 1, "c.tensorliststack_p", "tensorliststack_p"], [214, 0, 1, "c.tensorliststack_s", "tensorliststack_s"], [224, 0, 1, "c.unstack_p", "unstack_p"], [224, 0, 1, "c.unstack_s", "unstack_s"]], "Flatten": [[80, 1, 1, "c.Flatten", "core_mask"], [80, 1, 1, "c.Flatten", "input"], [80, 1, 1, "c.Flatten", "output"], [80, 1, 1, "c.Flatten", "param"]], "Tensorlistreserve": [[212, 1, 1, "c.Tensorlistreserve", "element_shape"], [212, 1, 1, "c.Tensorlistreserve", "num_elements"], [212, 1, 1, "c.Tensorlistreserve", "shape_size"], [212, 1, 1, "c.Tensorlistreserve", "tensor_c_shape"], [212, 1, 1, "c.Tensorlistreserve", "tensor_c_shape_size"], [212, 1, 1, "c.Tensorlistreserve", "tensor_list_c_element_shape"], [212, 1, 1, "c.Tensorlistreserve", "tensor_list_c_element_shape_size"]], "anytype_crop_anycore": [[52, 1, 1, "c.anytype_crop_anycore", "axis"], [52, 1, 1, "c.anytype_crop_anycore", "core_mask"], [52, 1, 1, "c.anytype_crop_anycore", "in_shape"], [52, 1, 1, "c.anytype_crop_anycore", "input"], [52, 1, 1, "c.anytype_crop_anycore", "offset"], [52, 1, 1, "c.anytype_crop_anycore", "out_shape"], [52, 1, 1, "c.anytype_crop_anycore", "output"], [52, 1, 1, "c.anytype_crop_anycore", "type_size"]], "anytype_expand_dims_anycore": [[72, 1, 1, "c.anytype_expand_dims_anycore", "core_mask"], [72, 1, 1, "c.anytype_expand_dims_anycore", "dst"], [72, 1, 1, "c.anytype_expand_dims_anycore", "src"], [72, 1, 1, "c.anytype_expand_dims_anycore", "total_copy_size"]], "anytype_fillv2_p": [[79, 1, 1, "c.anytype_fillv2_p", "core_mask"], [79, 1, 1, "c.anytype_fillv2_p", "length"], [79, 1, 1, "c.anytype_fillv2_p", "output"], [79, 1, 1, "c.anytype_fillv2_p", "type_size"], [79, 1, 1, "c.anytype_fillv2_p", "value"]], "anytype_fillv2_s": [[79, 1, 1, "c.anytype_fillv2_s", "core_mask"], [79, 1, 1, "c.anytype_fillv2_s", "length"], [79, 1, 1, "c.anytype_fillv2_s", "output"], [79, 1, 1, "c.anytype_fillv2_s", "type_size"], [79, 1, 1, "c.anytype_fillv2_s", "value"]], "anytype_reverse_sequence_anycore": [[159, 1, 1, "c.anytype_reverse_sequence_anycore", "core_mask"], [159, 1, 1, "c.anytype_reverse_sequence_anycore", "dst"], [159, 1, 1, "c.anytype_reverse_sequence_anycore", "param"], [159, 1, 1, "c.anytype_reverse_sequence_anycore", "seq_lengths"], [159, 1, 1, "c.anytype_reverse_sequence_anycore", "src"]], "anytype_reversev2_anycore": [[160, 1, 1, "c.anytype_reversev2_anycore", "core_mask"], [160, 1, 1, "c.anytype_reversev2_anycore", "dst"], [160, 1, 1, "c.anytype_reversev2_anycore", "param"], [160, 1, 1, "c.anytype_reversev2_anycore", "src"]], "anytype_squeeze_anycore": [[198, 1, 1, "c.anytype_squeeze_anycore", "core_mask"], [198, 1, 1, "c.anytype_squeeze_anycore", "dst"], [198, 1, 1, "c.anytype_squeeze_anycore", "src"], [198, 1, 1, "c.anytype_squeeze_anycore", "total_copy_size"]], "anytype_unsqueeze_anycore": [[223, 1, 1, "c.anytype_unsqueeze_anycore", "core_mask"], [223, 1, 1, "c.anytype_unsqueeze_anycore", "dst"], [223, 1, 1, "c.anytype_unsqueeze_anycore", "src"], [223, 1, 1, "c.anytype_unsqueeze_anycore", "total_copy_size"]], "assert": [[26, 1, 1, "c.assert", "Input"], [26, 1, 1, "c.assert", "output"]], "c128_Unique_p": [[221, 1, 1, "c.c128_Unique_p", "input"], [221, 1, 1, "c.c128_Unique_p", "input_len"], [221, 1, 1, "c.c128_Unique_p", "output0"], [221, 1, 1, "c.c128_Unique_p", "output0_len"]], "c128_Unique_s": [[221, 1, 1, "c.c128_Unique_s", "core_mask"], [221, 1, 1, "c.c128_Unique_s", "input"], [221, 1, 1, "c.c128_Unique_s", "input_len"], [221, 1, 1, "c.c128_Unique_s", "output0"], [221, 1, 1, "c.c128_Unique_s", "output0_len"]], "c128_abs_p": [[10, 1, 1, "c.c128_abs_p", "dst_data"], [10, 1, 1, "c.c128_abs_p", "length"], [10, 1, 1, "c.c128_abs_p", "src_data"]], "c128_abs_s": [[10, 1, 1, "c.c128_abs_s", "core_mask"], [10, 1, 1, "c.c128_abs_s", "dst_data"], [10, 1, 1, "c.c128_abs_s", "length"], [10, 1, 1, "c.c128_abs_s", "src_data"]], "c128_addext_p": [[17, 1, 1, "c.c128_addext_p", "alpha"], [17, 1, 1, "c.c128_addext_p", "in0"], [17, 1, 1, "c.c128_addext_p", "in1"], [17, 1, 1, "c.c128_addext_p", "out"], [17, 1, 1, "c.c128_addext_p", "size"]], "c128_addext_s": [[17, 1, 1, "c.c128_addext_s", "alpha"], [17, 1, 1, "c.c128_addext_s", "core_mask"], [17, 1, 1, "c.c128_addext_s", "in0"], [17, 1, 1, "c.c128_addext_s", "in1"], [17, 1, 1, "c.c128_addext_s", "out"], [17, 1, 1, "c.c128_addext_s", "size"]], "c128_addn_p": [[19, 1, 1, "c.c128_addn_p", "input0"], [19, 1, 1, "c.c128_addn_p", "input1"], [19, 1, 1, "c.c128_addn_p", "length"], [19, 1, 1, "c.c128_addn_p", "output"]], "c128_addn_s": [[19, 1, 1, "c.c128_addn_s", "core_mask"], [19, 1, 1, "c.c128_addn_s", "input0"], [19, 1, 1, "c.c128_addn_s", "input1"], [19, 1, 1, "c.c128_addn_s", "length"], [19, 1, 1, "c.c128_addn_s", "output"]], "c128_addrelu6_p": [[17, 1, 1, "c.c128_addrelu6_p", "in0"], [17, 1, 1, "c.c128_addrelu6_p", "in1"], [17, 1, 1, "c.c128_addrelu6_p", "out"], [17, 1, 1, "c.c128_addrelu6_p", "size"]], "c128_addrelu6_s": [[17, 1, 1, "c.c128_addrelu6_s", "core_mask"], [17, 1, 1, "c.c128_addrelu6_s", "in0"], [17, 1, 1, "c.c128_addrelu6_s", "in1"], [17, 1, 1, "c.c128_addrelu6_s", "out"], [17, 1, 1, "c.c128_addrelu6_s", "size"]], "c128_addrelu_p": [[17, 1, 1, "c.c128_addrelu_p", "in0"], [17, 1, 1, "c.c128_addrelu_p", "in1"], [17, 1, 1, "c.c128_addrelu_p", "out"], [17, 1, 1, "c.c128_addrelu_p", "size"]], "c128_addrelu_s": [[17, 1, 1, "c.c128_addrelu_s", "core_mask"], [17, 1, 1, "c.c128_addrelu_s", "in0"], [17, 1, 1, "c.c128_addrelu_s", "in1"], [17, 1, 1, "c.c128_addrelu_s", "out"], [17, 1, 1, "c.c128_addrelu_s", "size"]], "c128_allgather_p": [[22, 1, 1, "c.c128_allgather_p", "data_size"], [22, 1, 1, "c.c128_allgather_p", "input"], [22, 1, 1, "c.c128_allgather_p", "input_rank"], [22, 1, 1, "c.c128_allgather_p", "output"], [22, 1, 1, "c.c128_allgather_p", "output_rank"]], "c128_allgather_s": [[22, 1, 1, "c.c128_allgather_s", "core_mask"], [22, 1, 1, "c.c128_allgather_s", "data_size"], [22, 1, 1, "c.c128_allgather_s", "input"], [22, 1, 1, "c.c128_allgather_s", "input_rank"], [22, 1, 1, "c.c128_allgather_s", "output"], [22, 1, 1, "c.c128_allgather_s", "output_rank"]], "c128_assign_p": [[27, 1, 1, "c.c128_assign_p", "dst"], [27, 1, 1, "c.c128_assign_p", "length"], [27, 1, 1, "c.c128_assign_p", "src"]], "c128_assign_s": [[27, 1, 1, "c.c128_assign_s", "core_mask"], [27, 1, 1, "c.c128_assign_s", "dst"], [27, 1, 1, "c.c128_assign_s", "length"], [27, 1, 1, "c.c128_assign_s", "src"]], "c128_assignadd_p": [[28, 1, 1, "c.c128_assignadd_p", "input"], [28, 1, 1, "c.c128_assignadd_p", "length"], [28, 1, 1, "c.c128_assignadd_p", "output"]], "c128_assignadd_s": [[28, 1, 1, "c.c128_assignadd_s", "core_mask"], [28, 1, 1, "c.c128_assignadd_s", "input"], [28, 1, 1, "c.c128_assignadd_s", "length"], [28, 1, 1, "c.c128_assignadd_s", "output"]], "c128_batchtospace_p": [[35, 1, 1, "c.c128_batchtospace_p", "block_size"], [35, 1, 1, "c.c128_batchtospace_p", "crops"], [35, 1, 1, "c.c128_batchtospace_p", "data_size"], [35, 1, 1, "c.c128_batchtospace_p", "input"], [35, 1, 1, "c.c128_batchtospace_p", "input_shape"], [35, 1, 1, "c.c128_batchtospace_p", "output"]], "c128_batchtospace_s": [[35, 1, 1, "c.c128_batchtospace_s", "block_size"], [35, 1, 1, "c.c128_batchtospace_s", "core_mask"], [35, 1, 1, "c.c128_batchtospace_s", "crops"], [35, 1, 1, "c.c128_batchtospace_s", "data_size"], [35, 1, 1, "c.c128_batchtospace_s", "input"], [35, 1, 1, "c.c128_batchtospace_s", "input_shape"], [35, 1, 1, "c.c128_batchtospace_s", "output"]], "c128_batchtospacend_p": [[36, 1, 1, "c.c128_batchtospacend_p", "block_size"], [36, 1, 1, "c.c128_batchtospacend_p", "crops"], [36, 1, 1, "c.c128_batchtospacend_p", "data_size"], [36, 1, 1, "c.c128_batchtospacend_p", "input"], [36, 1, 1, "c.c128_batchtospacend_p", "input_shape"], [36, 1, 1, "c.c128_batchtospacend_p", "output"]], "c128_batchtospacend_s": [[36, 1, 1, "c.c128_batchtospacend_s", "block_size"], [36, 1, 1, "c.c128_batchtospacend_s", "core_mask"], [36, 1, 1, "c.c128_batchtospacend_s", "crops"], [36, 1, 1, "c.c128_batchtospacend_s", "data_size"], [36, 1, 1, "c.c128_batchtospacend_s", "input"], [36, 1, 1, "c.c128_batchtospacend_s", "input_shape"], [36, 1, 1, "c.c128_batchtospacend_s", "output"]], "c128_biasadd_p": [[37, 1, 1, "c.c128_biasadd_p", "data_format"], [37, 1, 1, "c.c128_biasadd_p", "dims"], [37, 1, 1, "c.c128_biasadd_p", "input_bias"], [37, 1, 1, "c.c128_biasadd_p", "input_x"], [37, 1, 1, "c.c128_biasadd_p", "length"], [37, 1, 1, "c.c128_biasadd_p", "output"], [37, 1, 1, "c.c128_biasadd_p", "shape_size"]], "c128_biasadd_s": [[37, 1, 1, "c.c128_biasadd_s", "core_mask"], [37, 1, 1, "c.c128_biasadd_s", "data_format"], [37, 1, 1, "c.c128_biasadd_s", "dims"], [37, 1, 1, "c.c128_biasadd_s", "input_bias"], [37, 1, 1, "c.c128_biasadd_s", "input_x"], [37, 1, 1, "c.c128_biasadd_s", "length"], [37, 1, 1, "c.c128_biasadd_s", "output"], [37, 1, 1, "c.c128_biasadd_s", "shape_size"]], "c128_broadcastto_p": [[41, 1, 1, "c.c128_broadcastto_p", "data_size"], [41, 1, 1, "c.c128_broadcastto_p", "input"], [41, 1, 1, "c.c128_broadcastto_p", "input_shape"], [41, 1, 1, "c.c128_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.c128_broadcastto_p", "output"], [41, 1, 1, "c.c128_broadcastto_p", "output_shape"], [41, 1, 1, "c.c128_broadcastto_p", "output_shape_size"]], "c128_broadcastto_s": [[41, 1, 1, "c.c128_broadcastto_s", "core_mask"], [41, 1, 1, "c.c128_broadcastto_s", "data_size"], [41, 1, 1, "c.c128_broadcastto_s", "input"], [41, 1, 1, "c.c128_broadcastto_s", "input_shape"], [41, 1, 1, "c.c128_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.c128_broadcastto_s", "output"], [41, 1, 1, "c.c128_broadcastto_s", "output_shape"], [41, 1, 1, "c.c128_broadcastto_s", "output_shape_size"]], "c128_concat_p": [[45, 1, 1, "c.c128_concat_p", "axis"], [45, 1, 1, "c.c128_concat_p", "input_ndim"], [45, 1, 1, "c.c128_concat_p", "input_shapes"], [45, 1, 1, "c.c128_concat_p", "inputs"], [45, 1, 1, "c.c128_concat_p", "num_inputs"], [45, 1, 1, "c.c128_concat_p", "output"]], "c128_concat_s": [[45, 1, 1, "c.c128_concat_s", "axis"], [45, 1, 1, "c.c128_concat_s", "core_mask"], [45, 1, 1, "c.c128_concat_s", "input_ndim"], [45, 1, 1, "c.c128_concat_s", "input_shapes"], [45, 1, 1, "c.c128_concat_s", "inputs"], [45, 1, 1, "c.c128_concat_s", "num_inputs"], [45, 1, 1, "c.c128_concat_s", "output"]], "c128_constant_of_shape_p": [[46, 1, 1, "c.c128_constant_of_shape_p", "end"], [46, 1, 1, "c.c128_constant_of_shape_p", "output"], [46, 1, 1, "c.c128_constant_of_shape_p", "start"], [46, 1, 1, "c.c128_constant_of_shape_p", "value_imag"], [46, 1, 1, "c.c128_constant_of_shape_p", "value_real"]], "c128_constant_of_shape_s": [[46, 1, 1, "c.c128_constant_of_shape_s", "core_mask"], [46, 1, 1, "c.c128_constant_of_shape_s", "end"], [46, 1, 1, "c.c128_constant_of_shape_s", "output"], [46, 1, 1, "c.c128_constant_of_shape_s", "start"], [46, 1, 1, "c.c128_constant_of_shape_s", "value_imag"], [46, 1, 1, "c.c128_constant_of_shape_s", "value_real"]], "c128_cumsum_p": [[54, 1, 1, "c.c128_cumsum_p", "axis_dim"], [54, 1, 1, "c.c128_cumsum_p", "exclusive"], [54, 1, 1, "c.c128_cumsum_p", "inner_dim"], [54, 1, 1, "c.c128_cumsum_p", "input"], [54, 1, 1, "c.c128_cumsum_p", "out_dim"], [54, 1, 1, "c.c128_cumsum_p", "output"]], "c128_cumsum_s": [[54, 1, 1, "c.c128_cumsum_s", "axis_dim"], [54, 1, 1, "c.c128_cumsum_s", "core_mask"], [54, 1, 1, "c.c128_cumsum_s", "exclusive"], [54, 1, 1, "c.c128_cumsum_s", "inner_dim"], [54, 1, 1, "c.c128_cumsum_s", "input"], [54, 1, 1, "c.c128_cumsum_s", "out_dim"], [54, 1, 1, "c.c128_cumsum_s", "output"]], "c128_depthtospace_p": [[59, 1, 1, "c.c128_depthtospace_p", "block_size"], [59, 1, 1, "c.c128_depthtospace_p", "data_size"], [59, 1, 1, "c.c128_depthtospace_p", "in_shape"], [59, 1, 1, "c.c128_depthtospace_p", "input"], [59, 1, 1, "c.c128_depthtospace_p", "output"]], "c128_depthtospace_s": [[59, 1, 1, "c.c128_depthtospace_s", "block_size"], [59, 1, 1, "c.c128_depthtospace_s", "core_mask"], [59, 1, 1, "c.c128_depthtospace_s", "data_size"], [59, 1, 1, "c.c128_depthtospace_s", "in_shape"], [59, 1, 1, "c.c128_depthtospace_s", "input"], [59, 1, 1, "c.c128_depthtospace_s", "output"]], "c128_div_fusion_p": [[61, 1, 1, "c.c128_div_fusion_p", "input0"], [61, 1, 1, "c.c128_div_fusion_p", "input1"], [61, 1, 1, "c.c128_div_fusion_p", "length"], [61, 1, 1, "c.c128_div_fusion_p", "output"]], "c128_div_fusion_s": [[61, 1, 1, "c.c128_div_fusion_s", "core_mask"], [61, 1, 1, "c.c128_div_fusion_s", "input0"], [61, 1, 1, "c.c128_div_fusion_s", "input1"], [61, 1, 1, "c.c128_div_fusion_s", "length"], [61, 1, 1, "c.c128_div_fusion_s", "output"]], "c128_eltwise_p": [[67, 1, 1, "c.c128_eltwise_p", "Input0"], [67, 1, 1, "c.c128_eltwise_p", "Input1"], [67, 1, 1, "c.c128_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.c128_eltwise_p", "length"], [67, 1, 1, "c.c128_eltwise_p", "output"]], "c128_eltwise_s": [[67, 1, 1, "c.c128_eltwise_s", "Input0"], [67, 1, 1, "c.c128_eltwise_s", "Input1"], [67, 1, 1, "c.c128_eltwise_s", "core_mask"], [67, 1, 1, "c.c128_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.c128_eltwise_s", "length"], [67, 1, 1, "c.c128_eltwise_s", "output"]], "c128_equal_p": [[70, 1, 1, "c.c128_equal_p", "Input0"], [70, 1, 1, "c.c128_equal_p", "Input1"], [70, 1, 1, "c.c128_equal_p", "length"], [70, 1, 1, "c.c128_equal_p", "output"]], "c128_equal_s": [[70, 1, 1, "c.c128_equal_s", "Input0"], [70, 1, 1, "c.c128_equal_s", "Input1"], [70, 1, 1, "c.c128_equal_s", "core_mask"], [70, 1, 1, "c.c128_equal_s", "length"], [70, 1, 1, "c.c128_equal_s", "output"]], "c128_expfusion_p": [[73, 1, 1, "c.c128_expfusion_p", "dst_data"], [73, 1, 1, "c.c128_expfusion_p", "in_scale"], [73, 1, 1, "c.c128_expfusion_p", "length"], [73, 1, 1, "c.c128_expfusion_p", "out_scale"], [73, 1, 1, "c.c128_expfusion_p", "scale"], [73, 1, 1, "c.c128_expfusion_p", "src_data"]], "c128_expfusion_s": [[73, 1, 1, "c.c128_expfusion_s", "core_mask"], [73, 1, 1, "c.c128_expfusion_s", "dst_data"], [73, 1, 1, "c.c128_expfusion_s", "in_scale"], [73, 1, 1, "c.c128_expfusion_s", "length"], [73, 1, 1, "c.c128_expfusion_s", "out_scale"], [73, 1, 1, "c.c128_expfusion_s", "scale"], [73, 1, 1, "c.c128_expfusion_s", "src_data"]], "c128_extract_features_p": [[55, 1, 1, "c.c128_extract_features_p", "num_strings"], [55, 1, 1, "c.c128_extract_features_p", "output_labels"], [55, 1, 1, "c.c128_extract_features_p", "output_weights"], [55, 1, 1, "c.c128_extract_features_p", "string_lengths"], [55, 1, 1, "c.c128_extract_features_p", "string_pointers"]], "c128_extract_features_s": [[55, 1, 1, "c.c128_extract_features_s", "core_mask"], [55, 1, 1, "c.c128_extract_features_s", "num_strings"], [55, 1, 1, "c.c128_extract_features_s", "output_labels"], [55, 1, 1, "c.c128_extract_features_s", "output_weights"], [55, 1, 1, "c.c128_extract_features_s", "string_lengths"], [55, 1, 1, "c.c128_extract_features_s", "string_pointers"]], "c128_fill_p": [[78, 1, 1, "c.c128_fill_p", "output"], [78, 1, 1, "c.c128_fill_p", "param"], [78, 1, 1, "c.c128_fill_p", "value"]], "c128_fill_s": [[78, 1, 1, "c.c128_fill_s", "core_mask"], [78, 1, 1, "c.c128_fill_s", "output"], [78, 1, 1, "c.c128_fill_s", "param"], [78, 1, 1, "c.c128_fill_s", "value"]], "c128_formattranspose_p": [[85, 1, 1, "c.c128_formattranspose_p", "batch"], [85, 1, 1, "c.c128_formattranspose_p", "channel"], [85, 1, 1, "c.c128_formattranspose_p", "dst_data"], [85, 1, 1, "c.c128_formattranspose_p", "dst_format"], [85, 1, 1, "c.c128_formattranspose_p", "plane"], [85, 1, 1, "c.c128_formattranspose_p", "src_data"], [85, 1, 1, "c.c128_formattranspose_p", "src_format"]], "c128_formattranspose_s": [[85, 1, 1, "c.c128_formattranspose_s", "batch"], [85, 1, 1, "c.c128_formattranspose_s", "channel"], [85, 1, 1, "c.c128_formattranspose_s", "core_mask"], [85, 1, 1, "c.c128_formattranspose_s", "dst_data"], [85, 1, 1, "c.c128_formattranspose_s", "dst_format"], [85, 1, 1, "c.c128_formattranspose_s", "plane"], [85, 1, 1, "c.c128_formattranspose_s", "src_data"], [85, 1, 1, "c.c128_formattranspose_s", "src_format"]], "c128_gather_nd_p": [[89, 1, 1, "c.c128_gather_nd_p", "indices"], [89, 1, 1, "c.c128_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.c128_gather_nd_p", "indices_shape"], [89, 1, 1, "c.c128_gather_nd_p", "input"], [89, 1, 1, "c.c128_gather_nd_p", "input_ndim"], [89, 1, 1, "c.c128_gather_nd_p", "input_shape"], [89, 1, 1, "c.c128_gather_nd_p", "output"]], "c128_gather_nd_s": [[89, 1, 1, "c.c128_gather_nd_s", "core_mask"], [89, 1, 1, "c.c128_gather_nd_s", "indices"], [89, 1, 1, "c.c128_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.c128_gather_nd_s", "indices_shape"], [89, 1, 1, "c.c128_gather_nd_s", "input"], [89, 1, 1, "c.c128_gather_nd_s", "input_ndim"], [89, 1, 1, "c.c128_gather_nd_s", "input_shape"], [89, 1, 1, "c.c128_gather_nd_s", "output"]], "c128_gather_p": [[88, 1, 1, "c.c128_gather_p", "axis"], [88, 1, 1, "c.c128_gather_p", "batch_dims"], [88, 1, 1, "c.c128_gather_p", "indices"], [88, 1, 1, "c.c128_gather_p", "indices_ndim"], [88, 1, 1, "c.c128_gather_p", "indices_shape"], [88, 1, 1, "c.c128_gather_p", "input"], [88, 1, 1, "c.c128_gather_p", "input_ndim"], [88, 1, 1, "c.c128_gather_p", "input_shape"], [88, 1, 1, "c.c128_gather_p", "output"]], "c128_gather_s": [[88, 1, 1, "c.c128_gather_s", "axis"], [88, 1, 1, "c.c128_gather_s", "batch_dims"], [88, 1, 1, "c.c128_gather_s", "core_mask"], [88, 1, 1, "c.c128_gather_s", "indices"], [88, 1, 1, "c.c128_gather_s", "indices_ndim"], [88, 1, 1, "c.c128_gather_s", "indices_shape"], [88, 1, 1, "c.c128_gather_s", "input"], [88, 1, 1, "c.c128_gather_s", "input_ndim"], [88, 1, 1, "c.c128_gather_s", "input_shape"], [88, 1, 1, "c.c128_gather_s", "output"]], "c128_gatherd_p": [[90, 1, 1, "c.c128_gatherd_p", "dim"], [90, 1, 1, "c.c128_gatherd_p", "index"], [90, 1, 1, "c.c128_gatherd_p", "index_shape"], [90, 1, 1, "c.c128_gatherd_p", "input_shape"], [90, 1, 1, "c.c128_gatherd_p", "input_shape_size"], [90, 1, 1, "c.c128_gatherd_p", "input_x"], [90, 1, 1, "c.c128_gatherd_p", "output"]], "c128_gatherd_s": [[90, 1, 1, "c.c128_gatherd_s", "core_mask"], [90, 1, 1, "c.c128_gatherd_s", "dim"], [90, 1, 1, "c.c128_gatherd_s", "index"], [90, 1, 1, "c.c128_gatherd_s", "index_shape"], [90, 1, 1, "c.c128_gatherd_s", "input_shape"], [90, 1, 1, "c.c128_gatherd_s", "input_shape_size"], [90, 1, 1, "c.c128_gatherd_s", "input_x"], [90, 1, 1, "c.c128_gatherd_s", "output"]], "c128_isfinite_p": [[99, 1, 1, "c.c128_isfinite_p", "Input"], [99, 1, 1, "c.c128_isfinite_p", "length"], [99, 1, 1, "c.c128_isfinite_p", "output"]], "c128_isfinite_s": [[99, 1, 1, "c.c128_isfinite_s", "Input"], [99, 1, 1, "c.c128_isfinite_s", "core_mask"], [99, 1, 1, "c.c128_isfinite_s", "length"], [99, 1, 1, "c.c128_isfinite_s", "output"]], "c128_matmulfusion_p": [[121, 1, 1, "c.c128_matmulfusion_p", "A"], [121, 1, 1, "c.c128_matmulfusion_p", "B"], [121, 1, 1, "c.c128_matmulfusion_p", "C"], [121, 1, 1, "c.c128_matmulfusion_p", "K"], [121, 1, 1, "c.c128_matmulfusion_p", "M"], [121, 1, 1, "c.c128_matmulfusion_p", "N"], [121, 1, 1, "c.c128_matmulfusion_p", "activation_type"], [121, 1, 1, "c.c128_matmulfusion_p", "bias"]], "c128_matmulfusion_s": [[121, 1, 1, "c.c128_matmulfusion_s", "A"], [121, 1, 1, "c.c128_matmulfusion_s", "B"], [121, 1, 1, "c.c128_matmulfusion_s", "C"], [121, 1, 1, "c.c128_matmulfusion_s", "K"], [121, 1, 1, "c.c128_matmulfusion_s", "M"], [121, 1, 1, "c.c128_matmulfusion_s", "N"], [121, 1, 1, "c.c128_matmulfusion_s", "activation_type"], [121, 1, 1, "c.c128_matmulfusion_s", "bias"], [121, 1, 1, "c.c128_matmulfusion_s", "core_mask"]], "c128_mul_p": [[130, 1, 1, "c.c128_mul_p", "input0"], [130, 1, 1, "c.c128_mul_p", "input1"], [130, 1, 1, "c.c128_mul_p", "length"], [130, 1, 1, "c.c128_mul_p", "output"]], "c128_mul_s": [[130, 1, 1, "c.c128_mul_s", "core_mask"], [130, 1, 1, "c.c128_mul_s", "input0"], [130, 1, 1, "c.c128_mul_s", "input1"], [130, 1, 1, "c.c128_mul_s", "length"], [130, 1, 1, "c.c128_mul_s", "output"]], "c128_neg_grad_p": [[133, 1, 1, "c.c128_neg_grad_p", "Input"], [133, 1, 1, "c.c128_neg_grad_p", "length"], [133, 1, 1, "c.c128_neg_grad_p", "output"]], "c128_neg_grad_s": [[133, 1, 1, "c.c128_neg_grad_s", "Input"], [133, 1, 1, "c.c128_neg_grad_s", "core_mask"], [133, 1, 1, "c.c128_neg_grad_s", "length"], [133, 1, 1, "c.c128_neg_grad_s", "output"]], "c128_neg_p": [[132, 1, 1, "c.c128_neg_p", "Input"], [132, 1, 1, "c.c128_neg_p", "length"], [132, 1, 1, "c.c128_neg_p", "output"]], "c128_neg_s": [[132, 1, 1, "c.c128_neg_s", "Input"], [132, 1, 1, "c.c128_neg_s", "core_mask"], [132, 1, 1, "c.c128_neg_s", "length"], [132, 1, 1, "c.c128_neg_s", "output"]], "c128_nonzero_p": [[137, 1, 1, "c.c128_nonzero_p", "dim_strides"], [137, 1, 1, "c.c128_nonzero_p", "input"], [137, 1, 1, "c.c128_nonzero_p", "input_rank"], [137, 1, 1, "c.c128_nonzero_p", "length"], [137, 1, 1, "c.c128_nonzero_p", "non_zero_num"], [137, 1, 1, "c.c128_nonzero_p", "output"], [137, 1, 1, "c.c128_nonzero_p", "shape"]], "c128_nonzero_s": [[137, 1, 1, "c.c128_nonzero_s", "core_mask"], [137, 1, 1, "c.c128_nonzero_s", "dim_strides"], [137, 1, 1, "c.c128_nonzero_s", "input"], [137, 1, 1, "c.c128_nonzero_s", "input_rank"], [137, 1, 1, "c.c128_nonzero_s", "length"], [137, 1, 1, "c.c128_nonzero_s", "non_zero_num"], [137, 1, 1, "c.c128_nonzero_s", "output"], [137, 1, 1, "c.c128_nonzero_s", "shape"]], "c128_not_equal_p": [[138, 1, 1, "c.c128_not_equal_p", "Input0"], [138, 1, 1, "c.c128_not_equal_p", "Input1"], [138, 1, 1, "c.c128_not_equal_p", "length"], [138, 1, 1, "c.c128_not_equal_p", "output"]], "c128_not_equal_s": [[138, 1, 1, "c.c128_not_equal_s", "Input0"], [138, 1, 1, "c.c128_not_equal_s", "Input1"], [138, 1, 1, "c.c128_not_equal_s", "core_mask"], [138, 1, 1, "c.c128_not_equal_s", "length"], [138, 1, 1, "c.c128_not_equal_s", "output"]], "c128_onehot_p": [[139, 1, 1, "c.c128_onehot_p", "axis"], [139, 1, 1, "c.c128_onehot_p", "depth"], [139, 1, 1, "c.c128_onehot_p", "indices"], [139, 1, 1, "c.c128_onehot_p", "indices_shape"], [139, 1, 1, "c.c128_onehot_p", "indices_shape_size"], [139, 1, 1, "c.c128_onehot_p", "on_off"], [139, 1, 1, "c.c128_onehot_p", "output"], [139, 1, 1, "c.c128_onehot_p", "support_neg_index"]], "c128_onehot_s": [[139, 1, 1, "c.c128_onehot_s", "axis"], [139, 1, 1, "c.c128_onehot_s", "core_mask"], [139, 1, 1, "c.c128_onehot_s", "depth"], [139, 1, 1, "c.c128_onehot_s", "indices"], [139, 1, 1, "c.c128_onehot_s", "indices_shape"], [139, 1, 1, "c.c128_onehot_s", "indices_shape_size"], [139, 1, 1, "c.c128_onehot_s", "on_off"], [139, 1, 1, "c.c128_onehot_s", "output"], [139, 1, 1, "c.c128_onehot_s", "support_neg_index"]], "c128_ones_like_p": [[140, 1, 1, "c.c128_ones_like_p", "length"], [140, 1, 1, "c.c128_ones_like_p", "output"]], "c128_ones_like_s": [[140, 1, 1, "c.c128_ones_like_s", "core_mask"], [140, 1, 1, "c.c128_ones_like_s", "length"], [140, 1, 1, "c.c128_ones_like_s", "output"]], "c128_padfusion_p": [[141, 1, 1, "c.c128_padfusion_p", "params"]], "c128_padfusion_s": [[141, 1, 1, "c.c128_padfusion_s", "core_mask"], [141, 1, 1, "c.c128_padfusion_s", "params"]], "c128_real_div_p": [[152, 1, 1, "c.c128_real_div_p", "input0"], [152, 1, 1, "c.c128_real_div_p", "input1"], [152, 1, 1, "c.c128_real_div_p", "length"], [152, 1, 1, "c.c128_real_div_p", "output"]], "c128_real_div_s": [[152, 1, 1, "c.c128_real_div_s", "core_mask"], [152, 1, 1, "c.c128_real_div_s", "input0"], [152, 1, 1, "c.c128_real_div_s", "input1"], [152, 1, 1, "c.c128_real_div_s", "length"], [152, 1, 1, "c.c128_real_div_s", "output"]], "c128_reciprocal_p": [[153, 1, 1, "c.c128_reciprocal_p", "Input"], [153, 1, 1, "c.c128_reciprocal_p", "length"], [153, 1, 1, "c.c128_reciprocal_p", "output"]], "c128_reciprocal_s": [[153, 1, 1, "c.c128_reciprocal_s", "Input"], [153, 1, 1, "c.c128_reciprocal_s", "core_mask"], [153, 1, 1, "c.c128_reciprocal_s", "length"], [153, 1, 1, "c.c128_reciprocal_s", "output"]], "c128_reduceall_p": [[21, 1, 1, "c.c128_reduceall_p", "axis_size"], [21, 1, 1, "c.c128_reduceall_p", "dst_data"], [21, 1, 1, "c.c128_reduceall_p", "inner_size"], [21, 1, 1, "c.c128_reduceall_p", "outer_size"], [21, 1, 1, "c.c128_reduceall_p", "src_data"]], "c128_reduceall_s": [[21, 1, 1, "c.c128_reduceall_s", "axis_size"], [21, 1, 1, "c.c128_reduceall_s", "core_mask"], [21, 1, 1, "c.c128_reduceall_s", "dst_data"], [21, 1, 1, "c.c128_reduceall_s", "inner_size"], [21, 1, 1, "c.c128_reduceall_s", "outer_size"], [21, 1, 1, "c.c128_reduceall_s", "src_data"]], "c128_reshape_p": [[156, 1, 1, "c.c128_reshape_p", "input"], [156, 1, 1, "c.c128_reshape_p", "length"], [156, 1, 1, "c.c128_reshape_p", "output"]], "c128_reshape_s": [[156, 1, 1, "c.c128_reshape_s", "core_mask"], [156, 1, 1, "c.c128_reshape_s", "input"], [156, 1, 1, "c.c128_reshape_s", "length"], [156, 1, 1, "c.c128_reshape_s", "output"]], "c128_rfft_p": [[161, 1, 1, "c.c128_rfft_p", "dir"], [161, 1, 1, "c.c128_rfft_p", "fft_size"], [161, 1, 1, "c.c128_rfft_p", "input"], [161, 1, 1, "c.c128_rfft_p", "output"], [161, 1, 1, "c.c128_rfft_p", "scratch_ptr"]], "c128_rfft_s": [[161, 1, 1, "c.c128_rfft_s", "core_mask"], [161, 1, 1, "c.c128_rfft_s", "dir"], [161, 1, 1, "c.c128_rfft_s", "fft_size1"], [161, 1, 1, "c.c128_rfft_s", "fft_size2"], [161, 1, 1, "c.c128_rfft_s", "input"], [161, 1, 1, "c.c128_rfft_s", "output"], [161, 1, 1, "c.c128_rfft_s", "scratch_ptr"], [161, 1, 1, "c.c128_rfft_s", "twiddle"]], "c128_rsqrt_p": [[164, 1, 1, "c.c128_rsqrt_p", "dst"], [164, 1, 1, "c.c128_rsqrt_p", "length"], [164, 1, 1, "c.c128_rsqrt_p", "src"]], "c128_rsqrt_s": [[164, 1, 1, "c.c128_rsqrt_s", "core_mask"], [164, 1, 1, "c.c128_rsqrt_s", "dst"], [164, 1, 1, "c.c128_rsqrt_s", "length"], [164, 1, 1, "c.c128_rsqrt_s", "src"]], "c128_scatter_elements_p": [[167, 1, 1, "c.c128_scatter_elements_p", "core_mask"], [167, 1, 1, "c.c128_scatter_elements_p", "indices"], [167, 1, 1, "c.c128_scatter_elements_p", "input"], [167, 1, 1, "c.c128_scatter_elements_p", "output"], [167, 1, 1, "c.c128_scatter_elements_p", "param"], [167, 1, 1, "c.c128_scatter_elements_p", "updates"]], "c128_scatter_elements_s": [[167, 1, 1, "c.c128_scatter_elements_s", "core_mask"], [167, 1, 1, "c.c128_scatter_elements_s", "indices"], [167, 1, 1, "c.c128_scatter_elements_s", "input"], [167, 1, 1, "c.c128_scatter_elements_s", "output"], [167, 1, 1, "c.c128_scatter_elements_s", "param"], [167, 1, 1, "c.c128_scatter_elements_s", "updates"]], "c128_scatter_nd_p": [[168, 1, 1, "c.c128_scatter_nd_p", "indices"], [168, 1, 1, "c.c128_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.c128_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.c128_scatter_nd_p", "output"], [168, 1, 1, "c.c128_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.c128_scatter_nd_p", "output_shape"], [168, 1, 1, "c.c128_scatter_nd_p", "updates"]], "c128_scatter_nd_s": [[168, 1, 1, "c.c128_scatter_nd_s", "core_mask"], [168, 1, 1, "c.c128_scatter_nd_s", "indices"], [168, 1, 1, "c.c128_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.c128_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.c128_scatter_nd_s", "output"], [168, 1, 1, "c.c128_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.c128_scatter_nd_s", "output_shape"], [168, 1, 1, "c.c128_scatter_nd_s", "updates"]], "c128_scatter_nd_update_p": [[169, 1, 1, "c.c128_scatter_nd_update_p", "indices"], [169, 1, 1, "c.c128_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.c128_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.c128_scatter_nd_update_p", "output"], [169, 1, 1, "c.c128_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.c128_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.c128_scatter_nd_update_p", "updates"]], "c128_scatter_nd_update_s": [[169, 1, 1, "c.c128_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.c128_scatter_nd_update_s", "indices"], [169, 1, 1, "c.c128_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.c128_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.c128_scatter_nd_update_s", "output"], [169, 1, 1, "c.c128_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.c128_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.c128_scatter_nd_update_s", "updates"]], "c128_select_p": [[170, 1, 1, "c.c128_select_p", "condition"], [170, 1, 1, "c.c128_select_p", "index_list1"], [170, 1, 1, "c.c128_select_p", "index_list2"], [170, 1, 1, "c.c128_select_p", "index_list3"], [170, 1, 1, "c.c128_select_p", "input0"], [170, 1, 1, "c.c128_select_p", "input1"], [170, 1, 1, "c.c128_select_p", "is_broadcast"], [170, 1, 1, "c.c128_select_p", "output"], [170, 1, 1, "c.c128_select_p", "output_dims"], [170, 1, 1, "c.c128_select_p", "output_dims_num"]], "c128_select_s": [[170, 1, 1, "c.c128_select_s", "condition"], [170, 1, 1, "c.c128_select_s", "core_mask"], [170, 1, 1, "c.c128_select_s", "index_list1"], [170, 1, 1, "c.c128_select_s", "index_list2"], [170, 1, 1, "c.c128_select_s", "index_list3"], [170, 1, 1, "c.c128_select_s", "input0"], [170, 1, 1, "c.c128_select_s", "input1"], [170, 1, 1, "c.c128_select_s", "is_broadcast"], [170, 1, 1, "c.c128_select_s", "output"], [170, 1, 1, "c.c128_select_s", "output_dims"], [170, 1, 1, "c.c128_select_s", "output_dims_num"]], "c128_slice_p": [[178, 1, 1, "c.c128_slice_p", "begin"], [178, 1, 1, "c.c128_slice_p", "input"], [178, 1, 1, "c.c128_slice_p", "input_shape"], [178, 1, 1, "c.c128_slice_p", "ndim"], [178, 1, 1, "c.c128_slice_p", "output"], [178, 1, 1, "c.c128_slice_p", "size"]], "c128_slice_s": [[178, 1, 1, "c.c128_slice_s", "begin"], [178, 1, 1, "c.c128_slice_s", "core_mask"], [178, 1, 1, "c.c128_slice_s", "input"], [178, 1, 1, "c.c128_slice_s", "input_shape"], [178, 1, 1, "c.c128_slice_s", "ndim"], [178, 1, 1, "c.c128_slice_s", "output"], [178, 1, 1, "c.c128_slice_s", "size"]], "c128_spacetobatch_p": [[183, 1, 1, "c.c128_spacetobatch_p", "block_size"], [183, 1, 1, "c.c128_spacetobatch_p", "data_size"], [183, 1, 1, "c.c128_spacetobatch_p", "input"], [183, 1, 1, "c.c128_spacetobatch_p", "input_shape"], [183, 1, 1, "c.c128_spacetobatch_p", "output"], [183, 1, 1, "c.c128_spacetobatch_p", "paddings"]], "c128_spacetobatch_s": [[183, 1, 1, "c.c128_spacetobatch_s", "block_size"], [183, 1, 1, "c.c128_spacetobatch_s", "core_mask"], [183, 1, 1, "c.c128_spacetobatch_s", "data_size"], [183, 1, 1, "c.c128_spacetobatch_s", "input"], [183, 1, 1, "c.c128_spacetobatch_s", "input_shape"], [183, 1, 1, "c.c128_spacetobatch_s", "output"], [183, 1, 1, "c.c128_spacetobatch_s", "paddings"]], "c128_spacetobatchnd_p": [[184, 1, 1, "c.c128_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.c128_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.c128_spacetobatchnd_p", "input"], [184, 1, 1, "c.c128_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.c128_spacetobatchnd_p", "output"], [184, 1, 1, "c.c128_spacetobatchnd_p", "paddings"]], "c128_spacetobatchnd_s": [[184, 1, 1, "c.c128_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.c128_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.c128_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.c128_spacetobatchnd_s", "input"], [184, 1, 1, "c.c128_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.c128_spacetobatchnd_s", "output"], [184, 1, 1, "c.c128_spacetobatchnd_s", "paddings"]], "c128_spacetodepth_p": [[185, 1, 1, "c.c128_spacetodepth_p", "block"], [185, 1, 1, "c.c128_spacetodepth_p", "data_size"], [185, 1, 1, "c.c128_spacetodepth_p", "in_shape"], [185, 1, 1, "c.c128_spacetodepth_p", "input"], [185, 1, 1, "c.c128_spacetodepth_p", "output"]], "c128_spacetodepth_s": [[185, 1, 1, "c.c128_spacetodepth_s", "block"], [185, 1, 1, "c.c128_spacetodepth_s", "core_mask"], [185, 1, 1, "c.c128_spacetodepth_s", "data_size"], [185, 1, 1, "c.c128_spacetodepth_s", "in_shape"], [185, 1, 1, "c.c128_spacetodepth_s", "input"], [185, 1, 1, "c.c128_spacetodepth_s", "output"]], "c128_sparsefillemptyrows_p": [[187, 1, 1, "c.c128_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_p", "values_ptr"]], "c128_sparsefillemptyrows_s": [[187, 1, 1, "c.c128_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.c128_sparsefillemptyrows_s", "values_ptr"]], "c128_sparsesegmentsum_p": [[189, 1, 1, "c.c128_sparsesegmentsum_p", "in_data"], [189, 1, 1, "c.c128_sparsesegmentsum_p", "in_data_shape"], [189, 1, 1, "c.c128_sparsesegmentsum_p", "in_data_shape_size"], [189, 1, 1, "c.c128_sparsesegmentsum_p", "in_indices"], [189, 1, 1, "c.c128_sparsesegmentsum_p", "in_indices_size"], [189, 1, 1, "c.c128_sparsesegmentsum_p", "in_segment_ids"], [189, 1, 1, "c.c128_sparsesegmentsum_p", "out_data"], [189, 1, 1, "c.c128_sparsesegmentsum_p", "out_data_shape"]], "c128_sparsesegmentsum_s": [[189, 1, 1, "c.c128_sparsesegmentsum_s", "core_mask"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "in_data"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "in_data_shape"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "in_data_shape_size"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "in_indices"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "in_indices_size"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "in_segment_ids"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "out_data"], [189, 1, 1, "c.c128_sparsesegmentsum_s", "out_data_shape"]], "c128_sparsetodense_p": [[190, 1, 1, "c.c128_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.c128_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.c128_sparsetodense_p", "output"], [190, 1, 1, "c.c128_sparsetodense_p", "output_strides"], [190, 1, 1, "c.c128_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.c128_sparsetodense_p", "sparse_values"]], "c128_sparsetodense_s": [[190, 1, 1, "c.c128_sparsetodense_s", "core_mask"], [190, 1, 1, "c.c128_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.c128_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.c128_sparsetodense_s", "output"], [190, 1, 1, "c.c128_sparsetodense_s", "output_strides"], [190, 1, 1, "c.c128_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.c128_sparsetodense_s", "sparse_values"]], "c128_splice_p": [[191, 1, 1, "c.c128_splice_p", "context_dim"], [191, 1, 1, "c.c128_splice_p", "dst_col"], [191, 1, 1, "c.c128_splice_p", "dst_data"], [191, 1, 1, "c.c128_splice_p", "dst_row"], [191, 1, 1, "c.c128_splice_p", "forward_indexes"], [191, 1, 1, "c.c128_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.c128_splice_p", "src_col"], [191, 1, 1, "c.c128_splice_p", "src_data"], [191, 1, 1, "c.c128_splice_p", "src_row"]], "c128_splice_s": [[191, 1, 1, "c.c128_splice_s", "context_dim"], [191, 1, 1, "c.c128_splice_s", "core_mask"], [191, 1, 1, "c.c128_splice_s", "dst_col"], [191, 1, 1, "c.c128_splice_s", "dst_data"], [191, 1, 1, "c.c128_splice_s", "dst_row"], [191, 1, 1, "c.c128_splice_s", "forward_indexes"], [191, 1, 1, "c.c128_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.c128_splice_s", "src_col"], [191, 1, 1, "c.c128_splice_s", "src_data"], [191, 1, 1, "c.c128_splice_s", "src_row"]], "c128_split_p": [[192, 1, 1, "c.c128_split_p", "axis"], [192, 1, 1, "c.c128_split_p", "input"], [192, 1, 1, "c.c128_split_p", "input_ndim"], [192, 1, 1, "c.c128_split_p", "input_shape"], [192, 1, 1, "c.c128_split_p", "num_split"], [192, 1, 1, "c.c128_split_p", "outputs"], [192, 1, 1, "c.c128_split_p", "split_sizes"]], "c128_split_s": [[192, 1, 1, "c.c128_split_s", "axis"], [192, 1, 1, "c.c128_split_s", "core_mask"], [192, 1, 1, "c.c128_split_s", "input"], [192, 1, 1, "c.c128_split_s", "input_ndim"], [192, 1, 1, "c.c128_split_s", "input_shape"], [192, 1, 1, "c.c128_split_s", "num_split"], [192, 1, 1, "c.c128_split_s", "outputs"], [192, 1, 1, "c.c128_split_s", "split_sizes"]], "c128_split_with_overlap_p": [[193, 1, 1, "c.c128_split_with_overlap_p", "axis"], [193, 1, 1, "c.c128_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.c128_split_with_overlap_p", "input"], [193, 1, 1, "c.c128_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.c128_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.c128_split_with_overlap_p", "num_split"], [193, 1, 1, "c.c128_split_with_overlap_p", "outputs"], [193, 1, 1, "c.c128_split_with_overlap_p", "start_indices"]], "c128_split_with_overlap_s": [[193, 1, 1, "c.c128_split_with_overlap_s", "axis"], [193, 1, 1, "c.c128_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.c128_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.c128_split_with_overlap_s", "input"], [193, 1, 1, "c.c128_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.c128_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.c128_split_with_overlap_s", "num_split"], [193, 1, 1, "c.c128_split_with_overlap_s", "outputs"], [193, 1, 1, "c.c128_split_with_overlap_s", "start_indices"]], "c128_sqrt_p": [[194, 1, 1, "c.c128_sqrt_p", "dst_data"], [194, 1, 1, "c.c128_sqrt_p", "length"], [194, 1, 1, "c.c128_sqrt_p", "src_data"]], "c128_sqrt_s": [[194, 1, 1, "c.c128_sqrt_s", "core_mask"], [194, 1, 1, "c.c128_sqrt_s", "dst_data"], [194, 1, 1, "c.c128_sqrt_s", "length"], [194, 1, 1, "c.c128_sqrt_s", "src_data"]], "c128_sqrtgrad_p": [[195, 1, 1, "c.c128_sqrtgrad_p", "input1"], [195, 1, 1, "c.c128_sqrtgrad_p", "input2"], [195, 1, 1, "c.c128_sqrtgrad_p", "output"], [195, 1, 1, "c.c128_sqrtgrad_p", "size"]], "c128_sqrtgrad_s": [[195, 1, 1, "c.c128_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.c128_sqrtgrad_s", "input1"], [195, 1, 1, "c.c128_sqrtgrad_s", "input2"], [195, 1, 1, "c.c128_sqrtgrad_s", "output"], [195, 1, 1, "c.c128_sqrtgrad_s", "size"]], "c128_square_p": [[196, 1, 1, "c.c128_square_p", "dst"], [196, 1, 1, "c.c128_square_p", "length"], [196, 1, 1, "c.c128_square_p", "src"]], "c128_square_s": [[196, 1, 1, "c.c128_square_s", "core_mask"], [196, 1, 1, "c.c128_square_s", "dst"], [196, 1, 1, "c.c128_square_s", "length"], [196, 1, 1, "c.c128_square_s", "src"]], "c128_squaredifference_p": [[197, 1, 1, "c.c128_squaredifference_p", "input0"], [197, 1, 1, "c.c128_squaredifference_p", "input1"], [197, 1, 1, "c.c128_squaredifference_p", "length"], [197, 1, 1, "c.c128_squaredifference_p", "output"]], "c128_squaredifference_s": [[197, 1, 1, "c.c128_squaredifference_s", "core_mask"], [197, 1, 1, "c.c128_squaredifference_s", "input0"], [197, 1, 1, "c.c128_squaredifference_s", "input1"], [197, 1, 1, "c.c128_squaredifference_s", "length"], [197, 1, 1, "c.c128_squaredifference_s", "output"]], "c128_stack_p": [[199, 1, 1, "c.c128_stack_p", "axis"], [199, 1, 1, "c.c128_stack_p", "input_ndim"], [199, 1, 1, "c.c128_stack_p", "input_shape"], [199, 1, 1, "c.c128_stack_p", "inputs"], [199, 1, 1, "c.c128_stack_p", "num_inputs"], [199, 1, 1, "c.c128_stack_p", "output"]], "c128_stack_s": [[199, 1, 1, "c.c128_stack_s", "axis"], [199, 1, 1, "c.c128_stack_s", "core_mask"], [199, 1, 1, "c.c128_stack_s", "input_ndim"], [199, 1, 1, "c.c128_stack_s", "input_shape"], [199, 1, 1, "c.c128_stack_s", "inputs"], [199, 1, 1, "c.c128_stack_s", "num_inputs"], [199, 1, 1, "c.c128_stack_s", "output"]], "c128_subext_p": [[202, 1, 1, "c.c128_subext_p", "alpha"], [202, 1, 1, "c.c128_subext_p", "input0"], [202, 1, 1, "c.c128_subext_p", "input1"], [202, 1, 1, "c.c128_subext_p", "output"], [202, 1, 1, "c.c128_subext_p", "size"]], "c128_subext_s": [[202, 1, 1, "c.c128_subext_s", "alpha"], [202, 1, 1, "c.c128_subext_s", "core_mask"], [202, 1, 1, "c.c128_subext_s", "input0"], [202, 1, 1, "c.c128_subext_s", "input1"], [202, 1, 1, "c.c128_subext_s", "output"], [202, 1, 1, "c.c128_subext_s", "size"]], "c128_subrelu6_p": [[202, 1, 1, "c.c128_subrelu6_p", "input0"], [202, 1, 1, "c.c128_subrelu6_p", "input1"], [202, 1, 1, "c.c128_subrelu6_p", "output"], [202, 1, 1, "c.c128_subrelu6_p", "size"]], "c128_subrelu6_s": [[202, 1, 1, "c.c128_subrelu6_s", "core_mask"], [202, 1, 1, "c.c128_subrelu6_s", "input0"], [202, 1, 1, "c.c128_subrelu6_s", "input1"], [202, 1, 1, "c.c128_subrelu6_s", "output"], [202, 1, 1, "c.c128_subrelu6_s", "size"]], "c128_subrelu_p": [[202, 1, 1, "c.c128_subrelu_p", "input0"], [202, 1, 1, "c.c128_subrelu_p", "input1"], [202, 1, 1, "c.c128_subrelu_p", "output"], [202, 1, 1, "c.c128_subrelu_p", "size"]], "c128_subrelu_s": [[202, 1, 1, "c.c128_subrelu_s", "core_mask"], [202, 1, 1, "c.c128_subrelu_s", "input0"], [202, 1, 1, "c.c128_subrelu_s", "input1"], [202, 1, 1, "c.c128_subrelu_s", "output"], [202, 1, 1, "c.c128_subrelu_s", "size"]], "c128_tensor_scatter_add_p": [[206, 1, 1, "c.c128_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "input"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "output"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.c128_tensor_scatter_add_p", "updates"]], "c128_tensor_scatter_add_s": [[206, 1, 1, "c.c128_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "input"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "output"], [206, 1, 1, "c.c128_tensor_scatter_add_s", "updates"]], "c128_tensorarrayread_p": [[208, 1, 1, "c.c128_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.c128_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.c128_tensorarrayread_p", "index"], [208, 1, 1, "c.c128_tensorarrayread_p", "output_data"], [208, 1, 1, "c.c128_tensorarrayread_p", "output_size"]], "c128_tensorarrayread_s": [[208, 1, 1, "c.c128_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.c128_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.c128_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.c128_tensorarrayread_s", "index"], [208, 1, 1, "c.c128_tensorarrayread_s", "output_data"], [208, 1, 1, "c.c128_tensorarrayread_s", "output_size"]], "c128_tensorlistfromtensor_p": [[210, 1, 1, "c.c128_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.c128_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.c128_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.c128_tensorlistfromtensor_p", "output_tensors"]], "c128_tensorlistfromtensor_s": [[210, 1, 1, "c.c128_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.c128_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.c128_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.c128_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.c128_tensorlistfromtensor_s", "output_tensors"]], "c128_tile_p": [[215, 1, 1, "c.c128_tile_p", "input"], [215, 1, 1, "c.c128_tile_p", "input_shape"], [215, 1, 1, "c.c128_tile_p", "output"], [215, 1, 1, "c.c128_tile_p", "stride"], [215, 1, 1, "c.c128_tile_p", "tile_dim"], [215, 1, 1, "c.c128_tile_p", "tile_num"]], "c128_tile_s": [[215, 1, 1, "c.c128_tile_s", "core_mask"], [215, 1, 1, "c.c128_tile_s", "input"], [215, 1, 1, "c.c128_tile_s", "input_shape"], [215, 1, 1, "c.c128_tile_s", "output"], [215, 1, 1, "c.c128_tile_s", "stride"], [215, 1, 1, "c.c128_tile_s", "tile_dim"], [215, 1, 1, "c.c128_tile_s", "tile_num"]], "c128_transpose_p": [[217, 1, 1, "c.c128_transpose_p", "in_data"], [217, 1, 1, "c.c128_transpose_p", "num_axes"], [217, 1, 1, "c.c128_transpose_p", "out_data"], [217, 1, 1, "c.c128_transpose_p", "out_strides"], [217, 1, 1, "c.c128_transpose_p", "output_shape"], [217, 1, 1, "c.c128_transpose_p", "perm"], [217, 1, 1, "c.c128_transpose_p", "strides"]], "c128_transpose_s": [[217, 1, 1, "c.c128_transpose_s", "core_mask"], [217, 1, 1, "c.c128_transpose_s", "in_data"], [217, 1, 1, "c.c128_transpose_s", "num_axes"], [217, 1, 1, "c.c128_transpose_s", "out_data"], [217, 1, 1, "c.c128_transpose_s", "out_strides"], [217, 1, 1, "c.c128_transpose_s", "output_shape"], [217, 1, 1, "c.c128_transpose_s", "perm"], [217, 1, 1, "c.c128_transpose_s", "strides"]], "c128_tril_p": [[218, 1, 1, "c.c128_tril_p", "dst"], [218, 1, 1, "c.c128_tril_p", "height"], [218, 1, 1, "c.c128_tril_p", "k"], [218, 1, 1, "c.c128_tril_p", "out_elems"], [218, 1, 1, "c.c128_tril_p", "src"], [218, 1, 1, "c.c128_tril_p", "width"]], "c128_tril_s": [[218, 1, 1, "c.c128_tril_s", "core_mask"], [218, 1, 1, "c.c128_tril_s", "dst"], [218, 1, 1, "c.c128_tril_s", "height"], [218, 1, 1, "c.c128_tril_s", "k"], [218, 1, 1, "c.c128_tril_s", "out_elems"], [218, 1, 1, "c.c128_tril_s", "src"], [218, 1, 1, "c.c128_tril_s", "width"]], "c128_triu_p": [[219, 1, 1, "c.c128_triu_p", "dst"], [219, 1, 1, "c.c128_triu_p", "height"], [219, 1, 1, "c.c128_triu_p", "k"], [219, 1, 1, "c.c128_triu_p", "out_elems"], [219, 1, 1, "c.c128_triu_p", "src"], [219, 1, 1, "c.c128_triu_p", "width"]], "c128_triu_s": [[219, 1, 1, "c.c128_triu_s", "core_mask"], [219, 1, 1, "c.c128_triu_s", "dst"], [219, 1, 1, "c.c128_triu_s", "height"], [219, 1, 1, "c.c128_triu_s", "k"], [219, 1, 1, "c.c128_triu_s", "out_elems"], [219, 1, 1, "c.c128_triu_s", "src"], [219, 1, 1, "c.c128_triu_s", "width"]], "c128_unsorted_segment_sum_p": [[222, 1, 1, "c.c128_unsorted_segment_sum_p", "dim0"], [222, 1, 1, "c.c128_unsorted_segment_sum_p", "dim1"], [222, 1, 1, "c.c128_unsorted_segment_sum_p", "id_max"], [222, 1, 1, "c.c128_unsorted_segment_sum_p", "index"], [222, 1, 1, "c.c128_unsorted_segment_sum_p", "input"], [222, 1, 1, "c.c128_unsorted_segment_sum_p", "output"]], "c128_unsorted_segment_sum_s": [[222, 1, 1, "c.c128_unsorted_segment_sum_s", "core_mask"], [222, 1, 1, "c.c128_unsorted_segment_sum_s", "dim0"], [222, 1, 1, "c.c128_unsorted_segment_sum_s", "dim1"], [222, 1, 1, "c.c128_unsorted_segment_sum_s", "id_max"], [222, 1, 1, "c.c128_unsorted_segment_sum_s", "index"], [222, 1, 1, "c.c128_unsorted_segment_sum_s", "input"], [222, 1, 1, "c.c128_unsorted_segment_sum_s", "output"]], "c128_where_p": [[225, 1, 1, "c.c128_where_p", "condition"], [225, 1, 1, "c.c128_where_p", "input0"], [225, 1, 1, "c.c128_where_p", "input1"], [225, 1, 1, "c.c128_where_p", "length"], [225, 1, 1, "c.c128_where_p", "output"]], "c128_where_s": [[225, 1, 1, "c.c128_where_s", "condition"], [225, 1, 1, "c.c128_where_s", "core_mask"], [225, 1, 1, "c.c128_where_s", "input0"], [225, 1, 1, "c.c128_where_s", "input1"], [225, 1, 1, "c.c128_where_s", "length"], [225, 1, 1, "c.c128_where_s", "output"]], "c128_zerolike_p": [[226, 1, 1, "c.c128_zerolike_p", "length"], [226, 1, 1, "c.c128_zerolike_p", "output"]], "c128_zerolike_s": [[226, 1, 1, "c.c128_zerolike_s", "core_mask"], [226, 1, 1, "c.c128_zerolike_s", "length"], [226, 1, 1, "c.c128_zerolike_s", "output"]], "c64_Unique_p": [[221, 1, 1, "c.c64_Unique_p", "input"], [221, 1, 1, "c.c64_Unique_p", "input_len"], [221, 1, 1, "c.c64_Unique_p", "output0"], [221, 1, 1, "c.c64_Unique_p", "output0_len"]], "c64_Unique_s": [[221, 1, 1, "c.c64_Unique_s", "core_mask"], [221, 1, 1, "c.c64_Unique_s", "input"], [221, 1, 1, "c.c64_Unique_s", "input_len"], [221, 1, 1, "c.c64_Unique_s", "output0"], [221, 1, 1, "c.c64_Unique_s", "output0_len"]], "c64_abs_p": [[10, 1, 1, "c.c64_abs_p", "dst_data"], [10, 1, 1, "c.c64_abs_p", "length"], [10, 1, 1, "c.c64_abs_p", "src_data"]], "c64_abs_s": [[10, 1, 1, "c.c64_abs_s", "core_mask"], [10, 1, 1, "c.c64_abs_s", "dst_data"], [10, 1, 1, "c.c64_abs_s", "length"], [10, 1, 1, "c.c64_abs_s", "src_data"]], "c64_addext_p": [[17, 1, 1, "c.c64_addext_p", "alpha"], [17, 1, 1, "c.c64_addext_p", "in0"], [17, 1, 1, "c.c64_addext_p", "in1"], [17, 1, 1, "c.c64_addext_p", "out"], [17, 1, 1, "c.c64_addext_p", "size"]], "c64_addext_s": [[17, 1, 1, "c.c64_addext_s", "alpha"], [17, 1, 1, "c.c64_addext_s", "core_mask"], [17, 1, 1, "c.c64_addext_s", "in0"], [17, 1, 1, "c.c64_addext_s", "in1"], [17, 1, 1, "c.c64_addext_s", "out"], [17, 1, 1, "c.c64_addext_s", "size"]], "c64_addn_p": [[19, 1, 1, "c.c64_addn_p", "input0"], [19, 1, 1, "c.c64_addn_p", "input1"], [19, 1, 1, "c.c64_addn_p", "length"], [19, 1, 1, "c.c64_addn_p", "output"]], "c64_addn_s": [[19, 1, 1, "c.c64_addn_s", "core_mask"], [19, 1, 1, "c.c64_addn_s", "input0"], [19, 1, 1, "c.c64_addn_s", "input1"], [19, 1, 1, "c.c64_addn_s", "length"], [19, 1, 1, "c.c64_addn_s", "output"]], "c64_addrelu6_p": [[17, 1, 1, "c.c64_addrelu6_p", "in0"], [17, 1, 1, "c.c64_addrelu6_p", "in1"], [17, 1, 1, "c.c64_addrelu6_p", "out"], [17, 1, 1, "c.c64_addrelu6_p", "size"]], "c64_addrelu6_s": [[17, 1, 1, "c.c64_addrelu6_s", "core_mask"], [17, 1, 1, "c.c64_addrelu6_s", "in0"], [17, 1, 1, "c.c64_addrelu6_s", "in1"], [17, 1, 1, "c.c64_addrelu6_s", "out"], [17, 1, 1, "c.c64_addrelu6_s", "size"]], "c64_addrelu_p": [[17, 1, 1, "c.c64_addrelu_p", "in0"], [17, 1, 1, "c.c64_addrelu_p", "in1"], [17, 1, 1, "c.c64_addrelu_p", "out"], [17, 1, 1, "c.c64_addrelu_p", "size"]], "c64_addrelu_s": [[17, 1, 1, "c.c64_addrelu_s", "core_mask"], [17, 1, 1, "c.c64_addrelu_s", "in0"], [17, 1, 1, "c.c64_addrelu_s", "in1"], [17, 1, 1, "c.c64_addrelu_s", "out"], [17, 1, 1, "c.c64_addrelu_s", "size"]], "c64_allgather_p": [[22, 1, 1, "c.c64_allgather_p", "data_size"], [22, 1, 1, "c.c64_allgather_p", "input"], [22, 1, 1, "c.c64_allgather_p", "input_rank"], [22, 1, 1, "c.c64_allgather_p", "output"], [22, 1, 1, "c.c64_allgather_p", "output_rank"]], "c64_allgather_s": [[22, 1, 1, "c.c64_allgather_s", "core_mask"], [22, 1, 1, "c.c64_allgather_s", "data_size"], [22, 1, 1, "c.c64_allgather_s", "input"], [22, 1, 1, "c.c64_allgather_s", "input_rank"], [22, 1, 1, "c.c64_allgather_s", "output"], [22, 1, 1, "c.c64_allgather_s", "output_rank"]], "c64_assign_p": [[27, 1, 1, "c.c64_assign_p", "dst"], [27, 1, 1, "c.c64_assign_p", "length"], [27, 1, 1, "c.c64_assign_p", "src"]], "c64_assign_s": [[27, 1, 1, "c.c64_assign_s", "core_mask"], [27, 1, 1, "c.c64_assign_s", "dst"], [27, 1, 1, "c.c64_assign_s", "length"], [27, 1, 1, "c.c64_assign_s", "src"]], "c64_assignadd_p": [[28, 1, 1, "c.c64_assignadd_p", "input"], [28, 1, 1, "c.c64_assignadd_p", "length"], [28, 1, 1, "c.c64_assignadd_p", "output"]], "c64_assignadd_s": [[28, 1, 1, "c.c64_assignadd_s", "core_mask"], [28, 1, 1, "c.c64_assignadd_s", "input"], [28, 1, 1, "c.c64_assignadd_s", "length"], [28, 1, 1, "c.c64_assignadd_s", "output"]], "c64_batchtospace_p": [[35, 1, 1, "c.c64_batchtospace_p", "block_size"], [35, 1, 1, "c.c64_batchtospace_p", "crops"], [35, 1, 1, "c.c64_batchtospace_p", "data_size"], [35, 1, 1, "c.c64_batchtospace_p", "input"], [35, 1, 1, "c.c64_batchtospace_p", "input_shape"], [35, 1, 1, "c.c64_batchtospace_p", "output"]], "c64_batchtospace_s": [[35, 1, 1, "c.c64_batchtospace_s", "block_size"], [35, 1, 1, "c.c64_batchtospace_s", "core_mask"], [35, 1, 1, "c.c64_batchtospace_s", "crops"], [35, 1, 1, "c.c64_batchtospace_s", "data_size"], [35, 1, 1, "c.c64_batchtospace_s", "input"], [35, 1, 1, "c.c64_batchtospace_s", "input_shape"], [35, 1, 1, "c.c64_batchtospace_s", "output"]], "c64_batchtospacend_p": [[36, 1, 1, "c.c64_batchtospacend_p", "block_size"], [36, 1, 1, "c.c64_batchtospacend_p", "crops"], [36, 1, 1, "c.c64_batchtospacend_p", "data_size"], [36, 1, 1, "c.c64_batchtospacend_p", "input"], [36, 1, 1, "c.c64_batchtospacend_p", "input_shape"], [36, 1, 1, "c.c64_batchtospacend_p", "output"]], "c64_batchtospacend_s": [[36, 1, 1, "c.c64_batchtospacend_s", "block_size"], [36, 1, 1, "c.c64_batchtospacend_s", "core_mask"], [36, 1, 1, "c.c64_batchtospacend_s", "crops"], [36, 1, 1, "c.c64_batchtospacend_s", "data_size"], [36, 1, 1, "c.c64_batchtospacend_s", "input"], [36, 1, 1, "c.c64_batchtospacend_s", "input_shape"], [36, 1, 1, "c.c64_batchtospacend_s", "output"]], "c64_biasadd_p": [[37, 1, 1, "c.c64_biasadd_p", "data_format"], [37, 1, 1, "c.c64_biasadd_p", "dims"], [37, 1, 1, "c.c64_biasadd_p", "input_bias"], [37, 1, 1, "c.c64_biasadd_p", "input_x"], [37, 1, 1, "c.c64_biasadd_p", "length"], [37, 1, 1, "c.c64_biasadd_p", "output"], [37, 1, 1, "c.c64_biasadd_p", "shape_size"]], "c64_biasadd_s": [[37, 1, 1, "c.c64_biasadd_s", "core_mask"], [37, 1, 1, "c.c64_biasadd_s", "data_format"], [37, 1, 1, "c.c64_biasadd_s", "dims"], [37, 1, 1, "c.c64_biasadd_s", "input_bias"], [37, 1, 1, "c.c64_biasadd_s", "input_x"], [37, 1, 1, "c.c64_biasadd_s", "length"], [37, 1, 1, "c.c64_biasadd_s", "output"], [37, 1, 1, "c.c64_biasadd_s", "shape_size"]], "c64_broadcastto_p": [[41, 1, 1, "c.c64_broadcastto_p", "data_size"], [41, 1, 1, "c.c64_broadcastto_p", "input"], [41, 1, 1, "c.c64_broadcastto_p", "input_shape"], [41, 1, 1, "c.c64_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.c64_broadcastto_p", "output"], [41, 1, 1, "c.c64_broadcastto_p", "output_shape"], [41, 1, 1, "c.c64_broadcastto_p", "output_shape_size"]], "c64_broadcastto_s": [[41, 1, 1, "c.c64_broadcastto_s", "core_mask"], [41, 1, 1, "c.c64_broadcastto_s", "data_size"], [41, 1, 1, "c.c64_broadcastto_s", "input"], [41, 1, 1, "c.c64_broadcastto_s", "input_shape"], [41, 1, 1, "c.c64_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.c64_broadcastto_s", "output"], [41, 1, 1, "c.c64_broadcastto_s", "output_shape"], [41, 1, 1, "c.c64_broadcastto_s", "output_shape_size"]], "c64_concat_p": [[45, 1, 1, "c.c64_concat_p", "axis"], [45, 1, 1, "c.c64_concat_p", "input_ndim"], [45, 1, 1, "c.c64_concat_p", "input_shapes"], [45, 1, 1, "c.c64_concat_p", "inputs"], [45, 1, 1, "c.c64_concat_p", "num_inputs"], [45, 1, 1, "c.c64_concat_p", "output"]], "c64_concat_s": [[45, 1, 1, "c.c64_concat_s", "axis"], [45, 1, 1, "c.c64_concat_s", "core_mask"], [45, 1, 1, "c.c64_concat_s", "input_ndim"], [45, 1, 1, "c.c64_concat_s", "input_shapes"], [45, 1, 1, "c.c64_concat_s", "inputs"], [45, 1, 1, "c.c64_concat_s", "num_inputs"], [45, 1, 1, "c.c64_concat_s", "output"]], "c64_constant_of_shape_p": [[46, 1, 1, "c.c64_constant_of_shape_p", "end"], [46, 1, 1, "c.c64_constant_of_shape_p", "output"], [46, 1, 1, "c.c64_constant_of_shape_p", "start"], [46, 1, 1, "c.c64_constant_of_shape_p", "value_imag"], [46, 1, 1, "c.c64_constant_of_shape_p", "value_real"]], "c64_constant_of_shape_s": [[46, 1, 1, "c.c64_constant_of_shape_s", "core_mask"], [46, 1, 1, "c.c64_constant_of_shape_s", "end"], [46, 1, 1, "c.c64_constant_of_shape_s", "output"], [46, 1, 1, "c.c64_constant_of_shape_s", "start"], [46, 1, 1, "c.c64_constant_of_shape_s", "value_imag"], [46, 1, 1, "c.c64_constant_of_shape_s", "value_real"]], "c64_cumsum_p": [[54, 1, 1, "c.c64_cumsum_p", "axis_dim"], [54, 1, 1, "c.c64_cumsum_p", "exclusive"], [54, 1, 1, "c.c64_cumsum_p", "inner_dim"], [54, 1, 1, "c.c64_cumsum_p", "input"], [54, 1, 1, "c.c64_cumsum_p", "out_dim"], [54, 1, 1, "c.c64_cumsum_p", "output"]], "c64_cumsum_s": [[54, 1, 1, "c.c64_cumsum_s", "axis_dim"], [54, 1, 1, "c.c64_cumsum_s", "core_mask"], [54, 1, 1, "c.c64_cumsum_s", "exclusive"], [54, 1, 1, "c.c64_cumsum_s", "inner_dim"], [54, 1, 1, "c.c64_cumsum_s", "input"], [54, 1, 1, "c.c64_cumsum_s", "out_dim"], [54, 1, 1, "c.c64_cumsum_s", "output"]], "c64_depthtospace_p": [[59, 1, 1, "c.c64_depthtospace_p", "block_size"], [59, 1, 1, "c.c64_depthtospace_p", "data_size"], [59, 1, 1, "c.c64_depthtospace_p", "in_shape"], [59, 1, 1, "c.c64_depthtospace_p", "input"], [59, 1, 1, "c.c64_depthtospace_p", "output"]], "c64_depthtospace_s": [[59, 1, 1, "c.c64_depthtospace_s", "block_size"], [59, 1, 1, "c.c64_depthtospace_s", "core_mask"], [59, 1, 1, "c.c64_depthtospace_s", "data_size"], [59, 1, 1, "c.c64_depthtospace_s", "in_shape"], [59, 1, 1, "c.c64_depthtospace_s", "input"], [59, 1, 1, "c.c64_depthtospace_s", "output"]], "c64_div_fusion_p": [[61, 1, 1, "c.c64_div_fusion_p", "input0"], [61, 1, 1, "c.c64_div_fusion_p", "input1"], [61, 1, 1, "c.c64_div_fusion_p", "length"], [61, 1, 1, "c.c64_div_fusion_p", "output"]], "c64_div_fusion_s": [[61, 1, 1, "c.c64_div_fusion_s", "core_mask"], [61, 1, 1, "c.c64_div_fusion_s", "input0"], [61, 1, 1, "c.c64_div_fusion_s", "input1"], [61, 1, 1, "c.c64_div_fusion_s", "length"], [61, 1, 1, "c.c64_div_fusion_s", "output"]], "c64_eltwise_p": [[67, 1, 1, "c.c64_eltwise_p", "Input0"], [67, 1, 1, "c.c64_eltwise_p", "Input1"], [67, 1, 1, "c.c64_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.c64_eltwise_p", "length"], [67, 1, 1, "c.c64_eltwise_p", "output"]], "c64_eltwise_s": [[67, 1, 1, "c.c64_eltwise_s", "Input0"], [67, 1, 1, "c.c64_eltwise_s", "Input1"], [67, 1, 1, "c.c64_eltwise_s", "core_mask"], [67, 1, 1, "c.c64_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.c64_eltwise_s", "length"], [67, 1, 1, "c.c64_eltwise_s", "output"]], "c64_equal_p": [[70, 1, 1, "c.c64_equal_p", "Input0"], [70, 1, 1, "c.c64_equal_p", "Input1"], [70, 1, 1, "c.c64_equal_p", "length"], [70, 1, 1, "c.c64_equal_p", "output"]], "c64_equal_s": [[70, 1, 1, "c.c64_equal_s", "Input0"], [70, 1, 1, "c.c64_equal_s", "Input1"], [70, 1, 1, "c.c64_equal_s", "core_mask"], [70, 1, 1, "c.c64_equal_s", "length"], [70, 1, 1, "c.c64_equal_s", "output"]], "c64_expfusion_p": [[73, 1, 1, "c.c64_expfusion_p", "dst_data"], [73, 1, 1, "c.c64_expfusion_p", "in_scale"], [73, 1, 1, "c.c64_expfusion_p", "length"], [73, 1, 1, "c.c64_expfusion_p", "out_scale"], [73, 1, 1, "c.c64_expfusion_p", "scale"], [73, 1, 1, "c.c64_expfusion_p", "src_data"]], "c64_expfusion_s": [[73, 1, 1, "c.c64_expfusion_s", "core_mask"], [73, 1, 1, "c.c64_expfusion_s", "dst_data"], [73, 1, 1, "c.c64_expfusion_s", "in_scale"], [73, 1, 1, "c.c64_expfusion_s", "length"], [73, 1, 1, "c.c64_expfusion_s", "out_scale"], [73, 1, 1, "c.c64_expfusion_s", "scale"], [73, 1, 1, "c.c64_expfusion_s", "src_data"]], "c64_extract_features_p": [[55, 1, 1, "c.c64_extract_features_p", "num_strings"], [55, 1, 1, "c.c64_extract_features_p", "output_labels"], [55, 1, 1, "c.c64_extract_features_p", "output_weights"], [55, 1, 1, "c.c64_extract_features_p", "string_lengths"], [55, 1, 1, "c.c64_extract_features_p", "string_pointers"]], "c64_extract_features_s": [[55, 1, 1, "c.c64_extract_features_s", "core_mask"], [55, 1, 1, "c.c64_extract_features_s", "num_strings"], [55, 1, 1, "c.c64_extract_features_s", "output_labels"], [55, 1, 1, "c.c64_extract_features_s", "output_weights"], [55, 1, 1, "c.c64_extract_features_s", "string_lengths"], [55, 1, 1, "c.c64_extract_features_s", "string_pointers"]], "c64_fill_p": [[78, 1, 1, "c.c64_fill_p", "output"], [78, 1, 1, "c.c64_fill_p", "param"], [78, 1, 1, "c.c64_fill_p", "value"]], "c64_fill_s": [[78, 1, 1, "c.c64_fill_s", "core_mask"], [78, 1, 1, "c.c64_fill_s", "output"], [78, 1, 1, "c.c64_fill_s", "param"], [78, 1, 1, "c.c64_fill_s", "value"]], "c64_formattranspose_p": [[85, 1, 1, "c.c64_formattranspose_p", "batch"], [85, 1, 1, "c.c64_formattranspose_p", "channel"], [85, 1, 1, "c.c64_formattranspose_p", "dst_data"], [85, 1, 1, "c.c64_formattranspose_p", "dst_format"], [85, 1, 1, "c.c64_formattranspose_p", "plane"], [85, 1, 1, "c.c64_formattranspose_p", "src_data"], [85, 1, 1, "c.c64_formattranspose_p", "src_format"]], "c64_formattranspose_s": [[85, 1, 1, "c.c64_formattranspose_s", "batch"], [85, 1, 1, "c.c64_formattranspose_s", "channel"], [85, 1, 1, "c.c64_formattranspose_s", "core_mask"], [85, 1, 1, "c.c64_formattranspose_s", "dst_data"], [85, 1, 1, "c.c64_formattranspose_s", "dst_format"], [85, 1, 1, "c.c64_formattranspose_s", "plane"], [85, 1, 1, "c.c64_formattranspose_s", "src_data"], [85, 1, 1, "c.c64_formattranspose_s", "src_format"]], "c64_gather_nd_p": [[89, 1, 1, "c.c64_gather_nd_p", "indices"], [89, 1, 1, "c.c64_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.c64_gather_nd_p", "indices_shape"], [89, 1, 1, "c.c64_gather_nd_p", "input"], [89, 1, 1, "c.c64_gather_nd_p", "input_ndim"], [89, 1, 1, "c.c64_gather_nd_p", "input_shape"], [89, 1, 1, "c.c64_gather_nd_p", "output"]], "c64_gather_nd_s": [[89, 1, 1, "c.c64_gather_nd_s", "core_mask"], [89, 1, 1, "c.c64_gather_nd_s", "indices"], [89, 1, 1, "c.c64_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.c64_gather_nd_s", "indices_shape"], [89, 1, 1, "c.c64_gather_nd_s", "input"], [89, 1, 1, "c.c64_gather_nd_s", "input_ndim"], [89, 1, 1, "c.c64_gather_nd_s", "input_shape"], [89, 1, 1, "c.c64_gather_nd_s", "output"]], "c64_gather_p": [[88, 1, 1, "c.c64_gather_p", "axis"], [88, 1, 1, "c.c64_gather_p", "batch_dims"], [88, 1, 1, "c.c64_gather_p", "indices"], [88, 1, 1, "c.c64_gather_p", "indices_ndim"], [88, 1, 1, "c.c64_gather_p", "indices_shape"], [88, 1, 1, "c.c64_gather_p", "input"], [88, 1, 1, "c.c64_gather_p", "input_ndim"], [88, 1, 1, "c.c64_gather_p", "input_shape"], [88, 1, 1, "c.c64_gather_p", "output"]], "c64_gather_s": [[88, 1, 1, "c.c64_gather_s", "axis"], [88, 1, 1, "c.c64_gather_s", "batch_dims"], [88, 1, 1, "c.c64_gather_s", "core_mask"], [88, 1, 1, "c.c64_gather_s", "indices"], [88, 1, 1, "c.c64_gather_s", "indices_ndim"], [88, 1, 1, "c.c64_gather_s", "indices_shape"], [88, 1, 1, "c.c64_gather_s", "input"], [88, 1, 1, "c.c64_gather_s", "input_ndim"], [88, 1, 1, "c.c64_gather_s", "input_shape"], [88, 1, 1, "c.c64_gather_s", "output"]], "c64_gatherd_p": [[90, 1, 1, "c.c64_gatherd_p", "dim"], [90, 1, 1, "c.c64_gatherd_p", "index"], [90, 1, 1, "c.c64_gatherd_p", "index_shape"], [90, 1, 1, "c.c64_gatherd_p", "input_shape"], [90, 1, 1, "c.c64_gatherd_p", "input_shape_size"], [90, 1, 1, "c.c64_gatherd_p", "input_x"], [90, 1, 1, "c.c64_gatherd_p", "output"]], "c64_gatherd_s": [[90, 1, 1, "c.c64_gatherd_s", "core_mask"], [90, 1, 1, "c.c64_gatherd_s", "dim"], [90, 1, 1, "c.c64_gatherd_s", "index"], [90, 1, 1, "c.c64_gatherd_s", "index_shape"], [90, 1, 1, "c.c64_gatherd_s", "input_shape"], [90, 1, 1, "c.c64_gatherd_s", "input_shape_size"], [90, 1, 1, "c.c64_gatherd_s", "input_x"], [90, 1, 1, "c.c64_gatherd_s", "output"]], "c64_isfinite_p": [[99, 1, 1, "c.c64_isfinite_p", "Input"], [99, 1, 1, "c.c64_isfinite_p", "length"], [99, 1, 1, "c.c64_isfinite_p", "output"]], "c64_isfinite_s": [[99, 1, 1, "c.c64_isfinite_s", "Input"], [99, 1, 1, "c.c64_isfinite_s", "core_mask"], [99, 1, 1, "c.c64_isfinite_s", "length"], [99, 1, 1, "c.c64_isfinite_s", "output"]], "c64_matmulfusion_p": [[121, 1, 1, "c.c64_matmulfusion_p", "A"], [121, 1, 1, "c.c64_matmulfusion_p", "B"], [121, 1, 1, "c.c64_matmulfusion_p", "C"], [121, 1, 1, "c.c64_matmulfusion_p", "K"], [121, 1, 1, "c.c64_matmulfusion_p", "M"], [121, 1, 1, "c.c64_matmulfusion_p", "N"], [121, 1, 1, "c.c64_matmulfusion_p", "activation_type"], [121, 1, 1, "c.c64_matmulfusion_p", "bias"]], "c64_matmulfusion_s": [[121, 1, 1, "c.c64_matmulfusion_s", "A"], [121, 1, 1, "c.c64_matmulfusion_s", "B"], [121, 1, 1, "c.c64_matmulfusion_s", "C"], [121, 1, 1, "c.c64_matmulfusion_s", "K"], [121, 1, 1, "c.c64_matmulfusion_s", "M"], [121, 1, 1, "c.c64_matmulfusion_s", "N"], [121, 1, 1, "c.c64_matmulfusion_s", "activation_type"], [121, 1, 1, "c.c64_matmulfusion_s", "bias"], [121, 1, 1, "c.c64_matmulfusion_s", "core_mask"]], "c64_mul_p": [[130, 1, 1, "c.c64_mul_p", "input0"], [130, 1, 1, "c.c64_mul_p", "input1"], [130, 1, 1, "c.c64_mul_p", "length"], [130, 1, 1, "c.c64_mul_p", "output"]], "c64_mul_s": [[130, 1, 1, "c.c64_mul_s", "core_mask"], [130, 1, 1, "c.c64_mul_s", "input0"], [130, 1, 1, "c.c64_mul_s", "input1"], [130, 1, 1, "c.c64_mul_s", "length"], [130, 1, 1, "c.c64_mul_s", "output"]], "c64_neg_grad_p": [[133, 1, 1, "c.c64_neg_grad_p", "Input"], [133, 1, 1, "c.c64_neg_grad_p", "length"], [133, 1, 1, "c.c64_neg_grad_p", "output"]], "c64_neg_grad_s": [[133, 1, 1, "c.c64_neg_grad_s", "Input"], [133, 1, 1, "c.c64_neg_grad_s", "core_mask"], [133, 1, 1, "c.c64_neg_grad_s", "length"], [133, 1, 1, "c.c64_neg_grad_s", "output"]], "c64_neg_p": [[132, 1, 1, "c.c64_neg_p", "Input"], [132, 1, 1, "c.c64_neg_p", "length"], [132, 1, 1, "c.c64_neg_p", "output"]], "c64_neg_s": [[132, 1, 1, "c.c64_neg_s", "Input"], [132, 1, 1, "c.c64_neg_s", "core_mask"], [132, 1, 1, "c.c64_neg_s", "length"], [132, 1, 1, "c.c64_neg_s", "output"]], "c64_nonzero_p": [[137, 1, 1, "c.c64_nonzero_p", "dim_strides"], [137, 1, 1, "c.c64_nonzero_p", "input"], [137, 1, 1, "c.c64_nonzero_p", "input_rank"], [137, 1, 1, "c.c64_nonzero_p", "length"], [137, 1, 1, "c.c64_nonzero_p", "non_zero_num"], [137, 1, 1, "c.c64_nonzero_p", "output"], [137, 1, 1, "c.c64_nonzero_p", "shape"]], "c64_nonzero_s": [[137, 1, 1, "c.c64_nonzero_s", "core_mask"], [137, 1, 1, "c.c64_nonzero_s", "dim_strides"], [137, 1, 1, "c.c64_nonzero_s", "input"], [137, 1, 1, "c.c64_nonzero_s", "input_rank"], [137, 1, 1, "c.c64_nonzero_s", "length"], [137, 1, 1, "c.c64_nonzero_s", "non_zero_num"], [137, 1, 1, "c.c64_nonzero_s", "output"], [137, 1, 1, "c.c64_nonzero_s", "shape"]], "c64_not_equal_p": [[138, 1, 1, "c.c64_not_equal_p", "Input0"], [138, 1, 1, "c.c64_not_equal_p", "Input1"], [138, 1, 1, "c.c64_not_equal_p", "length"], [138, 1, 1, "c.c64_not_equal_p", "output"]], "c64_not_equal_s": [[138, 1, 1, "c.c64_not_equal_s", "Input0"], [138, 1, 1, "c.c64_not_equal_s", "Input1"], [138, 1, 1, "c.c64_not_equal_s", "core_mask"], [138, 1, 1, "c.c64_not_equal_s", "length"], [138, 1, 1, "c.c64_not_equal_s", "output"]], "c64_onehot_p": [[139, 1, 1, "c.c64_onehot_p", "axis"], [139, 1, 1, "c.c64_onehot_p", "depth"], [139, 1, 1, "c.c64_onehot_p", "indices"], [139, 1, 1, "c.c64_onehot_p", "indices_shape"], [139, 1, 1, "c.c64_onehot_p", "indices_shape_size"], [139, 1, 1, "c.c64_onehot_p", "on_off"], [139, 1, 1, "c.c64_onehot_p", "output"], [139, 1, 1, "c.c64_onehot_p", "support_neg_index"]], "c64_onehot_s": [[139, 1, 1, "c.c64_onehot_s", "axis"], [139, 1, 1, "c.c64_onehot_s", "core_mask"], [139, 1, 1, "c.c64_onehot_s", "depth"], [139, 1, 1, "c.c64_onehot_s", "indices"], [139, 1, 1, "c.c64_onehot_s", "indices_shape"], [139, 1, 1, "c.c64_onehot_s", "indices_shape_size"], [139, 1, 1, "c.c64_onehot_s", "on_off"], [139, 1, 1, "c.c64_onehot_s", "output"], [139, 1, 1, "c.c64_onehot_s", "support_neg_index"]], "c64_ones_like_p": [[140, 1, 1, "c.c64_ones_like_p", "length"], [140, 1, 1, "c.c64_ones_like_p", "output"]], "c64_ones_like_s": [[140, 1, 1, "c.c64_ones_like_s", "core_mask"], [140, 1, 1, "c.c64_ones_like_s", "length"], [140, 1, 1, "c.c64_ones_like_s", "output"]], "c64_padfusion_p": [[141, 1, 1, "c.c64_padfusion_p", "params"]], "c64_padfusion_s": [[141, 1, 1, "c.c64_padfusion_s", "core_mask"], [141, 1, 1, "c.c64_padfusion_s", "params"]], "c64_real_div_p": [[152, 1, 1, "c.c64_real_div_p", "input0"], [152, 1, 1, "c.c64_real_div_p", "input1"], [152, 1, 1, "c.c64_real_div_p", "length"], [152, 1, 1, "c.c64_real_div_p", "output"]], "c64_real_div_s": [[152, 1, 1, "c.c64_real_div_s", "core_mask"], [152, 1, 1, "c.c64_real_div_s", "input0"], [152, 1, 1, "c.c64_real_div_s", "input1"], [152, 1, 1, "c.c64_real_div_s", "length"], [152, 1, 1, "c.c64_real_div_s", "output"]], "c64_reciprocal_p": [[153, 1, 1, "c.c64_reciprocal_p", "Input"], [153, 1, 1, "c.c64_reciprocal_p", "length"], [153, 1, 1, "c.c64_reciprocal_p", "output"]], "c64_reciprocal_s": [[153, 1, 1, "c.c64_reciprocal_s", "Input"], [153, 1, 1, "c.c64_reciprocal_s", "core_mask"], [153, 1, 1, "c.c64_reciprocal_s", "length"], [153, 1, 1, "c.c64_reciprocal_s", "output"]], "c64_reduceall_p": [[21, 1, 1, "c.c64_reduceall_p", "axis_size"], [21, 1, 1, "c.c64_reduceall_p", "dst_data"], [21, 1, 1, "c.c64_reduceall_p", "inner_size"], [21, 1, 1, "c.c64_reduceall_p", "outer_size"], [21, 1, 1, "c.c64_reduceall_p", "src_data"]], "c64_reduceall_s": [[21, 1, 1, "c.c64_reduceall_s", "axis_size"], [21, 1, 1, "c.c64_reduceall_s", "core_mask"], [21, 1, 1, "c.c64_reduceall_s", "dst_data"], [21, 1, 1, "c.c64_reduceall_s", "inner_size"], [21, 1, 1, "c.c64_reduceall_s", "outer_size"], [21, 1, 1, "c.c64_reduceall_s", "src_data"]], "c64_reshape_p": [[156, 1, 1, "c.c64_reshape_p", "input"], [156, 1, 1, "c.c64_reshape_p", "length"], [156, 1, 1, "c.c64_reshape_p", "output"]], "c64_reshape_s": [[156, 1, 1, "c.c64_reshape_s", "core_mask"], [156, 1, 1, "c.c64_reshape_s", "input"], [156, 1, 1, "c.c64_reshape_s", "length"], [156, 1, 1, "c.c64_reshape_s", "output"]], "c64_rfft_p": [[161, 1, 1, "c.c64_rfft_p", "dir"], [161, 1, 1, "c.c64_rfft_p", "fft_size"], [161, 1, 1, "c.c64_rfft_p", "input"], [161, 1, 1, "c.c64_rfft_p", "output"], [161, 1, 1, "c.c64_rfft_p", "scratch_ptr"]], "c64_rfft_s": [[161, 1, 1, "c.c64_rfft_s", "core_mask"], [161, 1, 1, "c.c64_rfft_s", "dir"], [161, 1, 1, "c.c64_rfft_s", "fft_size1"], [161, 1, 1, "c.c64_rfft_s", "fft_size2"], [161, 1, 1, "c.c64_rfft_s", "input"], [161, 1, 1, "c.c64_rfft_s", "output"], [161, 1, 1, "c.c64_rfft_s", "scratch_ptr"], [161, 1, 1, "c.c64_rfft_s", "twiddle"]], "c64_rsqrt_p": [[164, 1, 1, "c.c64_rsqrt_p", "dst"], [164, 1, 1, "c.c64_rsqrt_p", "length"], [164, 1, 1, "c.c64_rsqrt_p", "src"]], "c64_rsqrt_s": [[164, 1, 1, "c.c64_rsqrt_s", "core_mask"], [164, 1, 1, "c.c64_rsqrt_s", "dst"], [164, 1, 1, "c.c64_rsqrt_s", "length"], [164, 1, 1, "c.c64_rsqrt_s", "src"]], "c64_scatter_elements_p": [[167, 1, 1, "c.c64_scatter_elements_p", "core_mask"], [167, 1, 1, "c.c64_scatter_elements_p", "indices"], [167, 1, 1, "c.c64_scatter_elements_p", "input"], [167, 1, 1, "c.c64_scatter_elements_p", "output"], [167, 1, 1, "c.c64_scatter_elements_p", "param"], [167, 1, 1, "c.c64_scatter_elements_p", "updates"]], "c64_scatter_elements_s": [[167, 1, 1, "c.c64_scatter_elements_s", "core_mask"], [167, 1, 1, "c.c64_scatter_elements_s", "indices"], [167, 1, 1, "c.c64_scatter_elements_s", "input"], [167, 1, 1, "c.c64_scatter_elements_s", "output"], [167, 1, 1, "c.c64_scatter_elements_s", "param"], [167, 1, 1, "c.c64_scatter_elements_s", "updates"]], "c64_scatter_nd_p": [[168, 1, 1, "c.c64_scatter_nd_p", "indices"], [168, 1, 1, "c.c64_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.c64_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.c64_scatter_nd_p", "output"], [168, 1, 1, "c.c64_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.c64_scatter_nd_p", "output_shape"], [168, 1, 1, "c.c64_scatter_nd_p", "updates"]], "c64_scatter_nd_s": [[168, 1, 1, "c.c64_scatter_nd_s", "core_mask"], [168, 1, 1, "c.c64_scatter_nd_s", "indices"], [168, 1, 1, "c.c64_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.c64_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.c64_scatter_nd_s", "output"], [168, 1, 1, "c.c64_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.c64_scatter_nd_s", "output_shape"], [168, 1, 1, "c.c64_scatter_nd_s", "updates"]], "c64_scatter_nd_update_p": [[169, 1, 1, "c.c64_scatter_nd_update_p", "indices"], [169, 1, 1, "c.c64_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.c64_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.c64_scatter_nd_update_p", "output"], [169, 1, 1, "c.c64_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.c64_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.c64_scatter_nd_update_p", "updates"]], "c64_scatter_nd_update_s": [[169, 1, 1, "c.c64_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.c64_scatter_nd_update_s", "indices"], [169, 1, 1, "c.c64_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.c64_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.c64_scatter_nd_update_s", "output"], [169, 1, 1, "c.c64_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.c64_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.c64_scatter_nd_update_s", "updates"]], "c64_select_p": [[170, 1, 1, "c.c64_select_p", "condition"], [170, 1, 1, "c.c64_select_p", "index_list1"], [170, 1, 1, "c.c64_select_p", "index_list2"], [170, 1, 1, "c.c64_select_p", "index_list3"], [170, 1, 1, "c.c64_select_p", "input0"], [170, 1, 1, "c.c64_select_p", "input1"], [170, 1, 1, "c.c64_select_p", "is_broadcast"], [170, 1, 1, "c.c64_select_p", "output"], [170, 1, 1, "c.c64_select_p", "output_dims"], [170, 1, 1, "c.c64_select_p", "output_dims_num"]], "c64_select_s": [[170, 1, 1, "c.c64_select_s", "condition"], [170, 1, 1, "c.c64_select_s", "core_mask"], [170, 1, 1, "c.c64_select_s", "index_list1"], [170, 1, 1, "c.c64_select_s", "index_list2"], [170, 1, 1, "c.c64_select_s", "index_list3"], [170, 1, 1, "c.c64_select_s", "input0"], [170, 1, 1, "c.c64_select_s", "input1"], [170, 1, 1, "c.c64_select_s", "is_broadcast"], [170, 1, 1, "c.c64_select_s", "output"], [170, 1, 1, "c.c64_select_s", "output_dims"], [170, 1, 1, "c.c64_select_s", "output_dims_num"]], "c64_slice_p": [[178, 1, 1, "c.c64_slice_p", "begin"], [178, 1, 1, "c.c64_slice_p", "input"], [178, 1, 1, "c.c64_slice_p", "input_shape"], [178, 1, 1, "c.c64_slice_p", "ndim"], [178, 1, 1, "c.c64_slice_p", "output"], [178, 1, 1, "c.c64_slice_p", "size"]], "c64_slice_s": [[178, 1, 1, "c.c64_slice_s", "begin"], [178, 1, 1, "c.c64_slice_s", "core_mask"], [178, 1, 1, "c.c64_slice_s", "input"], [178, 1, 1, "c.c64_slice_s", "input_shape"], [178, 1, 1, "c.c64_slice_s", "ndim"], [178, 1, 1, "c.c64_slice_s", "output"], [178, 1, 1, "c.c64_slice_s", "size"]], "c64_spacetobatch_p": [[183, 1, 1, "c.c64_spacetobatch_p", "block_size"], [183, 1, 1, "c.c64_spacetobatch_p", "data_size"], [183, 1, 1, "c.c64_spacetobatch_p", "input"], [183, 1, 1, "c.c64_spacetobatch_p", "input_shape"], [183, 1, 1, "c.c64_spacetobatch_p", "output"], [183, 1, 1, "c.c64_spacetobatch_p", "paddings"]], "c64_spacetobatch_s": [[183, 1, 1, "c.c64_spacetobatch_s", "block_size"], [183, 1, 1, "c.c64_spacetobatch_s", "core_mask"], [183, 1, 1, "c.c64_spacetobatch_s", "data_size"], [183, 1, 1, "c.c64_spacetobatch_s", "input"], [183, 1, 1, "c.c64_spacetobatch_s", "input_shape"], [183, 1, 1, "c.c64_spacetobatch_s", "output"], [183, 1, 1, "c.c64_spacetobatch_s", "paddings"]], "c64_spacetobatchnd_p": [[184, 1, 1, "c.c64_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.c64_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.c64_spacetobatchnd_p", "input"], [184, 1, 1, "c.c64_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.c64_spacetobatchnd_p", "output"], [184, 1, 1, "c.c64_spacetobatchnd_p", "paddings"]], "c64_spacetobatchnd_s": [[184, 1, 1, "c.c64_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.c64_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.c64_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.c64_spacetobatchnd_s", "input"], [184, 1, 1, "c.c64_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.c64_spacetobatchnd_s", "output"], [184, 1, 1, "c.c64_spacetobatchnd_s", "paddings"]], "c64_spacetodepth_p": [[185, 1, 1, "c.c64_spacetodepth_p", "block"], [185, 1, 1, "c.c64_spacetodepth_p", "data_size"], [185, 1, 1, "c.c64_spacetodepth_p", "in_shape"], [185, 1, 1, "c.c64_spacetodepth_p", "input"], [185, 1, 1, "c.c64_spacetodepth_p", "output"]], "c64_spacetodepth_s": [[185, 1, 1, "c.c64_spacetodepth_s", "block"], [185, 1, 1, "c.c64_spacetodepth_s", "core_mask"], [185, 1, 1, "c.c64_spacetodepth_s", "data_size"], [185, 1, 1, "c.c64_spacetodepth_s", "in_shape"], [185, 1, 1, "c.c64_spacetodepth_s", "input"], [185, 1, 1, "c.c64_spacetodepth_s", "output"]], "c64_sparsefillemptyrows_p": [[187, 1, 1, "c.c64_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_p", "values_ptr"]], "c64_sparsefillemptyrows_s": [[187, 1, 1, "c.c64_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.c64_sparsefillemptyrows_s", "values_ptr"]], "c64_sparsesegmentsum_p": [[189, 1, 1, "c.c64_sparsesegmentsum_p", "in_data"], [189, 1, 1, "c.c64_sparsesegmentsum_p", "in_data_shape"], [189, 1, 1, "c.c64_sparsesegmentsum_p", "in_data_shape_size"], [189, 1, 1, "c.c64_sparsesegmentsum_p", "in_indices"], [189, 1, 1, "c.c64_sparsesegmentsum_p", "in_indices_size"], [189, 1, 1, "c.c64_sparsesegmentsum_p", "in_segment_ids"], [189, 1, 1, "c.c64_sparsesegmentsum_p", "out_data"], [189, 1, 1, "c.c64_sparsesegmentsum_p", "out_data_shape"]], "c64_sparsesegmentsum_s": [[189, 1, 1, "c.c64_sparsesegmentsum_s", "core_mask"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "in_data"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "in_data_shape"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "in_data_shape_size"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "in_indices"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "in_indices_size"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "in_segment_ids"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "out_data"], [189, 1, 1, "c.c64_sparsesegmentsum_s", "out_data_shape"]], "c64_sparsetodense_p": [[190, 1, 1, "c.c64_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.c64_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.c64_sparsetodense_p", "output"], [190, 1, 1, "c.c64_sparsetodense_p", "output_strides"], [190, 1, 1, "c.c64_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.c64_sparsetodense_p", "sparse_values"]], "c64_sparsetodense_s": [[190, 1, 1, "c.c64_sparsetodense_s", "core_mask"], [190, 1, 1, "c.c64_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.c64_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.c64_sparsetodense_s", "output"], [190, 1, 1, "c.c64_sparsetodense_s", "output_strides"], [190, 1, 1, "c.c64_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.c64_sparsetodense_s", "sparse_values"]], "c64_splice_p": [[191, 1, 1, "c.c64_splice_p", "context_dim"], [191, 1, 1, "c.c64_splice_p", "dst_col"], [191, 1, 1, "c.c64_splice_p", "dst_data"], [191, 1, 1, "c.c64_splice_p", "dst_row"], [191, 1, 1, "c.c64_splice_p", "forward_indexes"], [191, 1, 1, "c.c64_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.c64_splice_p", "src_col"], [191, 1, 1, "c.c64_splice_p", "src_data"], [191, 1, 1, "c.c64_splice_p", "src_row"]], "c64_splice_s": [[191, 1, 1, "c.c64_splice_s", "context_dim"], [191, 1, 1, "c.c64_splice_s", "core_mask"], [191, 1, 1, "c.c64_splice_s", "dst_col"], [191, 1, 1, "c.c64_splice_s", "dst_data"], [191, 1, 1, "c.c64_splice_s", "dst_row"], [191, 1, 1, "c.c64_splice_s", "forward_indexes"], [191, 1, 1, "c.c64_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.c64_splice_s", "src_col"], [191, 1, 1, "c.c64_splice_s", "src_data"], [191, 1, 1, "c.c64_splice_s", "src_row"]], "c64_split_p": [[192, 1, 1, "c.c64_split_p", "axis"], [192, 1, 1, "c.c64_split_p", "input"], [192, 1, 1, "c.c64_split_p", "input_ndim"], [192, 1, 1, "c.c64_split_p", "input_shape"], [192, 1, 1, "c.c64_split_p", "num_split"], [192, 1, 1, "c.c64_split_p", "outputs"], [192, 1, 1, "c.c64_split_p", "split_sizes"]], "c64_split_s": [[192, 1, 1, "c.c64_split_s", "axis"], [192, 1, 1, "c.c64_split_s", "core_mask"], [192, 1, 1, "c.c64_split_s", "input"], [192, 1, 1, "c.c64_split_s", "input_ndim"], [192, 1, 1, "c.c64_split_s", "input_shape"], [192, 1, 1, "c.c64_split_s", "num_split"], [192, 1, 1, "c.c64_split_s", "outputs"], [192, 1, 1, "c.c64_split_s", "split_sizes"]], "c64_split_with_overlap_p": [[193, 1, 1, "c.c64_split_with_overlap_p", "axis"], [193, 1, 1, "c.c64_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.c64_split_with_overlap_p", "input"], [193, 1, 1, "c.c64_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.c64_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.c64_split_with_overlap_p", "num_split"], [193, 1, 1, "c.c64_split_with_overlap_p", "outputs"], [193, 1, 1, "c.c64_split_with_overlap_p", "start_indices"]], "c64_split_with_overlap_s": [[193, 1, 1, "c.c64_split_with_overlap_s", "axis"], [193, 1, 1, "c.c64_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.c64_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.c64_split_with_overlap_s", "input"], [193, 1, 1, "c.c64_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.c64_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.c64_split_with_overlap_s", "num_split"], [193, 1, 1, "c.c64_split_with_overlap_s", "outputs"], [193, 1, 1, "c.c64_split_with_overlap_s", "start_indices"]], "c64_sqrt_p": [[194, 1, 1, "c.c64_sqrt_p", "dst_data"], [194, 1, 1, "c.c64_sqrt_p", "length"], [194, 1, 1, "c.c64_sqrt_p", "src_data"]], "c64_sqrt_s": [[194, 1, 1, "c.c64_sqrt_s", "core_mask"], [194, 1, 1, "c.c64_sqrt_s", "dst_data"], [194, 1, 1, "c.c64_sqrt_s", "length"], [194, 1, 1, "c.c64_sqrt_s", "src_data"]], "c64_sqrtgrad_p": [[195, 1, 1, "c.c64_sqrtgrad_p", "input1"], [195, 1, 1, "c.c64_sqrtgrad_p", "input2"], [195, 1, 1, "c.c64_sqrtgrad_p", "output"], [195, 1, 1, "c.c64_sqrtgrad_p", "size"]], "c64_sqrtgrad_s": [[195, 1, 1, "c.c64_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.c64_sqrtgrad_s", "input1"], [195, 1, 1, "c.c64_sqrtgrad_s", "input2"], [195, 1, 1, "c.c64_sqrtgrad_s", "output"], [195, 1, 1, "c.c64_sqrtgrad_s", "size"]], "c64_square_p": [[196, 1, 1, "c.c64_square_p", "dst"], [196, 1, 1, "c.c64_square_p", "length"], [196, 1, 1, "c.c64_square_p", "src"]], "c64_square_s": [[196, 1, 1, "c.c64_square_s", "core_mask"], [196, 1, 1, "c.c64_square_s", "dst"], [196, 1, 1, "c.c64_square_s", "length"], [196, 1, 1, "c.c64_square_s", "src"]], "c64_squaredifference_p": [[197, 1, 1, "c.c64_squaredifference_p", "input0"], [197, 1, 1, "c.c64_squaredifference_p", "input1"], [197, 1, 1, "c.c64_squaredifference_p", "length"], [197, 1, 1, "c.c64_squaredifference_p", "output"]], "c64_squaredifference_s": [[197, 1, 1, "c.c64_squaredifference_s", "core_mask"], [197, 1, 1, "c.c64_squaredifference_s", "input0"], [197, 1, 1, "c.c64_squaredifference_s", "input1"], [197, 1, 1, "c.c64_squaredifference_s", "length"], [197, 1, 1, "c.c64_squaredifference_s", "output"]], "c64_stack_p": [[199, 1, 1, "c.c64_stack_p", "axis"], [199, 1, 1, "c.c64_stack_p", "input_ndim"], [199, 1, 1, "c.c64_stack_p", "input_shape"], [199, 1, 1, "c.c64_stack_p", "inputs"], [199, 1, 1, "c.c64_stack_p", "num_inputs"], [199, 1, 1, "c.c64_stack_p", "output"]], "c64_stack_s": [[199, 1, 1, "c.c64_stack_s", "axis"], [199, 1, 1, "c.c64_stack_s", "core_mask"], [199, 1, 1, "c.c64_stack_s", "input_ndim"], [199, 1, 1, "c.c64_stack_s", "input_shape"], [199, 1, 1, "c.c64_stack_s", "inputs"], [199, 1, 1, "c.c64_stack_s", "num_inputs"], [199, 1, 1, "c.c64_stack_s", "output"]], "c64_subext_p": [[202, 1, 1, "c.c64_subext_p", "alpha"], [202, 1, 1, "c.c64_subext_p", "input0"], [202, 1, 1, "c.c64_subext_p", "input1"], [202, 1, 1, "c.c64_subext_p", "output"], [202, 1, 1, "c.c64_subext_p", "size"]], "c64_subext_s": [[202, 1, 1, "c.c64_subext_s", "alpha"], [202, 1, 1, "c.c64_subext_s", "core_mask"], [202, 1, 1, "c.c64_subext_s", "input0"], [202, 1, 1, "c.c64_subext_s", "input1"], [202, 1, 1, "c.c64_subext_s", "output"], [202, 1, 1, "c.c64_subext_s", "size"]], "c64_subrelu6_p": [[202, 1, 1, "c.c64_subrelu6_p", "input0"], [202, 1, 1, "c.c64_subrelu6_p", "input1"], [202, 1, 1, "c.c64_subrelu6_p", "output"], [202, 1, 1, "c.c64_subrelu6_p", "size"]], "c64_subrelu6_s": [[202, 1, 1, "c.c64_subrelu6_s", "core_mask"], [202, 1, 1, "c.c64_subrelu6_s", "input0"], [202, 1, 1, "c.c64_subrelu6_s", "input1"], [202, 1, 1, "c.c64_subrelu6_s", "output"], [202, 1, 1, "c.c64_subrelu6_s", "size"]], "c64_subrelu_p": [[202, 1, 1, "c.c64_subrelu_p", "input0"], [202, 1, 1, "c.c64_subrelu_p", "input1"], [202, 1, 1, "c.c64_subrelu_p", "output"], [202, 1, 1, "c.c64_subrelu_p", "size"]], "c64_subrelu_s": [[202, 1, 1, "c.c64_subrelu_s", "core_mask"], [202, 1, 1, "c.c64_subrelu_s", "input0"], [202, 1, 1, "c.c64_subrelu_s", "input1"], [202, 1, 1, "c.c64_subrelu_s", "output"], [202, 1, 1, "c.c64_subrelu_s", "size"]], "c64_tensor_scatter_add_p": [[206, 1, 1, "c.c64_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "input"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "output"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.c64_tensor_scatter_add_p", "updates"]], "c64_tensor_scatter_add_s": [[206, 1, 1, "c.c64_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "input"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "output"], [206, 1, 1, "c.c64_tensor_scatter_add_s", "updates"]], "c64_tensorarrayread_p": [[208, 1, 1, "c.c64_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.c64_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.c64_tensorarrayread_p", "index"], [208, 1, 1, "c.c64_tensorarrayread_p", "output_data"], [208, 1, 1, "c.c64_tensorarrayread_p", "output_size"]], "c64_tensorarrayread_s": [[208, 1, 1, "c.c64_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.c64_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.c64_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.c64_tensorarrayread_s", "index"], [208, 1, 1, "c.c64_tensorarrayread_s", "output_data"], [208, 1, 1, "c.c64_tensorarrayread_s", "output_size"]], "c64_tensorlistfromtensor_p": [[210, 1, 1, "c.c64_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.c64_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.c64_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.c64_tensorlistfromtensor_p", "output_tensors"]], "c64_tensorlistfromtensor_s": [[210, 1, 1, "c.c64_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.c64_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.c64_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.c64_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.c64_tensorlistfromtensor_s", "output_tensors"]], "c64_tile_p": [[215, 1, 1, "c.c64_tile_p", "input"], [215, 1, 1, "c.c64_tile_p", "input_shape"], [215, 1, 1, "c.c64_tile_p", "output"], [215, 1, 1, "c.c64_tile_p", "stride"], [215, 1, 1, "c.c64_tile_p", "tile_dim"], [215, 1, 1, "c.c64_tile_p", "tile_num"]], "c64_tile_s": [[215, 1, 1, "c.c64_tile_s", "core_mask"], [215, 1, 1, "c.c64_tile_s", "input"], [215, 1, 1, "c.c64_tile_s", "input_shape"], [215, 1, 1, "c.c64_tile_s", "output"], [215, 1, 1, "c.c64_tile_s", "stride"], [215, 1, 1, "c.c64_tile_s", "tile_dim"], [215, 1, 1, "c.c64_tile_s", "tile_num"]], "c64_transpose_p": [[217, 1, 1, "c.c64_transpose_p", "in_data"], [217, 1, 1, "c.c64_transpose_p", "num_axes"], [217, 1, 1, "c.c64_transpose_p", "out_data"], [217, 1, 1, "c.c64_transpose_p", "out_strides"], [217, 1, 1, "c.c64_transpose_p", "output_shape"], [217, 1, 1, "c.c64_transpose_p", "perm"], [217, 1, 1, "c.c64_transpose_p", "strides"]], "c64_transpose_s": [[217, 1, 1, "c.c64_transpose_s", "core_mask"], [217, 1, 1, "c.c64_transpose_s", "in_data"], [217, 1, 1, "c.c64_transpose_s", "num_axes"], [217, 1, 1, "c.c64_transpose_s", "out_data"], [217, 1, 1, "c.c64_transpose_s", "out_strides"], [217, 1, 1, "c.c64_transpose_s", "output_shape"], [217, 1, 1, "c.c64_transpose_s", "perm"], [217, 1, 1, "c.c64_transpose_s", "strides"]], "c64_tril_p": [[218, 1, 1, "c.c64_tril_p", "dst"], [218, 1, 1, "c.c64_tril_p", "height"], [218, 1, 1, "c.c64_tril_p", "k"], [218, 1, 1, "c.c64_tril_p", "out_elems"], [218, 1, 1, "c.c64_tril_p", "src"], [218, 1, 1, "c.c64_tril_p", "width"]], "c64_tril_s": [[218, 1, 1, "c.c64_tril_s", "core_mask"], [218, 1, 1, "c.c64_tril_s", "dst"], [218, 1, 1, "c.c64_tril_s", "height"], [218, 1, 1, "c.c64_tril_s", "k"], [218, 1, 1, "c.c64_tril_s", "out_elems"], [218, 1, 1, "c.c64_tril_s", "src"], [218, 1, 1, "c.c64_tril_s", "width"]], "c64_triu_p": [[219, 1, 1, "c.c64_triu_p", "dst"], [219, 1, 1, "c.c64_triu_p", "height"], [219, 1, 1, "c.c64_triu_p", "k"], [219, 1, 1, "c.c64_triu_p", "out_elems"], [219, 1, 1, "c.c64_triu_p", "src"], [219, 1, 1, "c.c64_triu_p", "width"]], "c64_triu_s": [[219, 1, 1, "c.c64_triu_s", "core_mask"], [219, 1, 1, "c.c64_triu_s", "dst"], [219, 1, 1, "c.c64_triu_s", "height"], [219, 1, 1, "c.c64_triu_s", "k"], [219, 1, 1, "c.c64_triu_s", "out_elems"], [219, 1, 1, "c.c64_triu_s", "src"], [219, 1, 1, "c.c64_triu_s", "width"]], "c64_unsorted_segment_sum_p": [[222, 1, 1, "c.c64_unsorted_segment_sum_p", "dim0"], [222, 1, 1, "c.c64_unsorted_segment_sum_p", "dim1"], [222, 1, 1, "c.c64_unsorted_segment_sum_p", "id_max"], [222, 1, 1, "c.c64_unsorted_segment_sum_p", "index"], [222, 1, 1, "c.c64_unsorted_segment_sum_p", "input"], [222, 1, 1, "c.c64_unsorted_segment_sum_p", "output"]], "c64_unsorted_segment_sum_s": [[222, 1, 1, "c.c64_unsorted_segment_sum_s", "core_mask"], [222, 1, 1, "c.c64_unsorted_segment_sum_s", "dim0"], [222, 1, 1, "c.c64_unsorted_segment_sum_s", "dim1"], [222, 1, 1, "c.c64_unsorted_segment_sum_s", "id_max"], [222, 1, 1, "c.c64_unsorted_segment_sum_s", "index"], [222, 1, 1, "c.c64_unsorted_segment_sum_s", "input"], [222, 1, 1, "c.c64_unsorted_segment_sum_s", "output"]], "c64_where_p": [[225, 1, 1, "c.c64_where_p", "condition"], [225, 1, 1, "c.c64_where_p", "input0"], [225, 1, 1, "c.c64_where_p", "input1"], [225, 1, 1, "c.c64_where_p", "length"], [225, 1, 1, "c.c64_where_p", "output"]], "c64_where_s": [[225, 1, 1, "c.c64_where_s", "condition"], [225, 1, 1, "c.c64_where_s", "core_mask"], [225, 1, 1, "c.c64_where_s", "input0"], [225, 1, 1, "c.c64_where_s", "input1"], [225, 1, 1, "c.c64_where_s", "length"], [225, 1, 1, "c.c64_where_s", "output"]], "c64_zerolike_p": [[226, 1, 1, "c.c64_zerolike_p", "length"], [226, 1, 1, "c.c64_zerolike_p", "output"]], "c64_zerolike_s": [[226, 1, 1, "c.c64_zerolike_s", "core_mask"], [226, 1, 1, "c.c64_zerolike_s", "length"], [226, 1, 1, "c.c64_zerolike_s", "output"]], "castc128Toc128_p": [[42, 1, 1, "c.castc128Toc128_p", "input"], [42, 1, 1, "c.castc128Toc128_p", "length"], [42, 1, 1, "c.castc128Toc128_p", "output"]], "castc128Toc128_s": [[42, 1, 1, "c.castc128Toc128_s", "core_mask"], [42, 1, 1, "c.castc128Toc128_s", "input"], [42, 1, 1, "c.castc128Toc128_s", "length"], [42, 1, 1, "c.castc128Toc128_s", "output"]], "castc64Toc64_p": [[42, 1, 1, "c.castc64Toc64_p", "input"], [42, 1, 1, "c.castc64Toc64_p", "length"], [42, 1, 1, "c.castc64Toc64_p", "output"]], "castc64Toc64_s": [[42, 1, 1, "c.castc64Toc64_s", "core_mask"], [42, 1, 1, "c.castc64Toc64_s", "input"], [42, 1, 1, "c.castc64Toc64_s", "length"], [42, 1, 1, "c.castc64Toc64_s", "output"]], "castdpTodp_p": [[42, 1, 1, "c.castdpTodp_p", "input"], [42, 1, 1, "c.castdpTodp_p", "length"], [42, 1, 1, "c.castdpTodp_p", "output"]], "castdpTodp_s": [[42, 1, 1, "c.castdpTodp_s", "core_mask"], [42, 1, 1, "c.castdpTodp_s", "input"], [42, 1, 1, "c.castdpTodp_s", "length"], [42, 1, 1, "c.castdpTodp_s", "output"]], "castdpTofp_p": [[42, 1, 1, "c.castdpTofp_p", "input"], [42, 1, 1, "c.castdpTofp_p", "length"], [42, 1, 1, "c.castdpTofp_p", "output"]], "castdpTofp_s": [[42, 1, 1, "c.castdpTofp_s", "core_mask"], [42, 1, 1, "c.castdpTofp_s", "input"], [42, 1, 1, "c.castdpTofp_s", "length"], [42, 1, 1, "c.castdpTofp_s", "output"]], "castfpTofp_p": [[42, 1, 1, "c.castfpTofp_p", "input"], [42, 1, 1, "c.castfpTofp_p", "length"], [42, 1, 1, "c.castfpTofp_p", "output"]], "castfpTofp_s": [[42, 1, 1, "c.castfpTofp_s", "core_mask"], [42, 1, 1, "c.castfpTofp_s", "input"], [42, 1, 1, "c.castfpTofp_s", "length"], [42, 1, 1, "c.castfpTofp_s", "output"]], "castfpToint16_p": [[42, 1, 1, "c.castfpToint16_p", "input"], [42, 1, 1, "c.castfpToint16_p", "length"], [42, 1, 1, "c.castfpToint16_p", "output"]], "castfpToint16_s": [[42, 1, 1, "c.castfpToint16_s", "core_mask"], [42, 1, 1, "c.castfpToint16_s", "input"], [42, 1, 1, "c.castfpToint16_s", "length"], [42, 1, 1, "c.castfpToint16_s", "output"]], "castfpToint32_p": [[42, 1, 1, "c.castfpToint32_p", "input"], [42, 1, 1, "c.castfpToint32_p", "length"], [42, 1, 1, "c.castfpToint32_p", "output"]], "castfpToint32_s": [[42, 1, 1, "c.castfpToint32_s", "core_mask"], [42, 1, 1, "c.castfpToint32_s", "input"], [42, 1, 1, "c.castfpToint32_s", "length"], [42, 1, 1, "c.castfpToint32_s", "output"]], "castfpToint8_p": [[42, 1, 1, "c.castfpToint8_p", "input"], [42, 1, 1, "c.castfpToint8_p", "length"], [42, 1, 1, "c.castfpToint8_p", "output"]], "castfpToint8_s": [[42, 1, 1, "c.castfpToint8_s", "core_mask"], [42, 1, 1, "c.castfpToint8_s", "input"], [42, 1, 1, "c.castfpToint8_s", "length"], [42, 1, 1, "c.castfpToint8_s", "output"]], "casti16Tofp_p": [[42, 1, 1, "c.casti16Tofp_p", "input"], [42, 1, 1, "c.casti16Tofp_p", "length"], [42, 1, 1, "c.casti16Tofp_p", "output"]], "casti16Tofp_s": [[42, 1, 1, "c.casti16Tofp_s", "core_mask"], [42, 1, 1, "c.casti16Tofp_s", "input"], [42, 1, 1, "c.casti16Tofp_s", "length"], [42, 1, 1, "c.casti16Tofp_s", "output"]], "casti16Tointi16_p": [[42, 1, 1, "c.casti16Tointi16_p", "input"], [42, 1, 1, "c.casti16Tointi16_p", "length"], [42, 1, 1, "c.casti16Tointi16_p", "output"]], "casti16Tointi16_s": [[42, 1, 1, "c.casti16Tointi16_s", "core_mask"], [42, 1, 1, "c.casti16Tointi16_s", "input"], [42, 1, 1, "c.casti16Tointi16_s", "length"], [42, 1, 1, "c.casti16Tointi16_s", "output"]], "casti16Tointi32_p": [[42, 1, 1, "c.casti16Tointi32_p", "input"], [42, 1, 1, "c.casti16Tointi32_p", "length"], [42, 1, 1, "c.casti16Tointi32_p", "output"]], "casti16Tointi32_s": [[42, 1, 1, "c.casti16Tointi32_s", "core_mask"], [42, 1, 1, "c.casti16Tointi32_s", "input"], [42, 1, 1, "c.casti16Tointi32_s", "length"], [42, 1, 1, "c.casti16Tointi32_s", "output"]], "casti16Tointi8_p": [[42, 1, 1, "c.casti16Tointi8_p", "input"], [42, 1, 1, "c.casti16Tointi8_p", "length"], [42, 1, 1, "c.casti16Tointi8_p", "output"]], "casti16Tointi8_s": [[42, 1, 1, "c.casti16Tointi8_s", "core_mask"], [42, 1, 1, "c.casti16Tointi8_s", "input"], [42, 1, 1, "c.casti16Tointi8_s", "length"], [42, 1, 1, "c.casti16Tointi8_s", "output"]], "casti32Tofp_p": [[42, 1, 1, "c.casti32Tofp_p", "input"], [42, 1, 1, "c.casti32Tofp_p", "length"], [42, 1, 1, "c.casti32Tofp_p", "output"]], "casti32Tofp_s": [[42, 1, 1, "c.casti32Tofp_s", "core_mask"], [42, 1, 1, "c.casti32Tofp_s", "input"], [42, 1, 1, "c.casti32Tofp_s", "length"], [42, 1, 1, "c.casti32Tofp_s", "output"]], "casti32Tointi16_p": [[42, 1, 1, "c.casti32Tointi16_p", "input"], [42, 1, 1, "c.casti32Tointi16_p", "length"], [42, 1, 1, "c.casti32Tointi16_p", "output"]], "casti32Tointi16_s": [[42, 1, 1, "c.casti32Tointi16_s", "core_mask"], [42, 1, 1, "c.casti32Tointi16_s", "input"], [42, 1, 1, "c.casti32Tointi16_s", "length"], [42, 1, 1, "c.casti32Tointi16_s", "output"]], "casti32Tointi32_p": [[42, 1, 1, "c.casti32Tointi32_p", "input"], [42, 1, 1, "c.casti32Tointi32_p", "length"], [42, 1, 1, "c.casti32Tointi32_p", "output"]], "casti32Tointi32_s": [[42, 1, 1, "c.casti32Tointi32_s", "core_mask"], [42, 1, 1, "c.casti32Tointi32_s", "input"], [42, 1, 1, "c.casti32Tointi32_s", "length"], [42, 1, 1, "c.casti32Tointi32_s", "output"]], "casti32Tointi8_p": [[42, 1, 1, "c.casti32Tointi8_p", "input"], [42, 1, 1, "c.casti32Tointi8_p", "length"], [42, 1, 1, "c.casti32Tointi8_p", "output"]], "casti32Tointi8_s": [[42, 1, 1, "c.casti32Tointi8_s", "core_mask"], [42, 1, 1, "c.casti32Tointi8_s", "input"], [42, 1, 1, "c.casti32Tointi8_s", "length"], [42, 1, 1, "c.casti32Tointi8_s", "output"]], "casti8Tofp_p": [[42, 1, 1, "c.casti8Tofp_p", "input"], [42, 1, 1, "c.casti8Tofp_p", "length"], [42, 1, 1, "c.casti8Tofp_p", "output"]], "casti8Tofp_s": [[42, 1, 1, "c.casti8Tofp_s", "core_mask"], [42, 1, 1, "c.casti8Tofp_s", "input"], [42, 1, 1, "c.casti8Tofp_s", "length"], [42, 1, 1, "c.casti8Tofp_s", "output"]], "casti8Tointi16_p": [[42, 1, 1, "c.casti8Tointi16_p", "input"], [42, 1, 1, "c.casti8Tointi16_p", "length"], [42, 1, 1, "c.casti8Tointi16_p", "output"]], "casti8Tointi16_s": [[42, 1, 1, "c.casti8Tointi16_s", "core_mask"], [42, 1, 1, "c.casti8Tointi16_s", "input"], [42, 1, 1, "c.casti8Tointi16_s", "length"], [42, 1, 1, "c.casti8Tointi16_s", "output"]], "casti8Tointi32_p": [[42, 1, 1, "c.casti8Tointi32_p", "input"], [42, 1, 1, "c.casti8Tointi32_p", "length"], [42, 1, 1, "c.casti8Tointi32_p", "output"]], "casti8Tointi32_s": [[42, 1, 1, "c.casti8Tointi32_s", "core_mask"], [42, 1, 1, "c.casti8Tointi32_s", "input"], [42, 1, 1, "c.casti8Tointi32_s", "length"], [42, 1, 1, "c.casti8Tointi32_s", "output"]], "casti8Tointi8_p": [[42, 1, 1, "c.casti8Tointi8_p", "input"], [42, 1, 1, "c.casti8Tointi8_p", "length"], [42, 1, 1, "c.casti8Tointi8_p", "output"]], "casti8Tointi8_s": [[42, 1, 1, "c.casti8Tointi8_s", "core_mask"], [42, 1, 1, "c.casti8Tointi8_s", "input"], [42, 1, 1, "c.casti8Tointi8_s", "length"], [42, 1, 1, "c.casti8Tointi8_s", "output"]], "customnormalize_p": [[56, 1, 1, "c.customnormalize_p", "params"], [56, 1, 1, "c.customnormalize_p", "result"], [56, 1, 1, "c.customnormalize_p", "result_len"], [56, 1, 1, "c.customnormalize_p", "str"], [56, 1, 1, "c.customnormalize_p", "str_len"], [56, 1, 1, "c.customnormalize_p", "tmp_str"]], "customnormalize_s": [[56, 1, 1, "c.customnormalize_s", "core_mask"], [56, 1, 1, "c.customnormalize_s", "params"], [56, 1, 1, "c.customnormalize_s", "result"], [56, 1, 1, "c.customnormalize_s", "result_len"], [56, 1, 1, "c.customnormalize_s", "str"], [56, 1, 1, "c.customnormalize_s", "str_len"], [56, 1, 1, "c.customnormalize_s", "tmp_str"]], "custompredict_p": [[57, 1, 1, "c.custompredict_p", "input"], [57, 1, 1, "c.custompredict_p", "input_size"], [57, 1, 1, "c.custompredict_p", "output_label"], [57, 1, 1, "c.custompredict_p", "output_num"], [57, 1, 1, "c.custompredict_p", "output_weight"], [57, 1, 1, "c.custompredict_p", "weight_threshold"]], "custompredict_s": [[57, 1, 1, "c.custompredict_s", "core_mask"], [57, 1, 1, "c.custompredict_s", "input"], [57, 1, 1, "c.custompredict_s", "input_size"], [57, 1, 1, "c.custompredict_s", "output_label"], [57, 1, 1, "c.custompredict_s", "output_num"], [57, 1, 1, "c.custompredict_s", "output_weight"], [57, 1, 1, "c.custompredict_s", "weight_threshold"]], "dp_Unique_p": [[221, 1, 1, "c.dp_Unique_p", "input"], [221, 1, 1, "c.dp_Unique_p", "input_len"], [221, 1, 1, "c.dp_Unique_p", "output0"], [221, 1, 1, "c.dp_Unique_p", "output0_len"]], "dp_Unique_s": [[221, 1, 1, "c.dp_Unique_s", "core_mask"], [221, 1, 1, "c.dp_Unique_s", "input"], [221, 1, 1, "c.dp_Unique_s", "input_len"], [221, 1, 1, "c.dp_Unique_s", "output0"], [221, 1, 1, "c.dp_Unique_s", "output0_len"]], "dp_abs_p": [[10, 1, 1, "c.dp_abs_p", "dst_data"], [10, 1, 1, "c.dp_abs_p", "length"], [10, 1, 1, "c.dp_abs_p", "src_data"]], "dp_abs_s": [[10, 1, 1, "c.dp_abs_s", "core_mask"], [10, 1, 1, "c.dp_abs_s", "dst_data"], [10, 1, 1, "c.dp_abs_s", "length"], [10, 1, 1, "c.dp_abs_s", "src_data"]], "dp_addext_p": [[17, 1, 1, "c.dp_addext_p", "alpha"], [17, 1, 1, "c.dp_addext_p", "in0"], [17, 1, 1, "c.dp_addext_p", "in1"], [17, 1, 1, "c.dp_addext_p", "out"], [17, 1, 1, "c.dp_addext_p", "size"]], "dp_addext_s": [[17, 1, 1, "c.dp_addext_s", "alpha"], [17, 1, 1, "c.dp_addext_s", "core_mask"], [17, 1, 1, "c.dp_addext_s", "in0"], [17, 1, 1, "c.dp_addext_s", "in1"], [17, 1, 1, "c.dp_addext_s", "out"], [17, 1, 1, "c.dp_addext_s", "size"]], "dp_addn_p": [[19, 1, 1, "c.dp_addn_p", "input0"], [19, 1, 1, "c.dp_addn_p", "input1"], [19, 1, 1, "c.dp_addn_p", "length"], [19, 1, 1, "c.dp_addn_p", "output"]], "dp_addn_s": [[19, 1, 1, "c.dp_addn_s", "core_mask"], [19, 1, 1, "c.dp_addn_s", "input0"], [19, 1, 1, "c.dp_addn_s", "input1"], [19, 1, 1, "c.dp_addn_s", "length"], [19, 1, 1, "c.dp_addn_s", "output"]], "dp_addrelu6_p": [[17, 1, 1, "c.dp_addrelu6_p", "in0"], [17, 1, 1, "c.dp_addrelu6_p", "in1"], [17, 1, 1, "c.dp_addrelu6_p", "out"], [17, 1, 1, "c.dp_addrelu6_p", "size"]], "dp_addrelu6_s": [[17, 1, 1, "c.dp_addrelu6_s", "core_mask"], [17, 1, 1, "c.dp_addrelu6_s", "in0"], [17, 1, 1, "c.dp_addrelu6_s", "in1"], [17, 1, 1, "c.dp_addrelu6_s", "out"], [17, 1, 1, "c.dp_addrelu6_s", "size"]], "dp_addrelu_p": [[17, 1, 1, "c.dp_addrelu_p", "in0"], [17, 1, 1, "c.dp_addrelu_p", "in1"], [17, 1, 1, "c.dp_addrelu_p", "out"], [17, 1, 1, "c.dp_addrelu_p", "size"]], "dp_addrelu_s": [[17, 1, 1, "c.dp_addrelu_s", "core_mask"], [17, 1, 1, "c.dp_addrelu_s", "in0"], [17, 1, 1, "c.dp_addrelu_s", "in1"], [17, 1, 1, "c.dp_addrelu_s", "out"], [17, 1, 1, "c.dp_addrelu_s", "size"]], "dp_allgather_p": [[22, 1, 1, "c.dp_allgather_p", "data_size"], [22, 1, 1, "c.dp_allgather_p", "input"], [22, 1, 1, "c.dp_allgather_p", "input_rank"], [22, 1, 1, "c.dp_allgather_p", "output"], [22, 1, 1, "c.dp_allgather_p", "output_rank"]], "dp_allgather_s": [[22, 1, 1, "c.dp_allgather_s", "core_mask"], [22, 1, 1, "c.dp_allgather_s", "data_size"], [22, 1, 1, "c.dp_allgather_s", "input"], [22, 1, 1, "c.dp_allgather_s", "input_rank"], [22, 1, 1, "c.dp_allgather_s", "output"], [22, 1, 1, "c.dp_allgather_s", "output_rank"]], "dp_and_p": [[112, 1, 1, "c.dp_and_p", "input0"], [112, 1, 1, "c.dp_and_p", "input1"], [112, 1, 1, "c.dp_and_p", "length"], [112, 1, 1, "c.dp_and_p", "output"]], "dp_and_s": [[112, 1, 1, "c.dp_and_s", "core_mask"], [112, 1, 1, "c.dp_and_s", "input0"], [112, 1, 1, "c.dp_and_s", "input1"], [112, 1, 1, "c.dp_and_s", "length"], [112, 1, 1, "c.dp_and_s", "output"]], "dp_assign_p": [[27, 1, 1, "c.dp_assign_p", "dst"], [27, 1, 1, "c.dp_assign_p", "length"], [27, 1, 1, "c.dp_assign_p", "src"]], "dp_assign_s": [[27, 1, 1, "c.dp_assign_s", "core_mask"], [27, 1, 1, "c.dp_assign_s", "dst"], [27, 1, 1, "c.dp_assign_s", "length"], [27, 1, 1, "c.dp_assign_s", "src"]], "dp_assignadd_p": [[28, 1, 1, "c.dp_assignadd_p", "input"], [28, 1, 1, "c.dp_assignadd_p", "length"], [28, 1, 1, "c.dp_assignadd_p", "output"]], "dp_assignadd_s": [[28, 1, 1, "c.dp_assignadd_s", "core_mask"], [28, 1, 1, "c.dp_assignadd_s", "input"], [28, 1, 1, "c.dp_assignadd_s", "length"], [28, 1, 1, "c.dp_assignadd_s", "output"]], "dp_batchtospace_p": [[35, 1, 1, "c.dp_batchtospace_p", "block_size"], [35, 1, 1, "c.dp_batchtospace_p", "crops"], [35, 1, 1, "c.dp_batchtospace_p", "data_size"], [35, 1, 1, "c.dp_batchtospace_p", "input"], [35, 1, 1, "c.dp_batchtospace_p", "input_shape"], [35, 1, 1, "c.dp_batchtospace_p", "output"]], "dp_batchtospace_s": [[35, 1, 1, "c.dp_batchtospace_s", "block_size"], [35, 1, 1, "c.dp_batchtospace_s", "core_mask"], [35, 1, 1, "c.dp_batchtospace_s", "crops"], [35, 1, 1, "c.dp_batchtospace_s", "data_size"], [35, 1, 1, "c.dp_batchtospace_s", "input"], [35, 1, 1, "c.dp_batchtospace_s", "input_shape"], [35, 1, 1, "c.dp_batchtospace_s", "output"]], "dp_batchtospacend_p": [[36, 1, 1, "c.dp_batchtospacend_p", "block_size"], [36, 1, 1, "c.dp_batchtospacend_p", "crops"], [36, 1, 1, "c.dp_batchtospacend_p", "data_size"], [36, 1, 1, "c.dp_batchtospacend_p", "input"], [36, 1, 1, "c.dp_batchtospacend_p", "input_shape"], [36, 1, 1, "c.dp_batchtospacend_p", "output"]], "dp_batchtospacend_s": [[36, 1, 1, "c.dp_batchtospacend_s", "block_size"], [36, 1, 1, "c.dp_batchtospacend_s", "core_mask"], [36, 1, 1, "c.dp_batchtospacend_s", "crops"], [36, 1, 1, "c.dp_batchtospacend_s", "data_size"], [36, 1, 1, "c.dp_batchtospacend_s", "input"], [36, 1, 1, "c.dp_batchtospacend_s", "input_shape"], [36, 1, 1, "c.dp_batchtospacend_s", "output"]], "dp_biasadd_p": [[37, 1, 1, "c.dp_biasadd_p", "data_format"], [37, 1, 1, "c.dp_biasadd_p", "dims"], [37, 1, 1, "c.dp_biasadd_p", "input_bias"], [37, 1, 1, "c.dp_biasadd_p", "input_x"], [37, 1, 1, "c.dp_biasadd_p", "length"], [37, 1, 1, "c.dp_biasadd_p", "output"], [37, 1, 1, "c.dp_biasadd_p", "shape_size"]], "dp_biasadd_s": [[37, 1, 1, "c.dp_biasadd_s", "core_mask"], [37, 1, 1, "c.dp_biasadd_s", "data_format"], [37, 1, 1, "c.dp_biasadd_s", "dims"], [37, 1, 1, "c.dp_biasadd_s", "input_bias"], [37, 1, 1, "c.dp_biasadd_s", "input_x"], [37, 1, 1, "c.dp_biasadd_s", "length"], [37, 1, 1, "c.dp_biasadd_s", "output"], [37, 1, 1, "c.dp_biasadd_s", "shape_size"]], "dp_broadcastto_p": [[41, 1, 1, "c.dp_broadcastto_p", "data_size"], [41, 1, 1, "c.dp_broadcastto_p", "input"], [41, 1, 1, "c.dp_broadcastto_p", "input_shape"], [41, 1, 1, "c.dp_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.dp_broadcastto_p", "output"], [41, 1, 1, "c.dp_broadcastto_p", "output_shape"], [41, 1, 1, "c.dp_broadcastto_p", "output_shape_size"]], "dp_broadcastto_s": [[41, 1, 1, "c.dp_broadcastto_s", "core_mask"], [41, 1, 1, "c.dp_broadcastto_s", "data_size"], [41, 1, 1, "c.dp_broadcastto_s", "input"], [41, 1, 1, "c.dp_broadcastto_s", "input_shape"], [41, 1, 1, "c.dp_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.dp_broadcastto_s", "output"], [41, 1, 1, "c.dp_broadcastto_s", "output_shape"], [41, 1, 1, "c.dp_broadcastto_s", "output_shape_size"]], "dp_ceil_p": [[43, 1, 1, "c.dp_ceil_p", "input_size"], [43, 1, 1, "c.dp_ceil_p", "input_x"], [43, 1, 1, "c.dp_ceil_p", "output"]], "dp_ceil_s": [[43, 1, 1, "c.dp_ceil_s", "core_mask"], [43, 1, 1, "c.dp_ceil_s", "input_size"], [43, 1, 1, "c.dp_ceil_s", "input_x"], [43, 1, 1, "c.dp_ceil_s", "output"]], "dp_clip_p": [[44, 1, 1, "c.dp_clip_p", "dst"], [44, 1, 1, "c.dp_clip_p", "length"], [44, 1, 1, "c.dp_clip_p", "max"], [44, 1, 1, "c.dp_clip_p", "min"], [44, 1, 1, "c.dp_clip_p", "src"]], "dp_clip_s": [[44, 1, 1, "c.dp_clip_s", "core_mask"], [44, 1, 1, "c.dp_clip_s", "dst"], [44, 1, 1, "c.dp_clip_s", "length"], [44, 1, 1, "c.dp_clip_s", "max"], [44, 1, 1, "c.dp_clip_s", "min"], [44, 1, 1, "c.dp_clip_s", "src"]], "dp_concat_p": [[45, 1, 1, "c.dp_concat_p", "axis"], [45, 1, 1, "c.dp_concat_p", "input_ndim"], [45, 1, 1, "c.dp_concat_p", "input_shapes"], [45, 1, 1, "c.dp_concat_p", "inputs"], [45, 1, 1, "c.dp_concat_p", "num_inputs"], [45, 1, 1, "c.dp_concat_p", "output"]], "dp_concat_s": [[45, 1, 1, "c.dp_concat_s", "axis"], [45, 1, 1, "c.dp_concat_s", "core_mask"], [45, 1, 1, "c.dp_concat_s", "input_ndim"], [45, 1, 1, "c.dp_concat_s", "input_shapes"], [45, 1, 1, "c.dp_concat_s", "inputs"], [45, 1, 1, "c.dp_concat_s", "num_inputs"], [45, 1, 1, "c.dp_concat_s", "output"]], "dp_constant_of_shape_p": [[46, 1, 1, "c.dp_constant_of_shape_p", "end"], [46, 1, 1, "c.dp_constant_of_shape_p", "output"], [46, 1, 1, "c.dp_constant_of_shape_p", "start"], [46, 1, 1, "c.dp_constant_of_shape_p", "value"]], "dp_constant_of_shape_s": [[46, 1, 1, "c.dp_constant_of_shape_s", "core_mask"], [46, 1, 1, "c.dp_constant_of_shape_s", "end"], [46, 1, 1, "c.dp_constant_of_shape_s", "output"], [46, 1, 1, "c.dp_constant_of_shape_s", "start"], [46, 1, 1, "c.dp_constant_of_shape_s", "value"]], "dp_cos_p": [[51, 1, 1, "c.dp_cos_p", "dst_data"], [51, 1, 1, "c.dp_cos_p", "length"], [51, 1, 1, "c.dp_cos_p", "src_data"]], "dp_cos_s": [[51, 1, 1, "c.dp_cos_s", "core_mask"], [51, 1, 1, "c.dp_cos_s", "dst_data"], [51, 1, 1, "c.dp_cos_s", "length"], [51, 1, 1, "c.dp_cos_s", "src_data"]], "dp_cumsum_p": [[54, 1, 1, "c.dp_cumsum_p", "axis_dim"], [54, 1, 1, "c.dp_cumsum_p", "exclusive"], [54, 1, 1, "c.dp_cumsum_p", "inner_dim"], [54, 1, 1, "c.dp_cumsum_p", "input"], [54, 1, 1, "c.dp_cumsum_p", "out_dim"], [54, 1, 1, "c.dp_cumsum_p", "output"]], "dp_cumsum_s": [[54, 1, 1, "c.dp_cumsum_s", "axis_dim"], [54, 1, 1, "c.dp_cumsum_s", "core_mask"], [54, 1, 1, "c.dp_cumsum_s", "exclusive"], [54, 1, 1, "c.dp_cumsum_s", "inner_dim"], [54, 1, 1, "c.dp_cumsum_s", "input"], [54, 1, 1, "c.dp_cumsum_s", "out_dim"], [54, 1, 1, "c.dp_cumsum_s", "output"]], "dp_depthtospace_p": [[59, 1, 1, "c.dp_depthtospace_p", "block_size"], [59, 1, 1, "c.dp_depthtospace_p", "data_size"], [59, 1, 1, "c.dp_depthtospace_p", "in_shape"], [59, 1, 1, "c.dp_depthtospace_p", "input"], [59, 1, 1, "c.dp_depthtospace_p", "output"]], "dp_depthtospace_s": [[59, 1, 1, "c.dp_depthtospace_s", "block_size"], [59, 1, 1, "c.dp_depthtospace_s", "core_mask"], [59, 1, 1, "c.dp_depthtospace_s", "data_size"], [59, 1, 1, "c.dp_depthtospace_s", "in_shape"], [59, 1, 1, "c.dp_depthtospace_s", "input"], [59, 1, 1, "c.dp_depthtospace_s", "output"]], "dp_div_fusion_p": [[61, 1, 1, "c.dp_div_fusion_p", "input0"], [61, 1, 1, "c.dp_div_fusion_p", "input1"], [61, 1, 1, "c.dp_div_fusion_p", "length"], [61, 1, 1, "c.dp_div_fusion_p", "output"]], "dp_div_fusion_s": [[61, 1, 1, "c.dp_div_fusion_s", "core_mask"], [61, 1, 1, "c.dp_div_fusion_s", "input0"], [61, 1, 1, "c.dp_div_fusion_s", "input1"], [61, 1, 1, "c.dp_div_fusion_s", "length"], [61, 1, 1, "c.dp_div_fusion_s", "output"]], "dp_eltwise_p": [[67, 1, 1, "c.dp_eltwise_p", "Input0"], [67, 1, 1, "c.dp_eltwise_p", "Input1"], [67, 1, 1, "c.dp_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.dp_eltwise_p", "length"], [67, 1, 1, "c.dp_eltwise_p", "output"]], "dp_eltwise_s": [[67, 1, 1, "c.dp_eltwise_s", "Input0"], [67, 1, 1, "c.dp_eltwise_s", "Input1"], [67, 1, 1, "c.dp_eltwise_s", "core_mask"], [67, 1, 1, "c.dp_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.dp_eltwise_s", "length"], [67, 1, 1, "c.dp_eltwise_s", "output"]], "dp_equal_p": [[70, 1, 1, "c.dp_equal_p", "Input0"], [70, 1, 1, "c.dp_equal_p", "Input1"], [70, 1, 1, "c.dp_equal_p", "length"], [70, 1, 1, "c.dp_equal_p", "output"]], "dp_equal_s": [[70, 1, 1, "c.dp_equal_s", "Input0"], [70, 1, 1, "c.dp_equal_s", "Input1"], [70, 1, 1, "c.dp_equal_s", "core_mask"], [70, 1, 1, "c.dp_equal_s", "length"], [70, 1, 1, "c.dp_equal_s", "output"]], "dp_erf_p": [[71, 1, 1, "c.dp_erf_p", "input"], [71, 1, 1, "c.dp_erf_p", "length"], [71, 1, 1, "c.dp_erf_p", "output"]], "dp_erf_s": [[71, 1, 1, "c.dp_erf_s", "core_mask"], [71, 1, 1, "c.dp_erf_s", "input"], [71, 1, 1, "c.dp_erf_s", "length"], [71, 1, 1, "c.dp_erf_s", "output"]], "dp_expfusion_p": [[73, 1, 1, "c.dp_expfusion_p", "dst_data"], [73, 1, 1, "c.dp_expfusion_p", "in_scale"], [73, 1, 1, "c.dp_expfusion_p", "length"], [73, 1, 1, "c.dp_expfusion_p", "out_scale"], [73, 1, 1, "c.dp_expfusion_p", "scale"], [73, 1, 1, "c.dp_expfusion_p", "src_data"]], "dp_expfusion_s": [[73, 1, 1, "c.dp_expfusion_s", "core_mask"], [73, 1, 1, "c.dp_expfusion_s", "dst_data"], [73, 1, 1, "c.dp_expfusion_s", "in_scale"], [73, 1, 1, "c.dp_expfusion_s", "length"], [73, 1, 1, "c.dp_expfusion_s", "out_scale"], [73, 1, 1, "c.dp_expfusion_s", "scale"], [73, 1, 1, "c.dp_expfusion_s", "src_data"]], "dp_extract_features_p": [[55, 1, 1, "c.dp_extract_features_p", "num_strings"], [55, 1, 1, "c.dp_extract_features_p", "output_labels"], [55, 1, 1, "c.dp_extract_features_p", "output_weights"], [55, 1, 1, "c.dp_extract_features_p", "string_lengths"], [55, 1, 1, "c.dp_extract_features_p", "string_pointers"]], "dp_extract_features_s": [[55, 1, 1, "c.dp_extract_features_s", "core_mask"], [55, 1, 1, "c.dp_extract_features_s", "num_strings"], [55, 1, 1, "c.dp_extract_features_s", "output_labels"], [55, 1, 1, "c.dp_extract_features_s", "output_weights"], [55, 1, 1, "c.dp_extract_features_s", "string_lengths"], [55, 1, 1, "c.dp_extract_features_s", "string_pointers"]], "dp_fill_p": [[78, 1, 1, "c.dp_fill_p", "output"], [78, 1, 1, "c.dp_fill_p", "param"], [78, 1, 1, "c.dp_fill_p", "value"]], "dp_fill_s": [[78, 1, 1, "c.dp_fill_s", "core_mask"], [78, 1, 1, "c.dp_fill_s", "output"], [78, 1, 1, "c.dp_fill_s", "param"], [78, 1, 1, "c.dp_fill_s", "value"]], "dp_floor_p": [[82, 1, 1, "c.dp_floor_p", "dst_data"], [82, 1, 1, "c.dp_floor_p", "length"], [82, 1, 1, "c.dp_floor_p", "src_data"]], "dp_floor_s": [[82, 1, 1, "c.dp_floor_s", "core_mask"], [82, 1, 1, "c.dp_floor_s", "dst_data"], [82, 1, 1, "c.dp_floor_s", "length"], [82, 1, 1, "c.dp_floor_s", "src_data"]], "dp_floordiv_p": [[83, 1, 1, "c.dp_floordiv_p", "dst_data"], [83, 1, 1, "c.dp_floordiv_p", "length"], [83, 1, 1, "c.dp_floordiv_p", "src_data0"], [83, 1, 1, "c.dp_floordiv_p", "src_data1"]], "dp_floordiv_s": [[83, 1, 1, "c.dp_floordiv_s", "core_mask"], [83, 1, 1, "c.dp_floordiv_s", "dst_data"], [83, 1, 1, "c.dp_floordiv_s", "length"], [83, 1, 1, "c.dp_floordiv_s", "src_data0"], [83, 1, 1, "c.dp_floordiv_s", "src_data1"]], "dp_floormod_p": [[84, 1, 1, "c.dp_floormod_p", "input0"], [84, 1, 1, "c.dp_floormod_p", "input1"], [84, 1, 1, "c.dp_floormod_p", "output"], [84, 1, 1, "c.dp_floormod_p", "size"]], "dp_floormod_s": [[84, 1, 1, "c.dp_floormod_s", "core_mask"], [84, 1, 1, "c.dp_floormod_s", "input0"], [84, 1, 1, "c.dp_floormod_s", "input1"], [84, 1, 1, "c.dp_floormod_s", "output"], [84, 1, 1, "c.dp_floormod_s", "size"]], "dp_formattranspose_p": [[85, 1, 1, "c.dp_formattranspose_p", "batch"], [85, 1, 1, "c.dp_formattranspose_p", "channel"], [85, 1, 1, "c.dp_formattranspose_p", "dst_data"], [85, 1, 1, "c.dp_formattranspose_p", "dst_format"], [85, 1, 1, "c.dp_formattranspose_p", "plane"], [85, 1, 1, "c.dp_formattranspose_p", "src_data"], [85, 1, 1, "c.dp_formattranspose_p", "src_format"]], "dp_formattranspose_s": [[85, 1, 1, "c.dp_formattranspose_s", "batch"], [85, 1, 1, "c.dp_formattranspose_s", "channel"], [85, 1, 1, "c.dp_formattranspose_s", "core_mask"], [85, 1, 1, "c.dp_formattranspose_s", "dst_data"], [85, 1, 1, "c.dp_formattranspose_s", "dst_format"], [85, 1, 1, "c.dp_formattranspose_s", "plane"], [85, 1, 1, "c.dp_formattranspose_s", "src_data"], [85, 1, 1, "c.dp_formattranspose_s", "src_format"]], "dp_gather_nd_p": [[89, 1, 1, "c.dp_gather_nd_p", "indices"], [89, 1, 1, "c.dp_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.dp_gather_nd_p", "indices_shape"], [89, 1, 1, "c.dp_gather_nd_p", "input"], [89, 1, 1, "c.dp_gather_nd_p", "input_ndim"], [89, 1, 1, "c.dp_gather_nd_p", "input_shape"], [89, 1, 1, "c.dp_gather_nd_p", "output"]], "dp_gather_nd_s": [[89, 1, 1, "c.dp_gather_nd_s", "core_mask"], [89, 1, 1, "c.dp_gather_nd_s", "indices"], [89, 1, 1, "c.dp_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.dp_gather_nd_s", "indices_shape"], [89, 1, 1, "c.dp_gather_nd_s", "input"], [89, 1, 1, "c.dp_gather_nd_s", "input_ndim"], [89, 1, 1, "c.dp_gather_nd_s", "input_shape"], [89, 1, 1, "c.dp_gather_nd_s", "output"]], "dp_gather_p": [[88, 1, 1, "c.dp_gather_p", "axis"], [88, 1, 1, "c.dp_gather_p", "batch_dims"], [88, 1, 1, "c.dp_gather_p", "indices"], [88, 1, 1, "c.dp_gather_p", "indices_ndim"], [88, 1, 1, "c.dp_gather_p", "indices_shape"], [88, 1, 1, "c.dp_gather_p", "input"], [88, 1, 1, "c.dp_gather_p", "input_ndim"], [88, 1, 1, "c.dp_gather_p", "input_shape"], [88, 1, 1, "c.dp_gather_p", "output"]], "dp_gather_s": [[88, 1, 1, "c.dp_gather_s", "axis"], [88, 1, 1, "c.dp_gather_s", "batch_dims"], [88, 1, 1, "c.dp_gather_s", "core_mask"], [88, 1, 1, "c.dp_gather_s", "indices"], [88, 1, 1, "c.dp_gather_s", "indices_ndim"], [88, 1, 1, "c.dp_gather_s", "indices_shape"], [88, 1, 1, "c.dp_gather_s", "input"], [88, 1, 1, "c.dp_gather_s", "input_ndim"], [88, 1, 1, "c.dp_gather_s", "input_shape"], [88, 1, 1, "c.dp_gather_s", "output"]], "dp_gatherd_p": [[90, 1, 1, "c.dp_gatherd_p", "dim"], [90, 1, 1, "c.dp_gatherd_p", "index"], [90, 1, 1, "c.dp_gatherd_p", "index_shape"], [90, 1, 1, "c.dp_gatherd_p", "input_shape"], [90, 1, 1, "c.dp_gatherd_p", "input_shape_size"], [90, 1, 1, "c.dp_gatherd_p", "input_x"], [90, 1, 1, "c.dp_gatherd_p", "output"]], "dp_gatherd_s": [[90, 1, 1, "c.dp_gatherd_s", "core_mask"], [90, 1, 1, "c.dp_gatherd_s", "dim"], [90, 1, 1, "c.dp_gatherd_s", "index"], [90, 1, 1, "c.dp_gatherd_s", "index_shape"], [90, 1, 1, "c.dp_gatherd_s", "input_shape"], [90, 1, 1, "c.dp_gatherd_s", "input_shape_size"], [90, 1, 1, "c.dp_gatherd_s", "input_x"], [90, 1, 1, "c.dp_gatherd_s", "output"]], "dp_greater_s": [[92, 1, 1, "c.dp_greater_s", "core_mask"], [92, 1, 1, "c.dp_greater_s", "element_num"], [92, 1, 1, "c.dp_greater_s", "in_elements_num0"], [92, 1, 1, "c.dp_greater_s", "input1"], [92, 1, 1, "c.dp_greater_s", "input2"], [92, 1, 1, "c.dp_greater_s", "optimize"], [92, 1, 1, "c.dp_greater_s", "output"]], "dp_greaterequal_s": [[93, 1, 1, "c.dp_greaterequal_s", "core_mask"], [93, 1, 1, "c.dp_greaterequal_s", "element_num"], [93, 1, 1, "c.dp_greaterequal_s", "in_elements_num0"], [93, 1, 1, "c.dp_greaterequal_s", "input1"], [93, 1, 1, "c.dp_greaterequal_s", "input2"], [93, 1, 1, "c.dp_greaterequal_s", "optimize"], [93, 1, 1, "c.dp_greaterequal_s", "output"]], "dp_isfinite_p": [[99, 1, 1, "c.dp_isfinite_p", "Input"], [99, 1, 1, "c.dp_isfinite_p", "length"], [99, 1, 1, "c.dp_isfinite_p", "output"]], "dp_isfinite_s": [[99, 1, 1, "c.dp_isfinite_s", "Input"], [99, 1, 1, "c.dp_isfinite_s", "core_mask"], [99, 1, 1, "c.dp_isfinite_s", "length"], [99, 1, 1, "c.dp_isfinite_s", "output"]], "dp_less_p": [[104, 1, 1, "c.dp_less_p", "Input0"], [104, 1, 1, "c.dp_less_p", "Input1"], [104, 1, 1, "c.dp_less_p", "in_elements_num0"], [104, 1, 1, "c.dp_less_p", "length"], [104, 1, 1, "c.dp_less_p", "optimize"], [104, 1, 1, "c.dp_less_p", "output"]], "dp_less_s": [[104, 1, 1, "c.dp_less_s", "Input0"], [104, 1, 1, "c.dp_less_s", "Input1"], [104, 1, 1, "c.dp_less_s", "core_mask"], [104, 1, 1, "c.dp_less_s", "in_elements_num0"], [104, 1, 1, "c.dp_less_s", "length"], [104, 1, 1, "c.dp_less_s", "optimize"], [104, 1, 1, "c.dp_less_s", "output"]], "dp_lessequal_p": [[105, 1, 1, "c.dp_lessequal_p", "Input0"], [105, 1, 1, "c.dp_lessequal_p", "Input1"], [105, 1, 1, "c.dp_lessequal_p", "in_elements_num0"], [105, 1, 1, "c.dp_lessequal_p", "length"], [105, 1, 1, "c.dp_lessequal_p", "optimize"], [105, 1, 1, "c.dp_lessequal_p", "output"]], "dp_lessequal_s": [[105, 1, 1, "c.dp_lessequal_s", "Input0"], [105, 1, 1, "c.dp_lessequal_s", "Input1"], [105, 1, 1, "c.dp_lessequal_s", "core_mask"], [105, 1, 1, "c.dp_lessequal_s", "in_elements_num0"], [105, 1, 1, "c.dp_lessequal_s", "length"], [105, 1, 1, "c.dp_lessequal_s", "optimize"], [105, 1, 1, "c.dp_lessequal_s", "output"]], "dp_log1p_p": [[108, 1, 1, "c.dp_log1p_p", "Input"], [108, 1, 1, "c.dp_log1p_p", "length"], [108, 1, 1, "c.dp_log1p_p", "output"]], "dp_log1p_s": [[108, 1, 1, "c.dp_log1p_s", "Input"], [108, 1, 1, "c.dp_log1p_s", "core_mask"], [108, 1, 1, "c.dp_log1p_s", "length"], [108, 1, 1, "c.dp_log1p_s", "output"]], "dp_logical_not_p": [[110, 1, 1, "c.dp_logical_not_p", "input"], [110, 1, 1, "c.dp_logical_not_p", "length"], [110, 1, 1, "c.dp_logical_not_p", "output"]], "dp_logical_not_s": [[110, 1, 1, "c.dp_logical_not_s", "core_mask"], [110, 1, 1, "c.dp_logical_not_s", "input"], [110, 1, 1, "c.dp_logical_not_s", "length"], [110, 1, 1, "c.dp_logical_not_s", "output"]], "dp_logical_or_p": [[111, 1, 1, "c.dp_logical_or_p", "input0"], [111, 1, 1, "c.dp_logical_or_p", "input1"], [111, 1, 1, "c.dp_logical_or_p", "length"], [111, 1, 1, "c.dp_logical_or_p", "output"]], "dp_logical_or_s": [[111, 1, 1, "c.dp_logical_or_s", "core_mask"], [111, 1, 1, "c.dp_logical_or_s", "input0"], [111, 1, 1, "c.dp_logical_or_s", "input1"], [111, 1, 1, "c.dp_logical_or_s", "length"], [111, 1, 1, "c.dp_logical_or_s", "output"]], "dp_lsh_projection_p": [[116, 1, 1, "c.dp_lsh_projection_p", "bits_per_hash"], [116, 1, 1, "c.dp_lsh_projection_p", "feature"], [116, 1, 1, "c.dp_lsh_projection_p", "feature_num"], [116, 1, 1, "c.dp_lsh_projection_p", "hash_group_num"], [116, 1, 1, "c.dp_lsh_projection_p", "hash_seed"], [116, 1, 1, "c.dp_lsh_projection_p", "output"], [116, 1, 1, "c.dp_lsh_projection_p", "weight"]], "dp_lsh_projection_s": [[116, 1, 1, "c.dp_lsh_projection_s", "bits_per_hash"], [116, 1, 1, "c.dp_lsh_projection_s", "core_mask"], [116, 1, 1, "c.dp_lsh_projection_s", "feature"], [116, 1, 1, "c.dp_lsh_projection_s", "feature_num"], [116, 1, 1, "c.dp_lsh_projection_s", "hash_group_num"], [116, 1, 1, "c.dp_lsh_projection_s", "hash_seed"], [116, 1, 1, "c.dp_lsh_projection_s", "output"], [116, 1, 1, "c.dp_lsh_projection_s", "weight"]], "dp_matmulfusion_p": [[121, 1, 1, "c.dp_matmulfusion_p", "A"], [121, 1, 1, "c.dp_matmulfusion_p", "B"], [121, 1, 1, "c.dp_matmulfusion_p", "C"], [121, 1, 1, "c.dp_matmulfusion_p", "K"], [121, 1, 1, "c.dp_matmulfusion_p", "M"], [121, 1, 1, "c.dp_matmulfusion_p", "N"], [121, 1, 1, "c.dp_matmulfusion_p", "activation_type"], [121, 1, 1, "c.dp_matmulfusion_p", "bias"]], "dp_matmulfusion_s": [[121, 1, 1, "c.dp_matmulfusion_s", "A"], [121, 1, 1, "c.dp_matmulfusion_s", "B"], [121, 1, 1, "c.dp_matmulfusion_s", "C"], [121, 1, 1, "c.dp_matmulfusion_s", "K"], [121, 1, 1, "c.dp_matmulfusion_s", "M"], [121, 1, 1, "c.dp_matmulfusion_s", "N"], [121, 1, 1, "c.dp_matmulfusion_s", "activation_type"], [121, 1, 1, "c.dp_matmulfusion_s", "bias"], [121, 1, 1, "c.dp_matmulfusion_s", "core_mask"]], "dp_maximum_p": [[122, 1, 1, "c.dp_maximum_p", "input0"], [122, 1, 1, "c.dp_maximum_p", "input1"], [122, 1, 1, "c.dp_maximum_p", "length"], [122, 1, 1, "c.dp_maximum_p", "output"]], "dp_maximum_s": [[122, 1, 1, "c.dp_maximum_s", "core_mask"], [122, 1, 1, "c.dp_maximum_s", "input0"], [122, 1, 1, "c.dp_maximum_s", "input1"], [122, 1, 1, "c.dp_maximum_s", "length"], [122, 1, 1, "c.dp_maximum_s", "output"]], "dp_maxpool_fusion_p": [[124, 1, 1, "c.dp_maxpool_fusion_p", "params"]], "dp_maxpool_fusion_s": [[124, 1, 1, "c.dp_maxpool_fusion_s", "core_mask"], [124, 1, 1, "c.dp_maxpool_fusion_s", "params"]], "dp_maxpool_grad_p": [[125, 1, 1, "c.dp_maxpool_grad_p", "params"]], "dp_maxpool_grad_s": [[125, 1, 1, "c.dp_maxpool_grad_s", "core_mask"], [125, 1, 1, "c.dp_maxpool_grad_s", "params"]], "dp_minimum_p": [[127, 1, 1, "c.dp_minimum_p", "input0"], [127, 1, 1, "c.dp_minimum_p", "input1"], [127, 1, 1, "c.dp_minimum_p", "length"], [127, 1, 1, "c.dp_minimum_p", "output"]], "dp_minimum_s": [[127, 1, 1, "c.dp_minimum_s", "core_mask"], [127, 1, 1, "c.dp_minimum_s", "input0"], [127, 1, 1, "c.dp_minimum_s", "input1"], [127, 1, 1, "c.dp_minimum_s", "length"], [127, 1, 1, "c.dp_minimum_s", "output"]], "dp_mod_p": [[129, 1, 1, "c.dp_mod_p", "input0"], [129, 1, 1, "c.dp_mod_p", "input1"], [129, 1, 1, "c.dp_mod_p", "length"], [129, 1, 1, "c.dp_mod_p", "output"]], "dp_mod_s": [[129, 1, 1, "c.dp_mod_s", "core_mask"], [129, 1, 1, "c.dp_mod_s", "input0"], [129, 1, 1, "c.dp_mod_s", "input1"], [129, 1, 1, "c.dp_mod_s", "length"], [129, 1, 1, "c.dp_mod_s", "output"]], "dp_mul_p": [[130, 1, 1, "c.dp_mul_p", "input0"], [130, 1, 1, "c.dp_mul_p", "input1"], [130, 1, 1, "c.dp_mul_p", "length"], [130, 1, 1, "c.dp_mul_p", "output"]], "dp_mul_s": [[130, 1, 1, "c.dp_mul_s", "core_mask"], [130, 1, 1, "c.dp_mul_s", "input0"], [130, 1, 1, "c.dp_mul_s", "input1"], [130, 1, 1, "c.dp_mul_s", "length"], [130, 1, 1, "c.dp_mul_s", "output"]], "dp_neg_grad_p": [[133, 1, 1, "c.dp_neg_grad_p", "Input"], [133, 1, 1, "c.dp_neg_grad_p", "length"], [133, 1, 1, "c.dp_neg_grad_p", "output"]], "dp_neg_grad_s": [[133, 1, 1, "c.dp_neg_grad_s", "Input"], [133, 1, 1, "c.dp_neg_grad_s", "core_mask"], [133, 1, 1, "c.dp_neg_grad_s", "length"], [133, 1, 1, "c.dp_neg_grad_s", "output"]], "dp_neg_p": [[132, 1, 1, "c.dp_neg_p", "Input"], [132, 1, 1, "c.dp_neg_p", "length"], [132, 1, 1, "c.dp_neg_p", "output"]], "dp_neg_s": [[132, 1, 1, "c.dp_neg_s", "Input"], [132, 1, 1, "c.dp_neg_s", "core_mask"], [132, 1, 1, "c.dp_neg_s", "length"], [132, 1, 1, "c.dp_neg_s", "output"]], "dp_nonzero_p": [[137, 1, 1, "c.dp_nonzero_p", "dim_strides"], [137, 1, 1, "c.dp_nonzero_p", "input"], [137, 1, 1, "c.dp_nonzero_p", "input_rank"], [137, 1, 1, "c.dp_nonzero_p", "length"], [137, 1, 1, "c.dp_nonzero_p", "non_zero_num"], [137, 1, 1, "c.dp_nonzero_p", "output"], [137, 1, 1, "c.dp_nonzero_p", "shape"]], "dp_nonzero_s": [[137, 1, 1, "c.dp_nonzero_s", "core_mask"], [137, 1, 1, "c.dp_nonzero_s", "dim_strides"], [137, 1, 1, "c.dp_nonzero_s", "input"], [137, 1, 1, "c.dp_nonzero_s", "input_rank"], [137, 1, 1, "c.dp_nonzero_s", "length"], [137, 1, 1, "c.dp_nonzero_s", "non_zero_num"], [137, 1, 1, "c.dp_nonzero_s", "output"], [137, 1, 1, "c.dp_nonzero_s", "shape"]], "dp_not_equal_p": [[138, 1, 1, "c.dp_not_equal_p", "Input0"], [138, 1, 1, "c.dp_not_equal_p", "Input1"], [138, 1, 1, "c.dp_not_equal_p", "length"], [138, 1, 1, "c.dp_not_equal_p", "output"]], "dp_not_equal_s": [[138, 1, 1, "c.dp_not_equal_s", "Input0"], [138, 1, 1, "c.dp_not_equal_s", "Input1"], [138, 1, 1, "c.dp_not_equal_s", "core_mask"], [138, 1, 1, "c.dp_not_equal_s", "length"], [138, 1, 1, "c.dp_not_equal_s", "output"]], "dp_onehot_p": [[139, 1, 1, "c.dp_onehot_p", "axis"], [139, 1, 1, "c.dp_onehot_p", "depth"], [139, 1, 1, "c.dp_onehot_p", "indices"], [139, 1, 1, "c.dp_onehot_p", "indices_shape"], [139, 1, 1, "c.dp_onehot_p", "indices_shape_size"], [139, 1, 1, "c.dp_onehot_p", "on_off"], [139, 1, 1, "c.dp_onehot_p", "output"], [139, 1, 1, "c.dp_onehot_p", "support_neg_index"]], "dp_onehot_s": [[139, 1, 1, "c.dp_onehot_s", "axis"], [139, 1, 1, "c.dp_onehot_s", "core_mask"], [139, 1, 1, "c.dp_onehot_s", "depth"], [139, 1, 1, "c.dp_onehot_s", "indices"], [139, 1, 1, "c.dp_onehot_s", "indices_shape"], [139, 1, 1, "c.dp_onehot_s", "indices_shape_size"], [139, 1, 1, "c.dp_onehot_s", "on_off"], [139, 1, 1, "c.dp_onehot_s", "output"], [139, 1, 1, "c.dp_onehot_s", "support_neg_index"]], "dp_ones_like_p": [[140, 1, 1, "c.dp_ones_like_p", "length"], [140, 1, 1, "c.dp_ones_like_p", "output"]], "dp_ones_like_s": [[140, 1, 1, "c.dp_ones_like_s", "core_mask"], [140, 1, 1, "c.dp_ones_like_s", "length"], [140, 1, 1, "c.dp_ones_like_s", "output"]], "dp_padfusion_p": [[141, 1, 1, "c.dp_padfusion_p", "params"]], "dp_padfusion_s": [[141, 1, 1, "c.dp_padfusion_s", "core_mask"], [141, 1, 1, "c.dp_padfusion_s", "params"]], "dp_pow_fusion_p": [[142, 1, 1, "c.dp_pow_fusion_p", "Input"], [142, 1, 1, "c.dp_pow_fusion_p", "broadcast"], [142, 1, 1, "c.dp_pow_fusion_p", "exponent"], [142, 1, 1, "c.dp_pow_fusion_p", "length_in"], [142, 1, 1, "c.dp_pow_fusion_p", "output"], [142, 1, 1, "c.dp_pow_fusion_p", "scale"], [142, 1, 1, "c.dp_pow_fusion_p", "shift"]], "dp_pow_fusion_s": [[142, 1, 1, "c.dp_pow_fusion_s", "Input"], [142, 1, 1, "c.dp_pow_fusion_s", "broadcast"], [142, 1, 1, "c.dp_pow_fusion_s", "core_mask"], [142, 1, 1, "c.dp_pow_fusion_s", "exponent"], [142, 1, 1, "c.dp_pow_fusion_s", "length_in"], [142, 1, 1, "c.dp_pow_fusion_s", "output"], [142, 1, 1, "c.dp_pow_fusion_s", "scale"], [142, 1, 1, "c.dp_pow_fusion_s", "shift"]], "dp_raggedrange_p": [[147, 1, 1, "c.dp_raggedrange_p", "deltas"], [147, 1, 1, "c.dp_raggedrange_p", "limits"], [147, 1, 1, "c.dp_raggedrange_p", "range_count"], [147, 1, 1, "c.dp_raggedrange_p", "splits"], [147, 1, 1, "c.dp_raggedrange_p", "starts"], [147, 1, 1, "c.dp_raggedrange_p", "values"]], "dp_raggedrange_s": [[147, 1, 1, "c.dp_raggedrange_s", "core_mask"], [147, 1, 1, "c.dp_raggedrange_s", "deltas"], [147, 1, 1, "c.dp_raggedrange_s", "limits"], [147, 1, 1, "c.dp_raggedrange_s", "range_count"], [147, 1, 1, "c.dp_raggedrange_s", "splits"], [147, 1, 1, "c.dp_raggedrange_s", "starts"], [147, 1, 1, "c.dp_raggedrange_s", "values"]], "dp_range_p": [[150, 1, 1, "c.dp_range_p", "delta"], [150, 1, 1, "c.dp_range_p", "length"], [150, 1, 1, "c.dp_range_p", "output"], [150, 1, 1, "c.dp_range_p", "start"]], "dp_range_s": [[150, 1, 1, "c.dp_range_s", "core_mask"], [150, 1, 1, "c.dp_range_s", "delta"], [150, 1, 1, "c.dp_range_s", "length"], [150, 1, 1, "c.dp_range_s", "output"], [150, 1, 1, "c.dp_range_s", "start"]], "dp_real_div_p": [[152, 1, 1, "c.dp_real_div_p", "input0"], [152, 1, 1, "c.dp_real_div_p", "input1"], [152, 1, 1, "c.dp_real_div_p", "length"], [152, 1, 1, "c.dp_real_div_p", "output"]], "dp_real_div_s": [[152, 1, 1, "c.dp_real_div_s", "core_mask"], [152, 1, 1, "c.dp_real_div_s", "input0"], [152, 1, 1, "c.dp_real_div_s", "input1"], [152, 1, 1, "c.dp_real_div_s", "length"], [152, 1, 1, "c.dp_real_div_s", "output"]], "dp_reciprocal_p": [[153, 1, 1, "c.dp_reciprocal_p", "Input"], [153, 1, 1, "c.dp_reciprocal_p", "length"], [153, 1, 1, "c.dp_reciprocal_p", "output"]], "dp_reciprocal_s": [[153, 1, 1, "c.dp_reciprocal_s", "Input"], [153, 1, 1, "c.dp_reciprocal_s", "core_mask"], [153, 1, 1, "c.dp_reciprocal_s", "length"], [153, 1, 1, "c.dp_reciprocal_s", "output"]], "dp_reduce_p": [[154, 1, 1, "c.dp_reduce_p", "core_mask"], [154, 1, 1, "c.dp_reduce_p", "dst_data"], [154, 1, 1, "c.dp_reduce_p", "param"], [154, 1, 1, "c.dp_reduce_p", "src_data"], [154, 1, 1, "c.dp_reduce_p", "tmp_dst_data"], [154, 1, 1, "c.dp_reduce_p", "tmp_src_data"]], "dp_reduce_s": [[154, 1, 1, "c.dp_reduce_s", "core_mask"], [154, 1, 1, "c.dp_reduce_s", "dst_data"], [154, 1, 1, "c.dp_reduce_s", "param"], [154, 1, 1, "c.dp_reduce_s", "src_data"]], "dp_reduceall_p": [[21, 1, 1, "c.dp_reduceall_p", "axis_size"], [21, 1, 1, "c.dp_reduceall_p", "dst_data"], [21, 1, 1, "c.dp_reduceall_p", "inner_size"], [21, 1, 1, "c.dp_reduceall_p", "outer_size"], [21, 1, 1, "c.dp_reduceall_p", "src_data"]], "dp_reduceall_s": [[21, 1, 1, "c.dp_reduceall_s", "axis_size"], [21, 1, 1, "c.dp_reduceall_s", "core_mask"], [21, 1, 1, "c.dp_reduceall_s", "dst_data"], [21, 1, 1, "c.dp_reduceall_s", "inner_size"], [21, 1, 1, "c.dp_reduceall_s", "outer_size"], [21, 1, 1, "c.dp_reduceall_s", "src_data"]], "dp_reducescatter_p": [[155, 1, 1, "c.dp_reducescatter_p", "data_size"], [155, 1, 1, "c.dp_reducescatter_p", "input_data"], [155, 1, 1, "c.dp_reducescatter_p", "output_data"], [155, 1, 1, "c.dp_reducescatter_p", "reduce_type"]], "dp_reducescatter_s": [[155, 1, 1, "c.dp_reducescatter_s", "core_mask"], [155, 1, 1, "c.dp_reducescatter_s", "data_size"], [155, 1, 1, "c.dp_reducescatter_s", "input_data"], [155, 1, 1, "c.dp_reducescatter_s", "output_data"], [155, 1, 1, "c.dp_reducescatter_s", "reduce_type"]], "dp_reshape_p": [[156, 1, 1, "c.dp_reshape_p", "input"], [156, 1, 1, "c.dp_reshape_p", "length"], [156, 1, 1, "c.dp_reshape_p", "output"]], "dp_reshape_s": [[156, 1, 1, "c.dp_reshape_s", "core_mask"], [156, 1, 1, "c.dp_reshape_s", "input"], [156, 1, 1, "c.dp_reshape_s", "length"], [156, 1, 1, "c.dp_reshape_s", "output"]], "dp_round_p": [[163, 1, 1, "c.dp_round_p", "input"], [163, 1, 1, "c.dp_round_p", "length"], [163, 1, 1, "c.dp_round_p", "output"]], "dp_round_s": [[163, 1, 1, "c.dp_round_s", "core_mask"], [163, 1, 1, "c.dp_round_s", "input"], [163, 1, 1, "c.dp_round_s", "length"], [163, 1, 1, "c.dp_round_s", "output"]], "dp_rsqrt_p": [[164, 1, 1, "c.dp_rsqrt_p", "dst"], [164, 1, 1, "c.dp_rsqrt_p", "length"], [164, 1, 1, "c.dp_rsqrt_p", "src"]], "dp_rsqrt_s": [[164, 1, 1, "c.dp_rsqrt_s", "core_mask"], [164, 1, 1, "c.dp_rsqrt_s", "dst"], [164, 1, 1, "c.dp_rsqrt_s", "length"], [164, 1, 1, "c.dp_rsqrt_s", "src"]], "dp_scalefusion_p": [[166, 1, 1, "c.dp_scalefusion_p", "bias"], [166, 1, 1, "c.dp_scalefusion_p", "dst_data"], [166, 1, 1, "c.dp_scalefusion_p", "length"], [166, 1, 1, "c.dp_scalefusion_p", "scale"], [166, 1, 1, "c.dp_scalefusion_p", "src_data"]], "dp_scalefusion_s": [[166, 1, 1, "c.dp_scalefusion_s", "bias"], [166, 1, 1, "c.dp_scalefusion_s", "core_mask"], [166, 1, 1, "c.dp_scalefusion_s", "dst_data"], [166, 1, 1, "c.dp_scalefusion_s", "length"], [166, 1, 1, "c.dp_scalefusion_s", "scale"], [166, 1, 1, "c.dp_scalefusion_s", "src_data"]], "dp_scatter_elements_p": [[167, 1, 1, "c.dp_scatter_elements_p", "core_mask"], [167, 1, 1, "c.dp_scatter_elements_p", "indices"], [167, 1, 1, "c.dp_scatter_elements_p", "input"], [167, 1, 1, "c.dp_scatter_elements_p", "output"], [167, 1, 1, "c.dp_scatter_elements_p", "param"], [167, 1, 1, "c.dp_scatter_elements_p", "updates"]], "dp_scatter_elements_s": [[167, 1, 1, "c.dp_scatter_elements_s", "core_mask"], [167, 1, 1, "c.dp_scatter_elements_s", "indices"], [167, 1, 1, "c.dp_scatter_elements_s", "input"], [167, 1, 1, "c.dp_scatter_elements_s", "output"], [167, 1, 1, "c.dp_scatter_elements_s", "param"], [167, 1, 1, "c.dp_scatter_elements_s", "updates"]], "dp_scatter_nd_p": [[168, 1, 1, "c.dp_scatter_nd_p", "indices"], [168, 1, 1, "c.dp_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.dp_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.dp_scatter_nd_p", "output"], [168, 1, 1, "c.dp_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.dp_scatter_nd_p", "output_shape"], [168, 1, 1, "c.dp_scatter_nd_p", "updates"]], "dp_scatter_nd_s": [[168, 1, 1, "c.dp_scatter_nd_s", "core_mask"], [168, 1, 1, "c.dp_scatter_nd_s", "indices"], [168, 1, 1, "c.dp_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.dp_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.dp_scatter_nd_s", "output"], [168, 1, 1, "c.dp_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.dp_scatter_nd_s", "output_shape"], [168, 1, 1, "c.dp_scatter_nd_s", "updates"]], "dp_scatter_nd_update_p": [[169, 1, 1, "c.dp_scatter_nd_update_p", "indices"], [169, 1, 1, "c.dp_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.dp_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.dp_scatter_nd_update_p", "output"], [169, 1, 1, "c.dp_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.dp_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.dp_scatter_nd_update_p", "updates"]], "dp_scatter_nd_update_s": [[169, 1, 1, "c.dp_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.dp_scatter_nd_update_s", "indices"], [169, 1, 1, "c.dp_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.dp_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.dp_scatter_nd_update_s", "output"], [169, 1, 1, "c.dp_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.dp_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.dp_scatter_nd_update_s", "updates"]], "dp_select_p": [[170, 1, 1, "c.dp_select_p", "condition"], [170, 1, 1, "c.dp_select_p", "index_list1"], [170, 1, 1, "c.dp_select_p", "index_list2"], [170, 1, 1, "c.dp_select_p", "index_list3"], [170, 1, 1, "c.dp_select_p", "input0"], [170, 1, 1, "c.dp_select_p", "input1"], [170, 1, 1, "c.dp_select_p", "is_broadcast"], [170, 1, 1, "c.dp_select_p", "output"], [170, 1, 1, "c.dp_select_p", "output_dims"], [170, 1, 1, "c.dp_select_p", "output_dims_num"]], "dp_select_s": [[170, 1, 1, "c.dp_select_s", "condition"], [170, 1, 1, "c.dp_select_s", "core_mask"], [170, 1, 1, "c.dp_select_s", "index_list1"], [170, 1, 1, "c.dp_select_s", "index_list2"], [170, 1, 1, "c.dp_select_s", "index_list3"], [170, 1, 1, "c.dp_select_s", "input0"], [170, 1, 1, "c.dp_select_s", "input1"], [170, 1, 1, "c.dp_select_s", "is_broadcast"], [170, 1, 1, "c.dp_select_s", "output"], [170, 1, 1, "c.dp_select_s", "output_dims"], [170, 1, 1, "c.dp_select_s", "output_dims_num"]], "dp_sin_p": [[175, 1, 1, "c.dp_sin_p", "dst_data"], [175, 1, 1, "c.dp_sin_p", "length"], [175, 1, 1, "c.dp_sin_p", "src_data"]], "dp_sin_s": [[175, 1, 1, "c.dp_sin_s", "core_mask"], [175, 1, 1, "c.dp_sin_s", "dst_data"], [175, 1, 1, "c.dp_sin_s", "length"], [175, 1, 1, "c.dp_sin_s", "src_data"]], "dp_slice_p": [[178, 1, 1, "c.dp_slice_p", "begin"], [178, 1, 1, "c.dp_slice_p", "input"], [178, 1, 1, "c.dp_slice_p", "input_shape"], [178, 1, 1, "c.dp_slice_p", "ndim"], [178, 1, 1, "c.dp_slice_p", "output"], [178, 1, 1, "c.dp_slice_p", "size"]], "dp_slice_s": [[178, 1, 1, "c.dp_slice_s", "begin"], [178, 1, 1, "c.dp_slice_s", "core_mask"], [178, 1, 1, "c.dp_slice_s", "input"], [178, 1, 1, "c.dp_slice_s", "input_shape"], [178, 1, 1, "c.dp_slice_s", "ndim"], [178, 1, 1, "c.dp_slice_s", "output"], [178, 1, 1, "c.dp_slice_s", "size"]], "dp_spacetobatch_p": [[183, 1, 1, "c.dp_spacetobatch_p", "block_size"], [183, 1, 1, "c.dp_spacetobatch_p", "data_size"], [183, 1, 1, "c.dp_spacetobatch_p", "input"], [183, 1, 1, "c.dp_spacetobatch_p", "input_shape"], [183, 1, 1, "c.dp_spacetobatch_p", "output"], [183, 1, 1, "c.dp_spacetobatch_p", "paddings"]], "dp_spacetobatch_s": [[183, 1, 1, "c.dp_spacetobatch_s", "block_size"], [183, 1, 1, "c.dp_spacetobatch_s", "core_mask"], [183, 1, 1, "c.dp_spacetobatch_s", "data_size"], [183, 1, 1, "c.dp_spacetobatch_s", "input"], [183, 1, 1, "c.dp_spacetobatch_s", "input_shape"], [183, 1, 1, "c.dp_spacetobatch_s", "output"], [183, 1, 1, "c.dp_spacetobatch_s", "paddings"]], "dp_spacetobatchnd_p": [[184, 1, 1, "c.dp_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.dp_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.dp_spacetobatchnd_p", "input"], [184, 1, 1, "c.dp_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.dp_spacetobatchnd_p", "output"], [184, 1, 1, "c.dp_spacetobatchnd_p", "paddings"]], "dp_spacetobatchnd_s": [[184, 1, 1, "c.dp_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.dp_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.dp_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.dp_spacetobatchnd_s", "input"], [184, 1, 1, "c.dp_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.dp_spacetobatchnd_s", "output"], [184, 1, 1, "c.dp_spacetobatchnd_s", "paddings"]], "dp_spacetodepth_p": [[185, 1, 1, "c.dp_spacetodepth_p", "block"], [185, 1, 1, "c.dp_spacetodepth_p", "data_size"], [185, 1, 1, "c.dp_spacetodepth_p", "in_shape"], [185, 1, 1, "c.dp_spacetodepth_p", "input"], [185, 1, 1, "c.dp_spacetodepth_p", "output"]], "dp_spacetodepth_s": [[185, 1, 1, "c.dp_spacetodepth_s", "block"], [185, 1, 1, "c.dp_spacetodepth_s", "core_mask"], [185, 1, 1, "c.dp_spacetodepth_s", "data_size"], [185, 1, 1, "c.dp_spacetodepth_s", "in_shape"], [185, 1, 1, "c.dp_spacetodepth_s", "input"], [185, 1, 1, "c.dp_spacetodepth_s", "output"]], "dp_sparsefillemptyrows_p": [[187, 1, 1, "c.dp_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_p", "values_ptr"]], "dp_sparsefillemptyrows_s": [[187, 1, 1, "c.dp_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.dp_sparsefillemptyrows_s", "values_ptr"]], "dp_sparsesegmentsum_p": [[189, 1, 1, "c.dp_sparsesegmentsum_p", "in_data"], [189, 1, 1, "c.dp_sparsesegmentsum_p", "in_data_shape"], [189, 1, 1, "c.dp_sparsesegmentsum_p", "in_data_shape_size"], [189, 1, 1, "c.dp_sparsesegmentsum_p", "in_indices"], [189, 1, 1, "c.dp_sparsesegmentsum_p", "in_indices_size"], [189, 1, 1, "c.dp_sparsesegmentsum_p", "in_segment_ids"], [189, 1, 1, "c.dp_sparsesegmentsum_p", "out_data"], [189, 1, 1, "c.dp_sparsesegmentsum_p", "out_data_shape"]], "dp_sparsesegmentsum_s": [[189, 1, 1, "c.dp_sparsesegmentsum_s", "core_mask"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "in_data"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "in_data_shape"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "in_data_shape_size"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "in_indices"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "in_indices_size"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "in_segment_ids"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "out_data"], [189, 1, 1, "c.dp_sparsesegmentsum_s", "out_data_shape"]], "dp_sparsetodense_p": [[190, 1, 1, "c.dp_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.dp_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.dp_sparsetodense_p", "output"], [190, 1, 1, "c.dp_sparsetodense_p", "output_strides"], [190, 1, 1, "c.dp_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.dp_sparsetodense_p", "sparse_values"]], "dp_sparsetodense_s": [[190, 1, 1, "c.dp_sparsetodense_s", "core_mask"], [190, 1, 1, "c.dp_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.dp_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.dp_sparsetodense_s", "output"], [190, 1, 1, "c.dp_sparsetodense_s", "output_strides"], [190, 1, 1, "c.dp_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.dp_sparsetodense_s", "sparse_values"]], "dp_splice_p": [[191, 1, 1, "c.dp_splice_p", "context_dim"], [191, 1, 1, "c.dp_splice_p", "dst_col"], [191, 1, 1, "c.dp_splice_p", "dst_data"], [191, 1, 1, "c.dp_splice_p", "dst_row"], [191, 1, 1, "c.dp_splice_p", "forward_indexes"], [191, 1, 1, "c.dp_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.dp_splice_p", "src_col"], [191, 1, 1, "c.dp_splice_p", "src_data"], [191, 1, 1, "c.dp_splice_p", "src_row"]], "dp_splice_s": [[191, 1, 1, "c.dp_splice_s", "context_dim"], [191, 1, 1, "c.dp_splice_s", "core_mask"], [191, 1, 1, "c.dp_splice_s", "dst_col"], [191, 1, 1, "c.dp_splice_s", "dst_data"], [191, 1, 1, "c.dp_splice_s", "dst_row"], [191, 1, 1, "c.dp_splice_s", "forward_indexes"], [191, 1, 1, "c.dp_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.dp_splice_s", "src_col"], [191, 1, 1, "c.dp_splice_s", "src_data"], [191, 1, 1, "c.dp_splice_s", "src_row"]], "dp_split_p": [[192, 1, 1, "c.dp_split_p", "axis"], [192, 1, 1, "c.dp_split_p", "input"], [192, 1, 1, "c.dp_split_p", "input_ndim"], [192, 1, 1, "c.dp_split_p", "input_shape"], [192, 1, 1, "c.dp_split_p", "num_split"], [192, 1, 1, "c.dp_split_p", "outputs"], [192, 1, 1, "c.dp_split_p", "split_sizes"]], "dp_split_s": [[192, 1, 1, "c.dp_split_s", "axis"], [192, 1, 1, "c.dp_split_s", "core_mask"], [192, 1, 1, "c.dp_split_s", "input"], [192, 1, 1, "c.dp_split_s", "input_ndim"], [192, 1, 1, "c.dp_split_s", "input_shape"], [192, 1, 1, "c.dp_split_s", "num_split"], [192, 1, 1, "c.dp_split_s", "outputs"], [192, 1, 1, "c.dp_split_s", "split_sizes"]], "dp_split_with_overlap_p": [[193, 1, 1, "c.dp_split_with_overlap_p", "axis"], [193, 1, 1, "c.dp_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.dp_split_with_overlap_p", "input"], [193, 1, 1, "c.dp_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.dp_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.dp_split_with_overlap_p", "num_split"], [193, 1, 1, "c.dp_split_with_overlap_p", "outputs"], [193, 1, 1, "c.dp_split_with_overlap_p", "start_indices"]], "dp_split_with_overlap_s": [[193, 1, 1, "c.dp_split_with_overlap_s", "axis"], [193, 1, 1, "c.dp_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.dp_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.dp_split_with_overlap_s", "input"], [193, 1, 1, "c.dp_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.dp_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.dp_split_with_overlap_s", "num_split"], [193, 1, 1, "c.dp_split_with_overlap_s", "outputs"], [193, 1, 1, "c.dp_split_with_overlap_s", "start_indices"]], "dp_sqrt_p": [[194, 1, 1, "c.dp_sqrt_p", "dst_data"], [194, 1, 1, "c.dp_sqrt_p", "length"], [194, 1, 1, "c.dp_sqrt_p", "src_data"]], "dp_sqrt_s": [[194, 1, 1, "c.dp_sqrt_s", "core_mask"], [194, 1, 1, "c.dp_sqrt_s", "dst_data"], [194, 1, 1, "c.dp_sqrt_s", "length"], [194, 1, 1, "c.dp_sqrt_s", "src_data"]], "dp_sqrtgrad_p": [[195, 1, 1, "c.dp_sqrtgrad_p", "input1"], [195, 1, 1, "c.dp_sqrtgrad_p", "input2"], [195, 1, 1, "c.dp_sqrtgrad_p", "output"], [195, 1, 1, "c.dp_sqrtgrad_p", "size"]], "dp_sqrtgrad_s": [[195, 1, 1, "c.dp_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.dp_sqrtgrad_s", "input1"], [195, 1, 1, "c.dp_sqrtgrad_s", "input2"], [195, 1, 1, "c.dp_sqrtgrad_s", "output"], [195, 1, 1, "c.dp_sqrtgrad_s", "size"]], "dp_square_p": [[196, 1, 1, "c.dp_square_p", "dst"], [196, 1, 1, "c.dp_square_p", "length"], [196, 1, 1, "c.dp_square_p", "src"]], "dp_square_s": [[196, 1, 1, "c.dp_square_s", "core_mask"], [196, 1, 1, "c.dp_square_s", "dst"], [196, 1, 1, "c.dp_square_s", "length"], [196, 1, 1, "c.dp_square_s", "src"]], "dp_squaredifference_p": [[197, 1, 1, "c.dp_squaredifference_p", "input0"], [197, 1, 1, "c.dp_squaredifference_p", "input1"], [197, 1, 1, "c.dp_squaredifference_p", "length"], [197, 1, 1, "c.dp_squaredifference_p", "output"]], "dp_squaredifference_s": [[197, 1, 1, "c.dp_squaredifference_s", "core_mask"], [197, 1, 1, "c.dp_squaredifference_s", "input0"], [197, 1, 1, "c.dp_squaredifference_s", "input1"], [197, 1, 1, "c.dp_squaredifference_s", "length"], [197, 1, 1, "c.dp_squaredifference_s", "output"]], "dp_stack_p": [[199, 1, 1, "c.dp_stack_p", "axis"], [199, 1, 1, "c.dp_stack_p", "input_ndim"], [199, 1, 1, "c.dp_stack_p", "input_shape"], [199, 1, 1, "c.dp_stack_p", "inputs"], [199, 1, 1, "c.dp_stack_p", "num_inputs"], [199, 1, 1, "c.dp_stack_p", "output"]], "dp_stack_s": [[199, 1, 1, "c.dp_stack_s", "axis"], [199, 1, 1, "c.dp_stack_s", "core_mask"], [199, 1, 1, "c.dp_stack_s", "input_ndim"], [199, 1, 1, "c.dp_stack_s", "input_shape"], [199, 1, 1, "c.dp_stack_s", "inputs"], [199, 1, 1, "c.dp_stack_s", "num_inputs"], [199, 1, 1, "c.dp_stack_s", "output"]], "dp_subext_p": [[202, 1, 1, "c.dp_subext_p", "alpha"], [202, 1, 1, "c.dp_subext_p", "input0"], [202, 1, 1, "c.dp_subext_p", "input1"], [202, 1, 1, "c.dp_subext_p", "output"], [202, 1, 1, "c.dp_subext_p", "size"]], "dp_subext_s": [[202, 1, 1, "c.dp_subext_s", "alpha"], [202, 1, 1, "c.dp_subext_s", "core_mask"], [202, 1, 1, "c.dp_subext_s", "input0"], [202, 1, 1, "c.dp_subext_s", "input1"], [202, 1, 1, "c.dp_subext_s", "output"], [202, 1, 1, "c.dp_subext_s", "size"]], "dp_subrelu6_p": [[202, 1, 1, "c.dp_subrelu6_p", "input0"], [202, 1, 1, "c.dp_subrelu6_p", "input1"], [202, 1, 1, "c.dp_subrelu6_p", "output"], [202, 1, 1, "c.dp_subrelu6_p", "size"]], "dp_subrelu6_s": [[202, 1, 1, "c.dp_subrelu6_s", "core_mask"], [202, 1, 1, "c.dp_subrelu6_s", "input0"], [202, 1, 1, "c.dp_subrelu6_s", "input1"], [202, 1, 1, "c.dp_subrelu6_s", "output"], [202, 1, 1, "c.dp_subrelu6_s", "size"]], "dp_subrelu_p": [[202, 1, 1, "c.dp_subrelu_p", "input0"], [202, 1, 1, "c.dp_subrelu_p", "input1"], [202, 1, 1, "c.dp_subrelu_p", "output"], [202, 1, 1, "c.dp_subrelu_p", "size"]], "dp_subrelu_s": [[202, 1, 1, "c.dp_subrelu_s", "core_mask"], [202, 1, 1, "c.dp_subrelu_s", "input0"], [202, 1, 1, "c.dp_subrelu_s", "input1"], [202, 1, 1, "c.dp_subrelu_s", "output"], [202, 1, 1, "c.dp_subrelu_s", "size"]], "dp_tensor_scatter_add_p": [[206, 1, 1, "c.dp_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "input"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "output"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.dp_tensor_scatter_add_p", "updates"]], "dp_tensor_scatter_add_s": [[206, 1, 1, "c.dp_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "input"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "output"], [206, 1, 1, "c.dp_tensor_scatter_add_s", "updates"]], "dp_tensorarrayread_p": [[208, 1, 1, "c.dp_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.dp_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.dp_tensorarrayread_p", "index"], [208, 1, 1, "c.dp_tensorarrayread_p", "output_data"], [208, 1, 1, "c.dp_tensorarrayread_p", "output_size"]], "dp_tensorarrayread_s": [[208, 1, 1, "c.dp_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.dp_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.dp_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.dp_tensorarrayread_s", "index"], [208, 1, 1, "c.dp_tensorarrayread_s", "output_data"], [208, 1, 1, "c.dp_tensorarrayread_s", "output_size"]], "dp_tensorlistfromtensor_p": [[210, 1, 1, "c.dp_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.dp_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.dp_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.dp_tensorlistfromtensor_p", "output_tensors"]], "dp_tensorlistfromtensor_s": [[210, 1, 1, "c.dp_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.dp_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.dp_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.dp_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.dp_tensorlistfromtensor_s", "output_tensors"]], "dp_tile_p": [[215, 1, 1, "c.dp_tile_p", "input"], [215, 1, 1, "c.dp_tile_p", "input_shape"], [215, 1, 1, "c.dp_tile_p", "output"], [215, 1, 1, "c.dp_tile_p", "stride"], [215, 1, 1, "c.dp_tile_p", "tile_dim"], [215, 1, 1, "c.dp_tile_p", "tile_num"]], "dp_tile_s": [[215, 1, 1, "c.dp_tile_s", "core_mask"], [215, 1, 1, "c.dp_tile_s", "input"], [215, 1, 1, "c.dp_tile_s", "input_shape"], [215, 1, 1, "c.dp_tile_s", "output"], [215, 1, 1, "c.dp_tile_s", "stride"], [215, 1, 1, "c.dp_tile_s", "tile_dim"], [215, 1, 1, "c.dp_tile_s", "tile_num"]], "dp_transpose_p": [[217, 1, 1, "c.dp_transpose_p", "in_data"], [217, 1, 1, "c.dp_transpose_p", "num_axes"], [217, 1, 1, "c.dp_transpose_p", "out_data"], [217, 1, 1, "c.dp_transpose_p", "out_strides"], [217, 1, 1, "c.dp_transpose_p", "output_shape"], [217, 1, 1, "c.dp_transpose_p", "perm"], [217, 1, 1, "c.dp_transpose_p", "strides"]], "dp_transpose_s": [[217, 1, 1, "c.dp_transpose_s", "core_mask"], [217, 1, 1, "c.dp_transpose_s", "in_data"], [217, 1, 1, "c.dp_transpose_s", "num_axes"], [217, 1, 1, "c.dp_transpose_s", "out_data"], [217, 1, 1, "c.dp_transpose_s", "out_strides"], [217, 1, 1, "c.dp_transpose_s", "output_shape"], [217, 1, 1, "c.dp_transpose_s", "perm"], [217, 1, 1, "c.dp_transpose_s", "strides"]], "dp_tril_p": [[218, 1, 1, "c.dp_tril_p", "dst"], [218, 1, 1, "c.dp_tril_p", "height"], [218, 1, 1, "c.dp_tril_p", "k"], [218, 1, 1, "c.dp_tril_p", "out_elems"], [218, 1, 1, "c.dp_tril_p", "src"], [218, 1, 1, "c.dp_tril_p", "width"]], "dp_tril_s": [[218, 1, 1, "c.dp_tril_s", "core_mask"], [218, 1, 1, "c.dp_tril_s", "dst"], [218, 1, 1, "c.dp_tril_s", "height"], [218, 1, 1, "c.dp_tril_s", "k"], [218, 1, 1, "c.dp_tril_s", "out_elems"], [218, 1, 1, "c.dp_tril_s", "src"], [218, 1, 1, "c.dp_tril_s", "width"]], "dp_triu_p": [[219, 1, 1, "c.dp_triu_p", "dst"], [219, 1, 1, "c.dp_triu_p", "height"], [219, 1, 1, "c.dp_triu_p", "k"], [219, 1, 1, "c.dp_triu_p", "out_elems"], [219, 1, 1, "c.dp_triu_p", "src"], [219, 1, 1, "c.dp_triu_p", "width"]], "dp_triu_s": [[219, 1, 1, "c.dp_triu_s", "core_mask"], [219, 1, 1, "c.dp_triu_s", "dst"], [219, 1, 1, "c.dp_triu_s", "height"], [219, 1, 1, "c.dp_triu_s", "k"], [219, 1, 1, "c.dp_triu_s", "out_elems"], [219, 1, 1, "c.dp_triu_s", "src"], [219, 1, 1, "c.dp_triu_s", "width"]], "dp_unsorted_segment_sum_p": [[222, 1, 1, "c.dp_unsorted_segment_sum_p", "dim0"], [222, 1, 1, "c.dp_unsorted_segment_sum_p", "dim1"], [222, 1, 1, "c.dp_unsorted_segment_sum_p", "id_max"], [222, 1, 1, "c.dp_unsorted_segment_sum_p", "index"], [222, 1, 1, "c.dp_unsorted_segment_sum_p", "input"], [222, 1, 1, "c.dp_unsorted_segment_sum_p", "output"]], "dp_unsorted_segment_sum_s": [[222, 1, 1, "c.dp_unsorted_segment_sum_s", "core_mask"], [222, 1, 1, "c.dp_unsorted_segment_sum_s", "dim0"], [222, 1, 1, "c.dp_unsorted_segment_sum_s", "dim1"], [222, 1, 1, "c.dp_unsorted_segment_sum_s", "id_max"], [222, 1, 1, "c.dp_unsorted_segment_sum_s", "index"], [222, 1, 1, "c.dp_unsorted_segment_sum_s", "input"], [222, 1, 1, "c.dp_unsorted_segment_sum_s", "output"]], "dp_where_p": [[225, 1, 1, "c.dp_where_p", "condition"], [225, 1, 1, "c.dp_where_p", "input0"], [225, 1, 1, "c.dp_where_p", "input1"], [225, 1, 1, "c.dp_where_p", "length"], [225, 1, 1, "c.dp_where_p", "output"]], "dp_where_s": [[225, 1, 1, "c.dp_where_s", "condition"], [225, 1, 1, "c.dp_where_s", "core_mask"], [225, 1, 1, "c.dp_where_s", "input0"], [225, 1, 1, "c.dp_where_s", "input1"], [225, 1, 1, "c.dp_where_s", "length"], [225, 1, 1, "c.dp_where_s", "output"]], "dp_zerolike_p": [[226, 1, 1, "c.dp_zerolike_p", "length"], [226, 1, 1, "c.dp_zerolike_p", "output"]], "dp_zerolike_s": [[226, 1, 1, "c.dp_zerolike_s", "core_mask"], [226, 1, 1, "c.dp_zerolike_s", "length"], [226, 1, 1, "c.dp_zerolike_s", "output"]], "fp_Gru_p": [[95, 1, 1, "c.fp_Gru_p", "buffer"], [95, 1, 1, "c.fp_Gru_p", "core_mask"], [95, 1, 1, "c.fp_Gru_p", "gru_param"], [95, 1, 1, "c.fp_Gru_p", "hidden_state"], [95, 1, 1, "c.fp_Gru_p", "input"], [95, 1, 1, "c.fp_Gru_p", "input_bias"], [95, 1, 1, "c.fp_Gru_p", "output"], [95, 1, 1, "c.fp_Gru_p", "state_bias"], [95, 1, 1, "c.fp_Gru_p", "weight_g"], [95, 1, 1, "c.fp_Gru_p", "weight_r"]], "fp_Gru_s": [[95, 1, 1, "c.fp_Gru_s", "buffer"], [95, 1, 1, "c.fp_Gru_s", "core_mask"], [95, 1, 1, "c.fp_Gru_s", "gru_param"], [95, 1, 1, "c.fp_Gru_s", "hidden_state"], [95, 1, 1, "c.fp_Gru_s", "input"], [95, 1, 1, "c.fp_Gru_s", "input_bias"], [95, 1, 1, "c.fp_Gru_s", "output"], [95, 1, 1, "c.fp_Gru_s", "state_bias"], [95, 1, 1, "c.fp_Gru_s", "weight_g"], [95, 1, 1, "c.fp_Gru_s", "weight_r"]], "fp_Lstm_p": [[117, 1, 1, "c.fp_Lstm_p", "buffer"], [117, 1, 1, "c.fp_Lstm_p", "cell_state"], [117, 1, 1, "c.fp_Lstm_p", "hidden_state"], [117, 1, 1, "c.fp_Lstm_p", "input"], [117, 1, 1, "c.fp_Lstm_p", "input_bias"], [117, 1, 1, "c.fp_Lstm_p", "lstm_param"], [117, 1, 1, "c.fp_Lstm_p", "output"], [117, 1, 1, "c.fp_Lstm_p", "state_bias"], [117, 1, 1, "c.fp_Lstm_p", "weight_h"], [117, 1, 1, "c.fp_Lstm_p", "weight_i"]], "fp_Lstm_s": [[117, 1, 1, "c.fp_Lstm_s", "buffer"], [117, 1, 1, "c.fp_Lstm_s", "cell_state"], [117, 1, 1, "c.fp_Lstm_s", "core_mask"], [117, 1, 1, "c.fp_Lstm_s", "hidden_state"], [117, 1, 1, "c.fp_Lstm_s", "input"], [117, 1, 1, "c.fp_Lstm_s", "input_bias"], [117, 1, 1, "c.fp_Lstm_s", "lstm_param"], [117, 1, 1, "c.fp_Lstm_s", "output"], [117, 1, 1, "c.fp_Lstm_s", "state_bias"], [117, 1, 1, "c.fp_Lstm_s", "weight_h"], [117, 1, 1, "c.fp_Lstm_s", "weight_i"]], "fp_QuantData_p": [[66, 1, 1, "c.fp_QuantData_p", "axis_num"], [66, 1, 1, "c.fp_QuantData_p", "element_num"], [66, 1, 1, "c.fp_QuantData_p", "quant_values"], [66, 1, 1, "c.fp_QuantData_p", "real_values"], [66, 1, 1, "c.fp_QuantData_p", "scale"], [66, 1, 1, "c.fp_QuantData_p", "segment_num"], [66, 1, 1, "c.fp_QuantData_p", "zp"]], "fp_QuantData_s": [[66, 1, 1, "c.fp_QuantData_s", "axis_num"], [66, 1, 1, "c.fp_QuantData_s", "core_mask"], [66, 1, 1, "c.fp_QuantData_s", "element_num"], [66, 1, 1, "c.fp_QuantData_s", "quant_values"], [66, 1, 1, "c.fp_QuantData_s", "real_values"], [66, 1, 1, "c.fp_QuantData_s", "scale"], [66, 1, 1, "c.fp_QuantData_s", "segment_num"], [66, 1, 1, "c.fp_QuantData_s", "zp"]], "fp_Unique_p": [[221, 1, 1, "c.fp_Unique_p", "input"], [221, 1, 1, "c.fp_Unique_p", "input_len"], [221, 1, 1, "c.fp_Unique_p", "output0"], [221, 1, 1, "c.fp_Unique_p", "output0_len"]], "fp_Unique_s": [[221, 1, 1, "c.fp_Unique_s", "core_mask"], [221, 1, 1, "c.fp_Unique_s", "input"], [221, 1, 1, "c.fp_Unique_s", "input_len"], [221, 1, 1, "c.fp_Unique_s", "output0"], [221, 1, 1, "c.fp_Unique_s", "output0_len"]], "fp_abs_p": [[10, 1, 1, "c.fp_abs_p", "dst_data"], [10, 1, 1, "c.fp_abs_p", "length"], [10, 1, 1, "c.fp_abs_p", "src_data"]], "fp_abs_s": [[10, 1, 1, "c.fp_abs_s", "core_mask"], [10, 1, 1, "c.fp_abs_s", "dst_data"], [10, 1, 1, "c.fp_abs_s", "length"], [10, 1, 1, "c.fp_abs_s", "src_data"]], "fp_absgrad_p": [[11, 1, 1, "c.fp_absgrad_p", "input0"], [11, 1, 1, "c.fp_absgrad_p", "input1"], [11, 1, 1, "c.fp_absgrad_p", "output"], [11, 1, 1, "c.fp_absgrad_p", "size"]], "fp_absgrad_s": [[11, 1, 1, "c.fp_absgrad_s", "core_mask"], [11, 1, 1, "c.fp_absgrad_s", "input0"], [11, 1, 1, "c.fp_absgrad_s", "input1"], [11, 1, 1, "c.fp_absgrad_s", "output"], [11, 1, 1, "c.fp_absgrad_s", "size"]], "fp_adam_p": [[14, 1, 1, "c.fp_adam_p", "beta1"], [14, 1, 1, "c.fp_adam_p", "beta1_power"], [14, 1, 1, "c.fp_adam_p", "beta2"], [14, 1, 1, "c.fp_adam_p", "beta2_power"], [14, 1, 1, "c.fp_adam_p", "end"], [14, 1, 1, "c.fp_adam_p", "eps"], [14, 1, 1, "c.fp_adam_p", "gradient"], [14, 1, 1, "c.fp_adam_p", "learning_rate"], [14, 1, 1, "c.fp_adam_p", "m"], [14, 1, 1, "c.fp_adam_p", "nesterov"], [14, 1, 1, "c.fp_adam_p", "start"], [14, 1, 1, "c.fp_adam_p", "v"], [14, 1, 1, "c.fp_adam_p", "weight"]], "fp_adam_s": [[14, 1, 1, "c.fp_adam_s", "beta1"], [14, 1, 1, "c.fp_adam_s", "beta1_power"], [14, 1, 1, "c.fp_adam_s", "beta2"], [14, 1, 1, "c.fp_adam_s", "beta2_power"], [14, 1, 1, "c.fp_adam_s", "core_mask"], [14, 1, 1, "c.fp_adam_s", "end"], [14, 1, 1, "c.fp_adam_s", "eps"], [14, 1, 1, "c.fp_adam_s", "gradient"], [14, 1, 1, "c.fp_adam_s", "learning_rate"], [14, 1, 1, "c.fp_adam_s", "m"], [14, 1, 1, "c.fp_adam_s", "nesterov"], [14, 1, 1, "c.fp_adam_s", "start"], [14, 1, 1, "c.fp_adam_s", "v"], [14, 1, 1, "c.fp_adam_s", "weight"]], "fp_adamweightdecay_p": [[15, 1, 1, "c.fp_adamweightdecay_p", "beta1"], [15, 1, 1, "c.fp_adamweightdecay_p", "beta2"], [15, 1, 1, "c.fp_adamweightdecay_p", "decay"], [15, 1, 1, "c.fp_adamweightdecay_p", "epsilon"], [15, 1, 1, "c.fp_adamweightdecay_p", "gradient"], [15, 1, 1, "c.fp_adamweightdecay_p", "length"], [15, 1, 1, "c.fp_adamweightdecay_p", "lr"], [15, 1, 1, "c.fp_adamweightdecay_p", "m"], [15, 1, 1, "c.fp_adamweightdecay_p", "v"], [15, 1, 1, "c.fp_adamweightdecay_p", "var"]], "fp_adamweightdecay_s": [[15, 1, 1, "c.fp_adamweightdecay_s", "beta1"], [15, 1, 1, "c.fp_adamweightdecay_s", "beta2"], [15, 1, 1, "c.fp_adamweightdecay_s", "core_mask"], [15, 1, 1, "c.fp_adamweightdecay_s", "decay"], [15, 1, 1, "c.fp_adamweightdecay_s", "end"], [15, 1, 1, "c.fp_adamweightdecay_s", "epsilon"], [15, 1, 1, "c.fp_adamweightdecay_s", "gradient"], [15, 1, 1, "c.fp_adamweightdecay_s", "lr"], [15, 1, 1, "c.fp_adamweightdecay_s", "m"], [15, 1, 1, "c.fp_adamweightdecay_s", "start"], [15, 1, 1, "c.fp_adamweightdecay_s", "v"], [15, 1, 1, "c.fp_adamweightdecay_s", "var"]], "fp_adder_p": [[16, 1, 1, "c.fp_adder_p", "bias"], [16, 1, 1, "c.fp_adder_p", "conv_param"], [16, 1, 1, "c.fp_adder_p", "core_mask"], [16, 1, 1, "c.fp_adder_p", "input_w"], [16, 1, 1, "c.fp_adder_p", "input_x"], [16, 1, 1, "c.fp_adder_p", "out_y"]], "fp_adder_s": [[16, 1, 1, "c.fp_adder_s", "bias"], [16, 1, 1, "c.fp_adder_s", "core_mask"], [16, 1, 1, "c.fp_adder_s", "input_w"], [16, 1, 1, "c.fp_adder_s", "input_x"], [16, 1, 1, "c.fp_adder_s", "out_y"], [16, 1, 1, "c.fp_adder_s", "param"]], "fp_addext_p": [[17, 1, 1, "c.fp_addext_p", "alpha"], [17, 1, 1, "c.fp_addext_p", "in0"], [17, 1, 1, "c.fp_addext_p", "in1"], [17, 1, 1, "c.fp_addext_p", "out"], [17, 1, 1, "c.fp_addext_p", "size"]], "fp_addext_s": [[17, 1, 1, "c.fp_addext_s", "alpha"], [17, 1, 1, "c.fp_addext_s", "core_mask"], [17, 1, 1, "c.fp_addext_s", "in0"], [17, 1, 1, "c.fp_addext_s", "in1"], [17, 1, 1, "c.fp_addext_s", "out"], [17, 1, 1, "c.fp_addext_s", "size"]], "fp_addgrad_p": [[18, 1, 1, "c.fp_addgrad_p", "dx1"], [18, 1, 1, "c.fp_addgrad_p", "dx2"], [18, 1, 1, "c.fp_addgrad_p", "dy"], [18, 1, 1, "c.fp_addgrad_p", "dy_dims"], [18, 1, 1, "c.fp_addgrad_p", "num_dims"], [18, 1, 1, "c.fp_addgrad_p", "x1_dims"], [18, 1, 1, "c.fp_addgrad_p", "x2_dims"]], "fp_addgrad_s": [[18, 1, 1, "c.fp_addgrad_s", "core_mask"], [18, 1, 1, "c.fp_addgrad_s", "dx1"], [18, 1, 1, "c.fp_addgrad_s", "dx2"], [18, 1, 1, "c.fp_addgrad_s", "dy"], [18, 1, 1, "c.fp_addgrad_s", "dy_dims"], [18, 1, 1, "c.fp_addgrad_s", "num_dims"], [18, 1, 1, "c.fp_addgrad_s", "x1_dims"], [18, 1, 1, "c.fp_addgrad_s", "x2_dims"]], "fp_addn_p": [[19, 1, 1, "c.fp_addn_p", "input0"], [19, 1, 1, "c.fp_addn_p", "input1"], [19, 1, 1, "c.fp_addn_p", "length"], [19, 1, 1, "c.fp_addn_p", "output"]], "fp_addn_s": [[19, 1, 1, "c.fp_addn_s", "core_mask"], [19, 1, 1, "c.fp_addn_s", "input0"], [19, 1, 1, "c.fp_addn_s", "input1"], [19, 1, 1, "c.fp_addn_s", "length"], [19, 1, 1, "c.fp_addn_s", "output"]], "fp_addrelu6_p": [[17, 1, 1, "c.fp_addrelu6_p", "in0"], [17, 1, 1, "c.fp_addrelu6_p", "in1"], [17, 1, 1, "c.fp_addrelu6_p", "out"], [17, 1, 1, "c.fp_addrelu6_p", "size"]], "fp_addrelu6_s": [[17, 1, 1, "c.fp_addrelu6_s", "core_mask"], [17, 1, 1, "c.fp_addrelu6_s", "in0"], [17, 1, 1, "c.fp_addrelu6_s", "in1"], [17, 1, 1, "c.fp_addrelu6_s", "out"], [17, 1, 1, "c.fp_addrelu6_s", "size"]], "fp_addrelu_p": [[17, 1, 1, "c.fp_addrelu_p", "in0"], [17, 1, 1, "c.fp_addrelu_p", "in1"], [17, 1, 1, "c.fp_addrelu_p", "out"], [17, 1, 1, "c.fp_addrelu_p", "size"]], "fp_addrelu_s": [[17, 1, 1, "c.fp_addrelu_s", "core_mask"], [17, 1, 1, "c.fp_addrelu_s", "in0"], [17, 1, 1, "c.fp_addrelu_s", "in1"], [17, 1, 1, "c.fp_addrelu_s", "out"], [17, 1, 1, "c.fp_addrelu_s", "size"]], "fp_affine_p": [[20, 1, 1, "c.fp_affine_p", "params"]], "fp_affine_s": [[20, 1, 1, "c.fp_affine_s", "core_mask"], [20, 1, 1, "c.fp_affine_s", "params"]], "fp_allgather_p": [[22, 1, 1, "c.fp_allgather_p", "data_size"], [22, 1, 1, "c.fp_allgather_p", "input"], [22, 1, 1, "c.fp_allgather_p", "input_rank"], [22, 1, 1, "c.fp_allgather_p", "output"], [22, 1, 1, "c.fp_allgather_p", "output_rank"]], "fp_allgather_s": [[22, 1, 1, "c.fp_allgather_s", "core_mask"], [22, 1, 1, "c.fp_allgather_s", "data_size"], [22, 1, 1, "c.fp_allgather_s", "input"], [22, 1, 1, "c.fp_allgather_s", "input_rank"], [22, 1, 1, "c.fp_allgather_s", "output"], [22, 1, 1, "c.fp_allgather_s", "output_rank"]], "fp_and_p": [[112, 1, 1, "c.fp_and_p", "input0"], [112, 1, 1, "c.fp_and_p", "input1"], [112, 1, 1, "c.fp_and_p", "length"], [112, 1, 1, "c.fp_and_p", "output"]], "fp_and_s": [[112, 1, 1, "c.fp_and_s", "core_mask"], [112, 1, 1, "c.fp_and_s", "input0"], [112, 1, 1, "c.fp_and_s", "input1"], [112, 1, 1, "c.fp_and_s", "length"], [112, 1, 1, "c.fp_and_s", "output"]], "fp_applymomentum_p": [[23, 1, 1, "c.fp_applymomentum_p", "accumulate"], [23, 1, 1, "c.fp_applymomentum_p", "gradient"], [23, 1, 1, "c.fp_applymomentum_p", "learning_rate"], [23, 1, 1, "c.fp_applymomentum_p", "length"], [23, 1, 1, "c.fp_applymomentum_p", "moment"], [23, 1, 1, "c.fp_applymomentum_p", "nesterov"], [23, 1, 1, "c.fp_applymomentum_p", "weight"]], "fp_applymomentum_s": [[23, 1, 1, "c.fp_applymomentum_s", "accumulate"], [23, 1, 1, "c.fp_applymomentum_s", "core_mask"], [23, 1, 1, "c.fp_applymomentum_s", "end"], [23, 1, 1, "c.fp_applymomentum_s", "gradient"], [23, 1, 1, "c.fp_applymomentum_s", "learning_rate"], [23, 1, 1, "c.fp_applymomentum_s", "moment"], [23, 1, 1, "c.fp_applymomentum_s", "nesterov"], [23, 1, 1, "c.fp_applymomentum_s", "start"], [23, 1, 1, "c.fp_applymomentum_s", "weight"]], "fp_argmax_p": [[24, 1, 1, "c.fp_argmax_p", "arg_elements"], [24, 1, 1, "c.fp_argmax_p", "axis"], [24, 1, 1, "c.fp_argmax_p", "in_shape"], [24, 1, 1, "c.fp_argmax_p", "in_strides"], [24, 1, 1, "c.fp_argmax_p", "index"], [24, 1, 1, "c.fp_argmax_p", "input"], [24, 1, 1, "c.fp_argmax_p", "input_shape_size"], [24, 1, 1, "c.fp_argmax_p", "out_strides"], [24, 1, 1, "c.fp_argmax_p", "out_value"], [24, 1, 1, "c.fp_argmax_p", "output"], [24, 1, 1, "c.fp_argmax_p", "output_value"], [24, 1, 1, "c.fp_argmax_p", "topk"]], "fp_argmax_s": [[24, 1, 1, "c.fp_argmax_s", "arg_elements"], [24, 1, 1, "c.fp_argmax_s", "axis"], [24, 1, 1, "c.fp_argmax_s", "core_mask"], [24, 1, 1, "c.fp_argmax_s", "in_shape"], [24, 1, 1, "c.fp_argmax_s", "in_strides"], [24, 1, 1, "c.fp_argmax_s", "index"], [24, 1, 1, "c.fp_argmax_s", "input"], [24, 1, 1, "c.fp_argmax_s", "input_shape_size"], [24, 1, 1, "c.fp_argmax_s", "out_strides"], [24, 1, 1, "c.fp_argmax_s", "out_value"], [24, 1, 1, "c.fp_argmax_s", "output"], [24, 1, 1, "c.fp_argmax_s", "output_value"], [24, 1, 1, "c.fp_argmax_s", "topk"]], "fp_argmin_p": [[25, 1, 1, "c.fp_argmin_p", "arg_elements"], [25, 1, 1, "c.fp_argmin_p", "axis"], [25, 1, 1, "c.fp_argmin_p", "in_shape"], [25, 1, 1, "c.fp_argmin_p", "in_strides"], [25, 1, 1, "c.fp_argmin_p", "index"], [25, 1, 1, "c.fp_argmin_p", "input"], [25, 1, 1, "c.fp_argmin_p", "input_shape_size"], [25, 1, 1, "c.fp_argmin_p", "out_strides"], [25, 1, 1, "c.fp_argmin_p", "out_value"], [25, 1, 1, "c.fp_argmin_p", "output"], [25, 1, 1, "c.fp_argmin_p", "output_value"], [25, 1, 1, "c.fp_argmin_p", "topk"]], "fp_argmin_s": [[25, 1, 1, "c.fp_argmin_s", "arg_elements"], [25, 1, 1, "c.fp_argmin_s", "axis"], [25, 1, 1, "c.fp_argmin_s", "core_mask"], [25, 1, 1, "c.fp_argmin_s", "in_shape"], [25, 1, 1, "c.fp_argmin_s", "in_strides"], [25, 1, 1, "c.fp_argmin_s", "index"], [25, 1, 1, "c.fp_argmin_s", "input"], [25, 1, 1, "c.fp_argmin_s", "input_shape_size"], [25, 1, 1, "c.fp_argmin_s", "out_strides"], [25, 1, 1, "c.fp_argmin_s", "out_value"], [25, 1, 1, "c.fp_argmin_s", "output"], [25, 1, 1, "c.fp_argmin_s", "output_value"], [25, 1, 1, "c.fp_argmin_s", "topk"]], "fp_assign_p": [[27, 1, 1, "c.fp_assign_p", "dst"], [27, 1, 1, "c.fp_assign_p", "length"], [27, 1, 1, "c.fp_assign_p", "src"]], "fp_assign_s": [[27, 1, 1, "c.fp_assign_s", "core_mask"], [27, 1, 1, "c.fp_assign_s", "dst"], [27, 1, 1, "c.fp_assign_s", "length"], [27, 1, 1, "c.fp_assign_s", "src"]], "fp_assignadd_p": [[28, 1, 1, "c.fp_assignadd_p", "input"], [28, 1, 1, "c.fp_assignadd_p", "length"], [28, 1, 1, "c.fp_assignadd_p", "output"]], "fp_assignadd_s": [[28, 1, 1, "c.fp_assignadd_s", "core_mask"], [28, 1, 1, "c.fp_assignadd_s", "input"], [28, 1, 1, "c.fp_assignadd_s", "length"], [28, 1, 1, "c.fp_assignadd_s", "output"]], "fp_attention_p": [[29, 1, 1, "c.fp_attention_p", "K"], [29, 1, 1, "c.fp_attention_p", "Q"], [29, 1, 1, "c.fp_attention_p", "QK"], [29, 1, 1, "c.fp_attention_p", "V"], [29, 1, 1, "c.fp_attention_p", "batch_size"], [29, 1, 1, "c.fp_attention_p", "head_dim"], [29, 1, 1, "c.fp_attention_p", "head_num"], [29, 1, 1, "c.fp_attention_p", "output"], [29, 1, 1, "c.fp_attention_p", "seq_len"], [29, 1, 1, "c.fp_attention_p", "softmax_out"]], "fp_attention_s": [[29, 1, 1, "c.fp_attention_s", "K"], [29, 1, 1, "c.fp_attention_s", "Q"], [29, 1, 1, "c.fp_attention_s", "QK"], [29, 1, 1, "c.fp_attention_s", "V"], [29, 1, 1, "c.fp_attention_s", "batch_size"], [29, 1, 1, "c.fp_attention_s", "core_mask"], [29, 1, 1, "c.fp_attention_s", "head_dim"], [29, 1, 1, "c.fp_attention_s", "head_num"], [29, 1, 1, "c.fp_attention_s", "output"], [29, 1, 1, "c.fp_attention_s", "seq_len"], [29, 1, 1, "c.fp_attention_s", "softmax_out"]], "fp_audio_spectrogram_p": [[30, 1, 1, "c.fp_audio_spectrogram_p", "params"], [30, 1, 1, "c.fp_audio_spectrogram_p", "workspace"]], "fp_audio_spectrogram_s": [[30, 1, 1, "c.fp_audio_spectrogram_s", "core_mask"], [30, 1, 1, "c.fp_audio_spectrogram_s", "params"], [30, 1, 1, "c.fp_audio_spectrogram_s", "workspace"]], "fp_avgpool_fusion_p": [[31, 1, 1, "c.fp_avgpool_fusion_p", "batch"], [31, 1, 1, "c.fp_avgpool_fusion_p", "channel"], [31, 1, 1, "c.fp_avgpool_fusion_p", "in_h"], [31, 1, 1, "c.fp_avgpool_fusion_p", "in_w"], [31, 1, 1, "c.fp_avgpool_fusion_p", "input"], [31, 1, 1, "c.fp_avgpool_fusion_p", "max_val"], [31, 1, 1, "c.fp_avgpool_fusion_p", "min_val"], [31, 1, 1, "c.fp_avgpool_fusion_p", "output"], [31, 1, 1, "c.fp_avgpool_fusion_p", "pad_bottom"], [31, 1, 1, "c.fp_avgpool_fusion_p", "pad_left"], [31, 1, 1, "c.fp_avgpool_fusion_p", "pad_right"], [31, 1, 1, "c.fp_avgpool_fusion_p", "pad_top"], [31, 1, 1, "c.fp_avgpool_fusion_p", "stride_h"], [31, 1, 1, "c.fp_avgpool_fusion_p", "stride_w"], [31, 1, 1, "c.fp_avgpool_fusion_p", "win_h"], [31, 1, 1, "c.fp_avgpool_fusion_p", "win_w"]], "fp_avgpool_fusion_s": [[31, 1, 1, "c.fp_avgpool_fusion_s", "batch"], [31, 1, 1, "c.fp_avgpool_fusion_s", "channel"], [31, 1, 1, "c.fp_avgpool_fusion_s", "core_mask"], [31, 1, 1, "c.fp_avgpool_fusion_s", "in_h"], [31, 1, 1, "c.fp_avgpool_fusion_s", "in_w"], [31, 1, 1, "c.fp_avgpool_fusion_s", "input"], [31, 1, 1, "c.fp_avgpool_fusion_s", "max_val"], [31, 1, 1, "c.fp_avgpool_fusion_s", "min_val"], [31, 1, 1, "c.fp_avgpool_fusion_s", "output"], [31, 1, 1, "c.fp_avgpool_fusion_s", "pad_bottom"], [31, 1, 1, "c.fp_avgpool_fusion_s", "pad_left"], [31, 1, 1, "c.fp_avgpool_fusion_s", "pad_right"], [31, 1, 1, "c.fp_avgpool_fusion_s", "pad_top"], [31, 1, 1, "c.fp_avgpool_fusion_s", "stride_h"], [31, 1, 1, "c.fp_avgpool_fusion_s", "stride_w"], [31, 1, 1, "c.fp_avgpool_fusion_s", "win_h"], [31, 1, 1, "c.fp_avgpool_fusion_s", "win_w"]], "fp_avgpoolinggrad_p": [[32, 1, 1, "c.fp_avgpoolinggrad_p", "batch"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "channel"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "end_idx"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "input"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "input_h"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "input_w"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "output"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "output_h"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "output_w"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "pad_l"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "pad_u"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "start_idx"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "stride_h"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "stride_w"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "window_h"], [32, 1, 1, "c.fp_avgpoolinggrad_p", "window_w"]], "fp_avgpoolinggrad_s": [[32, 1, 1, "c.fp_avgpoolinggrad_s", "batch"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "channel"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "core_mask"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "input"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "input_h"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "input_w"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "output"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "output_h"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "output_w"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "pad_l"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "pad_u"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "stride_h"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "stride_w"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "window_h"], [32, 1, 1, "c.fp_avgpoolinggrad_s", "window_w"]], "fp_batchnorm_p": [[33, 1, 1, "c.fp_batchnorm_p", "channel"], [33, 1, 1, "c.fp_batchnorm_p", "epsilon"], [33, 1, 1, "c.fp_batchnorm_p", "input"], [33, 1, 1, "c.fp_batchnorm_p", "mean"], [33, 1, 1, "c.fp_batchnorm_p", "output"], [33, 1, 1, "c.fp_batchnorm_p", "unit"], [33, 1, 1, "c.fp_batchnorm_p", "variance"]], "fp_batchnorm_s": [[33, 1, 1, "c.fp_batchnorm_s", "channel"], [33, 1, 1, "c.fp_batchnorm_s", "core_mask"], [33, 1, 1, "c.fp_batchnorm_s", "epsilon"], [33, 1, 1, "c.fp_batchnorm_s", "input"], [33, 1, 1, "c.fp_batchnorm_s", "mean"], [33, 1, 1, "c.fp_batchnorm_s", "output"], [33, 1, 1, "c.fp_batchnorm_s", "unit"], [33, 1, 1, "c.fp_batchnorm_s", "variance"]], "fp_batchnormgrad_p": [[34, 1, 1, "c.fp_batchnormgrad_p", "batch"], [34, 1, 1, "c.fp_batchnormgrad_p", "channel"], [34, 1, 1, "c.fp_batchnormgrad_p", "dbias"], [34, 1, 1, "c.fp_batchnormgrad_p", "dscale"], [34, 1, 1, "c.fp_batchnormgrad_p", "dx"], [34, 1, 1, "c.fp_batchnormgrad_p", "dy"], [34, 1, 1, "c.fp_batchnormgrad_p", "invar"], [34, 1, 1, "c.fp_batchnormgrad_p", "is_train"], [34, 1, 1, "c.fp_batchnormgrad_p", "mean"], [34, 1, 1, "c.fp_batchnormgrad_p", "scale"], [34, 1, 1, "c.fp_batchnormgrad_p", "x"]], "fp_batchnormgrad_s": [[34, 1, 1, "c.fp_batchnormgrad_s", "batch"], [34, 1, 1, "c.fp_batchnormgrad_s", "channel"], [34, 1, 1, "c.fp_batchnormgrad_s", "core_mask"], [34, 1, 1, "c.fp_batchnormgrad_s", "dbias"], [34, 1, 1, "c.fp_batchnormgrad_s", "dscale"], [34, 1, 1, "c.fp_batchnormgrad_s", "dx"], [34, 1, 1, "c.fp_batchnormgrad_s", "dy"], [34, 1, 1, "c.fp_batchnormgrad_s", "invar"], [34, 1, 1, "c.fp_batchnormgrad_s", "is_train"], [34, 1, 1, "c.fp_batchnormgrad_s", "mean"], [34, 1, 1, "c.fp_batchnormgrad_s", "scale"], [34, 1, 1, "c.fp_batchnormgrad_s", "x"]], "fp_batchtospace_p": [[35, 1, 1, "c.fp_batchtospace_p", "block_size"], [35, 1, 1, "c.fp_batchtospace_p", "crops"], [35, 1, 1, "c.fp_batchtospace_p", "data_size"], [35, 1, 1, "c.fp_batchtospace_p", "input"], [35, 1, 1, "c.fp_batchtospace_p", "input_shape"], [35, 1, 1, "c.fp_batchtospace_p", "output"]], "fp_batchtospace_s": [[35, 1, 1, "c.fp_batchtospace_s", "block_size"], [35, 1, 1, "c.fp_batchtospace_s", "core_mask"], [35, 1, 1, "c.fp_batchtospace_s", "crops"], [35, 1, 1, "c.fp_batchtospace_s", "data_size"], [35, 1, 1, "c.fp_batchtospace_s", "input"], [35, 1, 1, "c.fp_batchtospace_s", "input_shape"], [35, 1, 1, "c.fp_batchtospace_s", "output"]], "fp_batchtospacend_p": [[36, 1, 1, "c.fp_batchtospacend_p", "block_size"], [36, 1, 1, "c.fp_batchtospacend_p", "crops"], [36, 1, 1, "c.fp_batchtospacend_p", "data_size"], [36, 1, 1, "c.fp_batchtospacend_p", "input"], [36, 1, 1, "c.fp_batchtospacend_p", "input_shape"], [36, 1, 1, "c.fp_batchtospacend_p", "output"]], "fp_batchtospacend_s": [[36, 1, 1, "c.fp_batchtospacend_s", "block_size"], [36, 1, 1, "c.fp_batchtospacend_s", "core_mask"], [36, 1, 1, "c.fp_batchtospacend_s", "crops"], [36, 1, 1, "c.fp_batchtospacend_s", "data_size"], [36, 1, 1, "c.fp_batchtospacend_s", "input"], [36, 1, 1, "c.fp_batchtospacend_s", "input_shape"], [36, 1, 1, "c.fp_batchtospacend_s", "output"]], "fp_biasadd_p": [[37, 1, 1, "c.fp_biasadd_p", "data_format"], [37, 1, 1, "c.fp_biasadd_p", "dims"], [37, 1, 1, "c.fp_biasadd_p", "input_bias"], [37, 1, 1, "c.fp_biasadd_p", "input_x"], [37, 1, 1, "c.fp_biasadd_p", "length"], [37, 1, 1, "c.fp_biasadd_p", "output"], [37, 1, 1, "c.fp_biasadd_p", "shape_size"]], "fp_biasadd_s": [[37, 1, 1, "c.fp_biasadd_s", "core_mask"], [37, 1, 1, "c.fp_biasadd_s", "data_format"], [37, 1, 1, "c.fp_biasadd_s", "dims"], [37, 1, 1, "c.fp_biasadd_s", "input_bias"], [37, 1, 1, "c.fp_biasadd_s", "input_x"], [37, 1, 1, "c.fp_biasadd_s", "length"], [37, 1, 1, "c.fp_biasadd_s", "output"], [37, 1, 1, "c.fp_biasadd_s", "shape_size"]], "fp_biasaddgrad_p": [[38, 1, 1, "c.fp_biasaddgrad_p", "dbias"], [38, 1, 1, "c.fp_biasaddgrad_p", "dy"], [38, 1, 1, "c.fp_biasaddgrad_p", "dy_dims"], [38, 1, 1, "c.fp_biasaddgrad_p", "shape_size"]], "fp_biasaddgrad_s": [[38, 1, 1, "c.fp_biasaddgrad_s", "core_mask"], [38, 1, 1, "c.fp_biasaddgrad_s", "dbias"], [38, 1, 1, "c.fp_biasaddgrad_s", "dy"], [38, 1, 1, "c.fp_biasaddgrad_s", "dy_dims"], [38, 1, 1, "c.fp_biasaddgrad_s", "shape_size"]], "fp_binarycrossentropy_p": [[39, 1, 1, "c.fp_binarycrossentropy_p", "input_size"], [39, 1, 1, "c.fp_binarycrossentropy_p", "input_x"], [39, 1, 1, "c.fp_binarycrossentropy_p", "input_y"], [39, 1, 1, "c.fp_binarycrossentropy_p", "loss"], [39, 1, 1, "c.fp_binarycrossentropy_p", "reduction"], [39, 1, 1, "c.fp_binarycrossentropy_p", "tmp_loss"], [39, 1, 1, "c.fp_binarycrossentropy_p", "weight"], [39, 1, 1, "c.fp_binarycrossentropy_p", "weight_defined"]], "fp_binarycrossentropy_s": [[39, 1, 1, "c.fp_binarycrossentropy_s", "core_mask"], [39, 1, 1, "c.fp_binarycrossentropy_s", "input_size"], [39, 1, 1, "c.fp_binarycrossentropy_s", "input_x"], [39, 1, 1, "c.fp_binarycrossentropy_s", "input_y"], [39, 1, 1, "c.fp_binarycrossentropy_s", "loss"], [39, 1, 1, "c.fp_binarycrossentropy_s", "reduction"], [39, 1, 1, "c.fp_binarycrossentropy_s", "tmp_loss"], [39, 1, 1, "c.fp_binarycrossentropy_s", "weight"], [39, 1, 1, "c.fp_binarycrossentropy_s", "weight_defined"]], "fp_binarycrossentropygrad_p": [[40, 1, 1, "c.fp_binarycrossentropygrad_p", "dloss"], [40, 1, 1, "c.fp_binarycrossentropygrad_p", "dx"], [40, 1, 1, "c.fp_binarycrossentropygrad_p", "input_size"], [40, 1, 1, "c.fp_binarycrossentropygrad_p", "input_x"], [40, 1, 1, "c.fp_binarycrossentropygrad_p", "input_y"], [40, 1, 1, "c.fp_binarycrossentropygrad_p", "reduction"], [40, 1, 1, "c.fp_binarycrossentropygrad_p", "weight"], [40, 1, 1, "c.fp_binarycrossentropygrad_p", "weight_defined"]], "fp_binarycrossentropygrad_s": [[40, 1, 1, "c.fp_binarycrossentropygrad_s", "core_mask"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "dloss"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "dx"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "input_size"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "input_x"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "input_y"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "reduction"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "weight"], [40, 1, 1, "c.fp_binarycrossentropygrad_s", "weight_defined"]], "fp_broadcastto_p": [[41, 1, 1, "c.fp_broadcastto_p", "data_size"], [41, 1, 1, "c.fp_broadcastto_p", "input"], [41, 1, 1, "c.fp_broadcastto_p", "input_shape"], [41, 1, 1, "c.fp_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.fp_broadcastto_p", "output"], [41, 1, 1, "c.fp_broadcastto_p", "output_shape"], [41, 1, 1, "c.fp_broadcastto_p", "output_shape_size"]], "fp_broadcastto_s": [[41, 1, 1, "c.fp_broadcastto_s", "core_mask"], [41, 1, 1, "c.fp_broadcastto_s", "data_size"], [41, 1, 1, "c.fp_broadcastto_s", "input"], [41, 1, 1, "c.fp_broadcastto_s", "input_shape"], [41, 1, 1, "c.fp_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.fp_broadcastto_s", "output"], [41, 1, 1, "c.fp_broadcastto_s", "output_shape"], [41, 1, 1, "c.fp_broadcastto_s", "output_shape_size"]], "fp_ceil_p": [[43, 1, 1, "c.fp_ceil_p", "input_size"], [43, 1, 1, "c.fp_ceil_p", "input_x"], [43, 1, 1, "c.fp_ceil_p", "output"]], "fp_ceil_s": [[43, 1, 1, "c.fp_ceil_s", "core_mask"], [43, 1, 1, "c.fp_ceil_s", "input_size"], [43, 1, 1, "c.fp_ceil_s", "input_x"], [43, 1, 1, "c.fp_ceil_s", "output"]], "fp_celu_p": [[12, 1, 1, "c.fp_celu_p", "Input0"], [12, 1, 1, "c.fp_celu_p", "alpha"], [12, 1, 1, "c.fp_celu_p", "length"], [12, 1, 1, "c.fp_celu_p", "output"]], "fp_celu_s": [[12, 1, 1, "c.fp_celu_s", "Input0"], [12, 1, 1, "c.fp_celu_s", "alpha"], [12, 1, 1, "c.fp_celu_s", "core_mask"], [12, 1, 1, "c.fp_celu_s", "length"], [12, 1, 1, "c.fp_celu_s", "output"]], "fp_clip_p": [[12, 1, 1, "c.fp_clip_p", "Input0"], [12, 1, 1, "c.fp_clip_p", "length"], [12, 1, 1, "c.fp_clip_p", "max_val"], [12, 1, 1, "c.fp_clip_p", "min_val"], [12, 1, 1, "c.fp_clip_p", "output"]], "fp_clip_s": [[12, 1, 1, "c.fp_clip_s", "Input0"], [12, 1, 1, "c.fp_clip_s", "core_mask"], [12, 1, 1, "c.fp_clip_s", "length"], [12, 1, 1, "c.fp_clip_s", "max_val"], [12, 1, 1, "c.fp_clip_s", "min_val"], [12, 1, 1, "c.fp_clip_s", "output"]], "fp_concat_p": [[45, 1, 1, "c.fp_concat_p", "axis"], [45, 1, 1, "c.fp_concat_p", "input_ndim"], [45, 1, 1, "c.fp_concat_p", "input_shapes"], [45, 1, 1, "c.fp_concat_p", "inputs"], [45, 1, 1, "c.fp_concat_p", "num_inputs"], [45, 1, 1, "c.fp_concat_p", "output"]], "fp_concat_s": [[45, 1, 1, "c.fp_concat_s", "axis"], [45, 1, 1, "c.fp_concat_s", "core_mask"], [45, 1, 1, "c.fp_concat_s", "input_ndim"], [45, 1, 1, "c.fp_concat_s", "input_shapes"], [45, 1, 1, "c.fp_concat_s", "inputs"], [45, 1, 1, "c.fp_concat_s", "num_inputs"], [45, 1, 1, "c.fp_concat_s", "output"]], "fp_constant_of_shape_p": [[46, 1, 1, "c.fp_constant_of_shape_p", "end"], [46, 1, 1, "c.fp_constant_of_shape_p", "output"], [46, 1, 1, "c.fp_constant_of_shape_p", "start"], [46, 1, 1, "c.fp_constant_of_shape_p", "value"]], "fp_constant_of_shape_s": [[46, 1, 1, "c.fp_constant_of_shape_s", "core_mask"], [46, 1, 1, "c.fp_constant_of_shape_s", "end"], [46, 1, 1, "c.fp_constant_of_shape_s", "output"], [46, 1, 1, "c.fp_constant_of_shape_s", "start"], [46, 1, 1, "c.fp_constant_of_shape_s", "value"]], "fp_conv2d_p": [[47, 1, 1, "c.fp_conv2d_p", "bias"], [47, 1, 1, "c.fp_conv2d_p", "conv_param"], [47, 1, 1, "c.fp_conv2d_p", "core_mask"], [47, 1, 1, "c.fp_conv2d_p", "input_w"], [47, 1, 1, "c.fp_conv2d_p", "input_x"], [47, 1, 1, "c.fp_conv2d_p", "out_y"]], "fp_conv2d_s": [[47, 1, 1, "c.fp_conv2d_s", "bias"], [47, 1, 1, "c.fp_conv2d_s", "conv_param"], [47, 1, 1, "c.fp_conv2d_s", "core_mask"], [47, 1, 1, "c.fp_conv2d_s", "input_w"], [47, 1, 1, "c.fp_conv2d_s", "input_x"], [47, 1, 1, "c.fp_conv2d_s", "out_y"]], "fp_conv2dbackpropfilterfusion_p": [[49, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "conv_param"], [49, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "dw"], [49, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "dy"], [49, 1, 1, "c.fp_conv2dbackpropfilterfusion_p", "x"]], "fp_conv2dbackpropfilterfusion_s": [[49, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "conv_param"], [49, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "core_mask"], [49, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "dw"], [49, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "dy"], [49, 1, 1, "c.fp_conv2dbackpropfilterfusion_s", "x"]], "fp_conv2dbackpropinputfusion_p": [[50, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "conv_param"], [50, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "dx"], [50, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "dy"], [50, 1, 1, "c.fp_conv2dbackpropinputfusion_p", "w"]], "fp_conv2dbackpropinputfusion_s": [[50, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "conv_param"], [50, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "core_mask"], [50, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "dx"], [50, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "dy"], [50, 1, 1, "c.fp_conv2dbackpropinputfusion_s", "w"]], "fp_convtranspose_p": [[48, 1, 1, "c.fp_convtranspose_p", "bias"], [48, 1, 1, "c.fp_convtranspose_p", "conv_param"], [48, 1, 1, "c.fp_convtranspose_p", "core_mask"], [48, 1, 1, "c.fp_convtranspose_p", "input_w"], [48, 1, 1, "c.fp_convtranspose_p", "input_x"], [48, 1, 1, "c.fp_convtranspose_p", "out_y"]], "fp_convtranspose_s": [[48, 1, 1, "c.fp_convtranspose_s", "bias"], [48, 1, 1, "c.fp_convtranspose_s", "conv_param"], [48, 1, 1, "c.fp_convtranspose_s", "core_mask"], [48, 1, 1, "c.fp_convtranspose_s", "input_w"], [48, 1, 1, "c.fp_convtranspose_s", "input_x"], [48, 1, 1, "c.fp_convtranspose_s", "out_y"]], "fp_cos_p": [[51, 1, 1, "c.fp_cos_p", "dst_data"], [51, 1, 1, "c.fp_cos_p", "length"], [51, 1, 1, "c.fp_cos_p", "src_data"]], "fp_cos_s": [[51, 1, 1, "c.fp_cos_s", "core_mask"], [51, 1, 1, "c.fp_cos_s", "dst_data"], [51, 1, 1, "c.fp_cos_s", "length"], [51, 1, 1, "c.fp_cos_s", "src_data"]], "fp_crop_and_resize_anycore": [[53, 1, 1, "c.fp_crop_and_resize_anycore", "box_idx"], [53, 1, 1, "c.fp_crop_and_resize_anycore", "boxes"], [53, 1, 1, "c.fp_crop_and_resize_anycore", "core_mask"], [53, 1, 1, "c.fp_crop_and_resize_anycore", "dst"], [53, 1, 1, "c.fp_crop_and_resize_anycore", "extrapolation_value"], [53, 1, 1, "c.fp_crop_and_resize_anycore", "param"], [53, 1, 1, "c.fp_crop_and_resize_anycore", "src"]], "fp_cumsum_p": [[54, 1, 1, "c.fp_cumsum_p", "axis_dim"], [54, 1, 1, "c.fp_cumsum_p", "exclusive"], [54, 1, 1, "c.fp_cumsum_p", "inner_dim"], [54, 1, 1, "c.fp_cumsum_p", "input"], [54, 1, 1, "c.fp_cumsum_p", "out_dim"], [54, 1, 1, "c.fp_cumsum_p", "output"]], "fp_cumsum_s": [[54, 1, 1, "c.fp_cumsum_s", "axis_dim"], [54, 1, 1, "c.fp_cumsum_s", "core_mask"], [54, 1, 1, "c.fp_cumsum_s", "exclusive"], [54, 1, 1, "c.fp_cumsum_s", "inner_dim"], [54, 1, 1, "c.fp_cumsum_s", "input"], [54, 1, 1, "c.fp_cumsum_s", "out_dim"], [54, 1, 1, "c.fp_cumsum_s", "output"]], "fp_deconvgradfilter_p": [[58, 1, 1, "c.fp_deconvgradfilter_p", "dw_data"], [58, 1, 1, "c.fp_deconvgradfilter_p", "dy_data"], [58, 1, 1, "c.fp_deconvgradfilter_p", "param"], [58, 1, 1, "c.fp_deconvgradfilter_p", "x_data"]], "fp_deconvgradfilter_s": [[58, 1, 1, "c.fp_deconvgradfilter_s", "core_mask"], [58, 1, 1, "c.fp_deconvgradfilter_s", "dw_data"], [58, 1, 1, "c.fp_deconvgradfilter_s", "dy_data"], [58, 1, 1, "c.fp_deconvgradfilter_s", "param"], [58, 1, 1, "c.fp_deconvgradfilter_s", "x_data"]], "fp_depthtospace_p": [[59, 1, 1, "c.fp_depthtospace_p", "block_size"], [59, 1, 1, "c.fp_depthtospace_p", "data_size"], [59, 1, 1, "c.fp_depthtospace_p", "in_shape"], [59, 1, 1, "c.fp_depthtospace_p", "input"], [59, 1, 1, "c.fp_depthtospace_p", "output"]], "fp_depthtospace_s": [[59, 1, 1, "c.fp_depthtospace_s", "block_size"], [59, 1, 1, "c.fp_depthtospace_s", "core_mask"], [59, 1, 1, "c.fp_depthtospace_s", "data_size"], [59, 1, 1, "c.fp_depthtospace_s", "in_shape"], [59, 1, 1, "c.fp_depthtospace_s", "input"], [59, 1, 1, "c.fp_depthtospace_s", "output"]], "fp_detection_post_process_p": [[60, 1, 1, "c.fp_detection_post_process_p", "anchors"], [60, 1, 1, "c.fp_detection_post_process_p", "input_boxes"], [60, 1, 1, "c.fp_detection_post_process_p", "input_scores"], [60, 1, 1, "c.fp_detection_post_process_p", "output_boxes"], [60, 1, 1, "c.fp_detection_post_process_p", "output_classes"], [60, 1, 1, "c.fp_detection_post_process_p", "output_num"], [60, 1, 1, "c.fp_detection_post_process_p", "output_scores"], [60, 1, 1, "c.fp_detection_post_process_p", "param"]], "fp_detection_post_process_s": [[60, 1, 1, "c.fp_detection_post_process_s", "anchors"], [60, 1, 1, "c.fp_detection_post_process_s", "core_mask"], [60, 1, 1, "c.fp_detection_post_process_s", "input_boxes"], [60, 1, 1, "c.fp_detection_post_process_s", "input_scores"], [60, 1, 1, "c.fp_detection_post_process_s", "output_boxes"], [60, 1, 1, "c.fp_detection_post_process_s", "output_classes"], [60, 1, 1, "c.fp_detection_post_process_s", "output_num"], [60, 1, 1, "c.fp_detection_post_process_s", "output_scores"], [60, 1, 1, "c.fp_detection_post_process_s", "param"]], "fp_div_fusion_p": [[61, 1, 1, "c.fp_div_fusion_p", "input0"], [61, 1, 1, "c.fp_div_fusion_p", "input1"], [61, 1, 1, "c.fp_div_fusion_p", "length"], [61, 1, 1, "c.fp_div_fusion_p", "output"]], "fp_div_fusion_s": [[61, 1, 1, "c.fp_div_fusion_s", "core_mask"], [61, 1, 1, "c.fp_div_fusion_s", "input0"], [61, 1, 1, "c.fp_div_fusion_s", "input1"], [61, 1, 1, "c.fp_div_fusion_s", "length"], [61, 1, 1, "c.fp_div_fusion_s", "output"]], "fp_dropout_p": [[63, 1, 1, "c.fp_dropout_p", "input"], [63, 1, 1, "c.fp_dropout_p", "length"], [63, 1, 1, "c.fp_dropout_p", "mask"], [63, 1, 1, "c.fp_dropout_p", "output"], [63, 1, 1, "c.fp_dropout_p", "scale"]], "fp_dropout_s": [[63, 1, 1, "c.fp_dropout_s", "core_mask"], [63, 1, 1, "c.fp_dropout_s", "input"], [63, 1, 1, "c.fp_dropout_s", "length"], [63, 1, 1, "c.fp_dropout_s", "mask"], [63, 1, 1, "c.fp_dropout_s", "output"], [63, 1, 1, "c.fp_dropout_s", "scale"]], "fp_dropoutgrad_p": [[64, 1, 1, "c.fp_dropoutgrad_p", "input"], [64, 1, 1, "c.fp_dropoutgrad_p", "length"], [64, 1, 1, "c.fp_dropoutgrad_p", "mask"], [64, 1, 1, "c.fp_dropoutgrad_p", "output"], [64, 1, 1, "c.fp_dropoutgrad_p", "scale"]], "fp_dropoutgrad_s": [[64, 1, 1, "c.fp_dropoutgrad_s", "core_mask"], [64, 1, 1, "c.fp_dropoutgrad_s", "input"], [64, 1, 1, "c.fp_dropoutgrad_s", "length"], [64, 1, 1, "c.fp_dropoutgrad_s", "mask"], [64, 1, 1, "c.fp_dropoutgrad_s", "output"], [64, 1, 1, "c.fp_dropoutgrad_s", "scale"]], "fp_eltwise_p": [[67, 1, 1, "c.fp_eltwise_p", "Input0"], [67, 1, 1, "c.fp_eltwise_p", "Input1"], [67, 1, 1, "c.fp_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.fp_eltwise_p", "length"], [67, 1, 1, "c.fp_eltwise_p", "output"]], "fp_eltwise_s": [[67, 1, 1, "c.fp_eltwise_s", "Input0"], [67, 1, 1, "c.fp_eltwise_s", "Input1"], [67, 1, 1, "c.fp_eltwise_s", "core_mask"], [67, 1, 1, "c.fp_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.fp_eltwise_s", "length"], [67, 1, 1, "c.fp_eltwise_s", "output"]], "fp_elu_grad_p": [[13, 1, 1, "c.fp_elu_grad_p", "alpha"], [13, 1, 1, "c.fp_elu_grad_p", "dst"], [13, 1, 1, "c.fp_elu_grad_p", "length"], [13, 1, 1, "c.fp_elu_grad_p", "src0"], [13, 1, 1, "c.fp_elu_grad_p", "src1"]], "fp_elu_grad_s": [[13, 1, 1, "c.fp_elu_grad_s", "alpha"], [13, 1, 1, "c.fp_elu_grad_s", "core_mask"], [13, 1, 1, "c.fp_elu_grad_s", "dst"], [13, 1, 1, "c.fp_elu_grad_s", "length"], [13, 1, 1, "c.fp_elu_grad_s", "src0"], [13, 1, 1, "c.fp_elu_grad_s", "src1"]], "fp_elu_p": [[12, 1, 1, "c.fp_elu_p", "Input0"], [12, 1, 1, "c.fp_elu_p", "alpha"], [12, 1, 1, "c.fp_elu_p", "length"], [12, 1, 1, "c.fp_elu_p", "output"]], "fp_elu_s": [[12, 1, 1, "c.fp_elu_s", "Input0"], [12, 1, 1, "c.fp_elu_s", "alpha"], [12, 1, 1, "c.fp_elu_s", "core_mask"], [12, 1, 1, "c.fp_elu_s", "length"], [12, 1, 1, "c.fp_elu_s", "output"]], "fp_embeddinglookup_p": [[69, 1, 1, "c.fp_embeddinglookup_p", "Input0"], [69, 1, 1, "c.fp_embeddinglookup_p", "Input1"], [69, 1, 1, "c.fp_embeddinglookup_p", "length"], [69, 1, 1, "c.fp_embeddinglookup_p", "output"]], "fp_embeddinglookup_s": [[69, 1, 1, "c.fp_embeddinglookup_s", "core_mask"], [69, 1, 1, "c.fp_embeddinglookup_s", "ids"], [69, 1, 1, "c.fp_embeddinglookup_s", "ids_size_"], [69, 1, 1, "c.fp_embeddinglookup_s", "input_data"], [69, 1, 1, "c.fp_embeddinglookup_s", "is_regulated"], [69, 1, 1, "c.fp_embeddinglookup_s", "layer_num_"], [69, 1, 1, "c.fp_embeddinglookup_s", "layer_size_"], [69, 1, 1, "c.fp_embeddinglookup_s", "max_norm_"], [69, 1, 1, "c.fp_embeddinglookup_s", "output"]], "fp_equal_p": [[70, 1, 1, "c.fp_equal_p", "Input0"], [70, 1, 1, "c.fp_equal_p", "Input1"], [70, 1, 1, "c.fp_equal_p", "length"], [70, 1, 1, "c.fp_equal_p", "output"]], "fp_equal_s": [[70, 1, 1, "c.fp_equal_s", "Input0"], [70, 1, 1, "c.fp_equal_s", "Input1"], [70, 1, 1, "c.fp_equal_s", "core_mask"], [70, 1, 1, "c.fp_equal_s", "length"], [70, 1, 1, "c.fp_equal_s", "output"]], "fp_erf_p": [[71, 1, 1, "c.fp_erf_p", "input"], [71, 1, 1, "c.fp_erf_p", "length"], [71, 1, 1, "c.fp_erf_p", "output"]], "fp_erf_s": [[71, 1, 1, "c.fp_erf_s", "core_mask"], [71, 1, 1, "c.fp_erf_s", "input"], [71, 1, 1, "c.fp_erf_s", "length"], [71, 1, 1, "c.fp_erf_s", "output"]], "fp_expfusion_p": [[73, 1, 1, "c.fp_expfusion_p", "dst_data"], [73, 1, 1, "c.fp_expfusion_p", "in_scale"], [73, 1, 1, "c.fp_expfusion_p", "length"], [73, 1, 1, "c.fp_expfusion_p", "out_scale"], [73, 1, 1, "c.fp_expfusion_p", "scale"], [73, 1, 1, "c.fp_expfusion_p", "src_data"]], "fp_expfusion_s": [[73, 1, 1, "c.fp_expfusion_s", "core_mask"], [73, 1, 1, "c.fp_expfusion_s", "dst_data"], [73, 1, 1, "c.fp_expfusion_s", "in_scale"], [73, 1, 1, "c.fp_expfusion_s", "length"], [73, 1, 1, "c.fp_expfusion_s", "out_scale"], [73, 1, 1, "c.fp_expfusion_s", "scale"], [73, 1, 1, "c.fp_expfusion_s", "src_data"]], "fp_extract_features_p": [[55, 1, 1, "c.fp_extract_features_p", "num_strings"], [55, 1, 1, "c.fp_extract_features_p", "output_labels"], [55, 1, 1, "c.fp_extract_features_p", "output_weights"], [55, 1, 1, "c.fp_extract_features_p", "string_lengths"], [55, 1, 1, "c.fp_extract_features_p", "string_pointers"]], "fp_extract_features_s": [[55, 1, 1, "c.fp_extract_features_s", "core_mask"], [55, 1, 1, "c.fp_extract_features_s", "num_strings"], [55, 1, 1, "c.fp_extract_features_s", "output_labels"], [55, 1, 1, "c.fp_extract_features_s", "output_weights"], [55, 1, 1, "c.fp_extract_features_s", "string_lengths"], [55, 1, 1, "c.fp_extract_features_s", "string_pointers"]], "fp_fake_quant_with_min_max_vars_p": [[74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "length"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "max_val"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "min_val"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "output"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "quant_max"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "quant_min"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "src"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_p", "symmetric"]], "fp_fake_quant_with_min_max_vars_per_channel_p": [[75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "channel_num"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "length"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "max_val"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "min_val"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "output"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "quant_max"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "quant_min"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "src"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_p", "symmetric"]], "fp_fake_quant_with_min_max_vars_per_channel_s": [[75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "channel_num"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "core_mask"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "length"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "max_val"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "min_val"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "output"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "quant_max"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "quant_min"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "src"], [75, 1, 1, "c.fp_fake_quant_with_min_max_vars_per_channel_s", "symmetric"]], "fp_fake_quant_with_min_max_vars_s": [[74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "core_mask"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "length"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "max_val"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "min_val"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "output"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "quant_max"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "quant_min"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "src"], [74, 1, 1, "c.fp_fake_quant_with_min_max_vars_s", "symmetric"]], "fp_fill_p": [[78, 1, 1, "c.fp_fill_p", "output"], [78, 1, 1, "c.fp_fill_p", "param"], [78, 1, 1, "c.fp_fill_p", "value"]], "fp_fill_s": [[78, 1, 1, "c.fp_fill_s", "core_mask"], [78, 1, 1, "c.fp_fill_s", "output"], [78, 1, 1, "c.fp_fill_s", "param"], [78, 1, 1, "c.fp_fill_s", "value"]], "fp_flattengrad_p": [[81, 1, 1, "c.fp_flattengrad_p", "input"], [81, 1, 1, "c.fp_flattengrad_p", "origin_ndim"], [81, 1, 1, "c.fp_flattengrad_p", "origin_shape"], [81, 1, 1, "c.fp_flattengrad_p", "output"]], "fp_flattengrad_s": [[81, 1, 1, "c.fp_flattengrad_s", "core_mask"], [81, 1, 1, "c.fp_flattengrad_s", "input"], [81, 1, 1, "c.fp_flattengrad_s", "origin_ndim"], [81, 1, 1, "c.fp_flattengrad_s", "origin_shape"], [81, 1, 1, "c.fp_flattengrad_s", "output"]], "fp_floor_p": [[82, 1, 1, "c.fp_floor_p", "dst_data"], [82, 1, 1, "c.fp_floor_p", "length"], [82, 1, 1, "c.fp_floor_p", "src_data"]], "fp_floor_s": [[82, 1, 1, "c.fp_floor_s", "core_mask"], [82, 1, 1, "c.fp_floor_s", "dst_data"], [82, 1, 1, "c.fp_floor_s", "length"], [82, 1, 1, "c.fp_floor_s", "src_data"]], "fp_floordiv_p": [[83, 1, 1, "c.fp_floordiv_p", "dst_data"], [83, 1, 1, "c.fp_floordiv_p", "length"], [83, 1, 1, "c.fp_floordiv_p", "src_data0"], [83, 1, 1, "c.fp_floordiv_p", "src_data1"]], "fp_floordiv_s": [[83, 1, 1, "c.fp_floordiv_s", "core_mask"], [83, 1, 1, "c.fp_floordiv_s", "dst_data"], [83, 1, 1, "c.fp_floordiv_s", "length"], [83, 1, 1, "c.fp_floordiv_s", "src_data0"], [83, 1, 1, "c.fp_floordiv_s", "src_data1"]], "fp_floormod_p": [[84, 1, 1, "c.fp_floormod_p", "input0"], [84, 1, 1, "c.fp_floormod_p", "input1"], [84, 1, 1, "c.fp_floormod_p", "output"], [84, 1, 1, "c.fp_floormod_p", "size"]], "fp_floormod_s": [[84, 1, 1, "c.fp_floormod_s", "core_mask"], [84, 1, 1, "c.fp_floormod_s", "input0"], [84, 1, 1, "c.fp_floormod_s", "input1"], [84, 1, 1, "c.fp_floormod_s", "output"], [84, 1, 1, "c.fp_floormod_s", "size"]], "fp_formattranspose_p": [[85, 1, 1, "c.fp_formattranspose_p", "batch"], [85, 1, 1, "c.fp_formattranspose_p", "channel"], [85, 1, 1, "c.fp_formattranspose_p", "dst_data"], [85, 1, 1, "c.fp_formattranspose_p", "dst_format"], [85, 1, 1, "c.fp_formattranspose_p", "plane"], [85, 1, 1, "c.fp_formattranspose_p", "src_data"], [85, 1, 1, "c.fp_formattranspose_p", "src_format"]], "fp_formattranspose_s": [[85, 1, 1, "c.fp_formattranspose_s", "batch"], [85, 1, 1, "c.fp_formattranspose_s", "channel"], [85, 1, 1, "c.fp_formattranspose_s", "core_mask"], [85, 1, 1, "c.fp_formattranspose_s", "dst_data"], [85, 1, 1, "c.fp_formattranspose_s", "dst_format"], [85, 1, 1, "c.fp_formattranspose_s", "plane"], [85, 1, 1, "c.fp_formattranspose_s", "src_data"], [85, 1, 1, "c.fp_formattranspose_s", "src_format"]], "fp_fullconnection_p": [[86, 1, 1, "c.fp_fullconnection_p", "A"], [86, 1, 1, "c.fp_fullconnection_p", "B"], [86, 1, 1, "c.fp_fullconnection_p", "C"], [86, 1, 1, "c.fp_fullconnection_p", "K"], [86, 1, 1, "c.fp_fullconnection_p", "M"], [86, 1, 1, "c.fp_fullconnection_p", "N"], [86, 1, 1, "c.fp_fullconnection_p", "activation_type"], [86, 1, 1, "c.fp_fullconnection_p", "bias"]], "fp_fullconnection_s": [[86, 1, 1, "c.fp_fullconnection_s", "A"], [86, 1, 1, "c.fp_fullconnection_s", "B"], [86, 1, 1, "c.fp_fullconnection_s", "C"], [86, 1, 1, "c.fp_fullconnection_s", "K"], [86, 1, 1, "c.fp_fullconnection_s", "M"], [86, 1, 1, "c.fp_fullconnection_s", "N"], [86, 1, 1, "c.fp_fullconnection_s", "activation_type"], [86, 1, 1, "c.fp_fullconnection_s", "bias"], [86, 1, 1, "c.fp_fullconnection_s", "core_mask"]], "fp_fusedbatchnorm_p": [[87, 1, 1, "c.fp_fusedbatchnorm_p", "channel"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "epsilon"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "input"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "mean"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "offset"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "output"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "scale"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "unit"], [87, 1, 1, "c.fp_fusedbatchnorm_p", "variance"]], "fp_fusedbatchnorm_s": [[87, 1, 1, "c.fp_fusedbatchnorm_s", "channel"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "core_mask"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "epsilon"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "input"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "mean"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "offset"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "output"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "scale"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "unit"], [87, 1, 1, "c.fp_fusedbatchnorm_s", "variance"]], "fp_gather_nd_p": [[89, 1, 1, "c.fp_gather_nd_p", "indices"], [89, 1, 1, "c.fp_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.fp_gather_nd_p", "indices_shape"], [89, 1, 1, "c.fp_gather_nd_p", "input"], [89, 1, 1, "c.fp_gather_nd_p", "input_ndim"], [89, 1, 1, "c.fp_gather_nd_p", "input_shape"], [89, 1, 1, "c.fp_gather_nd_p", "output"]], "fp_gather_nd_s": [[89, 1, 1, "c.fp_gather_nd_s", "core_mask"], [89, 1, 1, "c.fp_gather_nd_s", "indices"], [89, 1, 1, "c.fp_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.fp_gather_nd_s", "indices_shape"], [89, 1, 1, "c.fp_gather_nd_s", "input"], [89, 1, 1, "c.fp_gather_nd_s", "input_ndim"], [89, 1, 1, "c.fp_gather_nd_s", "input_shape"], [89, 1, 1, "c.fp_gather_nd_s", "output"]], "fp_gather_p": [[88, 1, 1, "c.fp_gather_p", "axis"], [88, 1, 1, "c.fp_gather_p", "batch_dims"], [88, 1, 1, "c.fp_gather_p", "indices"], [88, 1, 1, "c.fp_gather_p", "indices_ndim"], [88, 1, 1, "c.fp_gather_p", "indices_shape"], [88, 1, 1, "c.fp_gather_p", "input"], [88, 1, 1, "c.fp_gather_p", "input_ndim"], [88, 1, 1, "c.fp_gather_p", "input_shape"], [88, 1, 1, "c.fp_gather_p", "output"]], "fp_gather_s": [[88, 1, 1, "c.fp_gather_s", "axis"], [88, 1, 1, "c.fp_gather_s", "batch_dims"], [88, 1, 1, "c.fp_gather_s", "core_mask"], [88, 1, 1, "c.fp_gather_s", "indices"], [88, 1, 1, "c.fp_gather_s", "indices_ndim"], [88, 1, 1, "c.fp_gather_s", "indices_shape"], [88, 1, 1, "c.fp_gather_s", "input"], [88, 1, 1, "c.fp_gather_s", "input_ndim"], [88, 1, 1, "c.fp_gather_s", "input_shape"], [88, 1, 1, "c.fp_gather_s", "output"]], "fp_gatherd_p": [[90, 1, 1, "c.fp_gatherd_p", "dim"], [90, 1, 1, "c.fp_gatherd_p", "index"], [90, 1, 1, "c.fp_gatherd_p", "index_shape"], [90, 1, 1, "c.fp_gatherd_p", "input_shape"], [90, 1, 1, "c.fp_gatherd_p", "input_shape_size"], [90, 1, 1, "c.fp_gatherd_p", "input_x"], [90, 1, 1, "c.fp_gatherd_p", "output"]], "fp_gatherd_s": [[90, 1, 1, "c.fp_gatherd_s", "core_mask"], [90, 1, 1, "c.fp_gatherd_s", "dim"], [90, 1, 1, "c.fp_gatherd_s", "index"], [90, 1, 1, "c.fp_gatherd_s", "index_shape"], [90, 1, 1, "c.fp_gatherd_s", "input_shape"], [90, 1, 1, "c.fp_gatherd_s", "input_shape_size"], [90, 1, 1, "c.fp_gatherd_s", "input_x"], [90, 1, 1, "c.fp_gatherd_s", "output"]], "fp_gelu_grad_p": [[13, 1, 1, "c.fp_gelu_grad_p", "dst"], [13, 1, 1, "c.fp_gelu_grad_p", "length"], [13, 1, 1, "c.fp_gelu_grad_p", "src0"], [13, 1, 1, "c.fp_gelu_grad_p", "src1"]], "fp_gelu_grad_s": [[13, 1, 1, "c.fp_gelu_grad_s", "core_mask"], [13, 1, 1, "c.fp_gelu_grad_s", "dst"], [13, 1, 1, "c.fp_gelu_grad_s", "length"], [13, 1, 1, "c.fp_gelu_grad_s", "src0"], [13, 1, 1, "c.fp_gelu_grad_s", "src1"]], "fp_gelu_p": [[12, 1, 1, "c.fp_gelu_p", "Input0"], [12, 1, 1, "c.fp_gelu_p", "approximate"], [12, 1, 1, "c.fp_gelu_p", "length"], [12, 1, 1, "c.fp_gelu_p", "output"]], "fp_gelu_s": [[12, 1, 1, "c.fp_gelu_s", "Input0"], [12, 1, 1, "c.fp_gelu_s", "approximate"], [12, 1, 1, "c.fp_gelu_s", "core_mask"], [12, 1, 1, "c.fp_gelu_s", "length"], [12, 1, 1, "c.fp_gelu_s", "output"]], "fp_glu_p": [[91, 1, 1, "c.fp_glu_p", "in_data"], [91, 1, 1, "c.fp_glu_p", "input_shape"], [91, 1, 1, "c.fp_glu_p", "len"], [91, 1, 1, "c.fp_glu_p", "ndim"], [91, 1, 1, "c.fp_glu_p", "num_split"], [91, 1, 1, "c.fp_glu_p", "out_data"], [91, 1, 1, "c.fp_glu_p", "split_data"], [91, 1, 1, "c.fp_glu_p", "split_dim"], [91, 1, 1, "c.fp_glu_p", "split_sizes"], [91, 1, 1, "c.fp_glu_p", "strides"]], "fp_glu_s": [[91, 1, 1, "c.fp_glu_s", "core_mask"], [91, 1, 1, "c.fp_glu_s", "in_data"], [91, 1, 1, "c.fp_glu_s", "input_shape"], [91, 1, 1, "c.fp_glu_s", "len"], [91, 1, 1, "c.fp_glu_s", "ndim"], [91, 1, 1, "c.fp_glu_s", "num_split"], [91, 1, 1, "c.fp_glu_s", "out_data"], [91, 1, 1, "c.fp_glu_s", "split_data"], [91, 1, 1, "c.fp_glu_s", "split_dim"], [91, 1, 1, "c.fp_glu_s", "split_sizes"], [91, 1, 1, "c.fp_glu_s", "strides"]], "fp_graddiv1l_p": [[62, 1, 1, "c.fp_graddiv1l_p", "dx1"], [62, 1, 1, "c.fp_graddiv1l_p", "dx2"], [62, 1, 1, "c.fp_graddiv1l_p", "dy"], [62, 1, 1, "c.fp_graddiv1l_p", "indices"], [62, 1, 1, "c.fp_graddiv1l_p", "large_multiples"], [62, 1, 1, "c.fp_graddiv1l_p", "large_shape"], [62, 1, 1, "c.fp_graddiv1l_p", "large_strides"], [62, 1, 1, "c.fp_graddiv1l_p", "ndims"], [62, 1, 1, "c.fp_graddiv1l_p", "out_shape"], [62, 1, 1, "c.fp_graddiv1l_p", "out_strides"], [62, 1, 1, "c.fp_graddiv1l_p", "small_multiples"], [62, 1, 1, "c.fp_graddiv1l_p", "small_shape"], [62, 1, 1, "c.fp_graddiv1l_p", "small_strides"], [62, 1, 1, "c.fp_graddiv1l_p", "tile_data0"], [62, 1, 1, "c.fp_graddiv1l_p", "tile_data1"], [62, 1, 1, "c.fp_graddiv1l_p", "tile_data2"], [62, 1, 1, "c.fp_graddiv1l_p", "x1"], [62, 1, 1, "c.fp_graddiv1l_p", "x2"]], "fp_graddiv1l_s": [[62, 1, 1, "c.fp_graddiv1l_s", "core_mask"], [62, 1, 1, "c.fp_graddiv1l_s", "dx1"], [62, 1, 1, "c.fp_graddiv1l_s", "dx2"], [62, 1, 1, "c.fp_graddiv1l_s", "dy"], [62, 1, 1, "c.fp_graddiv1l_s", "indices"], [62, 1, 1, "c.fp_graddiv1l_s", "large_multiples"], [62, 1, 1, "c.fp_graddiv1l_s", "large_shape"], [62, 1, 1, "c.fp_graddiv1l_s", "large_strides"], [62, 1, 1, "c.fp_graddiv1l_s", "ndims"], [62, 1, 1, "c.fp_graddiv1l_s", "out_shape"], [62, 1, 1, "c.fp_graddiv1l_s", "out_strides"], [62, 1, 1, "c.fp_graddiv1l_s", "small_multiples"], [62, 1, 1, "c.fp_graddiv1l_s", "small_shape"], [62, 1, 1, "c.fp_graddiv1l_s", "small_strides"], [62, 1, 1, "c.fp_graddiv1l_s", "tile_data0"], [62, 1, 1, "c.fp_graddiv1l_s", "tile_data1"], [62, 1, 1, "c.fp_graddiv1l_s", "tile_data2"], [62, 1, 1, "c.fp_graddiv1l_s", "x1"], [62, 1, 1, "c.fp_graddiv1l_s", "x2"]], "fp_graddiv2l_p": [[62, 1, 1, "c.fp_graddiv2l_p", "dx1"], [62, 1, 1, "c.fp_graddiv2l_p", "dx2"], [62, 1, 1, "c.fp_graddiv2l_p", "dy"], [62, 1, 1, "c.fp_graddiv2l_p", "indices"], [62, 1, 1, "c.fp_graddiv2l_p", "large_multiples"], [62, 1, 1, "c.fp_graddiv2l_p", "large_shape"], [62, 1, 1, "c.fp_graddiv2l_p", "large_strides"], [62, 1, 1, "c.fp_graddiv2l_p", "ndims"], [62, 1, 1, "c.fp_graddiv2l_p", "out_shape"], [62, 1, 1, "c.fp_graddiv2l_p", "out_strides"], [62, 1, 1, "c.fp_graddiv2l_p", "small_multiples"], [62, 1, 1, "c.fp_graddiv2l_p", "small_shape"], [62, 1, 1, "c.fp_graddiv2l_p", "small_strides"], [62, 1, 1, "c.fp_graddiv2l_p", "tile_data0"], [62, 1, 1, "c.fp_graddiv2l_p", "tile_data1"], [62, 1, 1, "c.fp_graddiv2l_p", "tile_data2"], [62, 1, 1, "c.fp_graddiv2l_p", "x1"], [62, 1, 1, "c.fp_graddiv2l_p", "x2"]], "fp_graddiv2l_s": [[62, 1, 1, "c.fp_graddiv2l_s", "core_mask"], [62, 1, 1, "c.fp_graddiv2l_s", "dx1"], [62, 1, 1, "c.fp_graddiv2l_s", "dx2"], [62, 1, 1, "c.fp_graddiv2l_s", "dy"], [62, 1, 1, "c.fp_graddiv2l_s", "indices"], [62, 1, 1, "c.fp_graddiv2l_s", "large_multiples"], [62, 1, 1, "c.fp_graddiv2l_s", "large_shape"], [62, 1, 1, "c.fp_graddiv2l_s", "large_strides"], [62, 1, 1, "c.fp_graddiv2l_s", "ndims"], [62, 1, 1, "c.fp_graddiv2l_s", "out_shape"], [62, 1, 1, "c.fp_graddiv2l_s", "out_strides"], [62, 1, 1, "c.fp_graddiv2l_s", "small_multiples"], [62, 1, 1, "c.fp_graddiv2l_s", "small_shape"], [62, 1, 1, "c.fp_graddiv2l_s", "small_strides"], [62, 1, 1, "c.fp_graddiv2l_s", "tile_data0"], [62, 1, 1, "c.fp_graddiv2l_s", "tile_data1"], [62, 1, 1, "c.fp_graddiv2l_s", "tile_data2"], [62, 1, 1, "c.fp_graddiv2l_s", "x1"], [62, 1, 1, "c.fp_graddiv2l_s", "x2"]], "fp_graddiv_p": [[62, 1, 1, "c.fp_graddiv_p", "dx1"], [62, 1, 1, "c.fp_graddiv_p", "dx2"], [62, 1, 1, "c.fp_graddiv_p", "dy"], [62, 1, 1, "c.fp_graddiv_p", "indices"], [62, 1, 1, "c.fp_graddiv_p", "large_multiples"], [62, 1, 1, "c.fp_graddiv_p", "large_shape"], [62, 1, 1, "c.fp_graddiv_p", "large_strides"], [62, 1, 1, "c.fp_graddiv_p", "ndims"], [62, 1, 1, "c.fp_graddiv_p", "out_shape"], [62, 1, 1, "c.fp_graddiv_p", "out_strides"], [62, 1, 1, "c.fp_graddiv_p", "small_multiples"], [62, 1, 1, "c.fp_graddiv_p", "small_shape"], [62, 1, 1, "c.fp_graddiv_p", "small_strides"], [62, 1, 1, "c.fp_graddiv_p", "tile_data0"], [62, 1, 1, "c.fp_graddiv_p", "tile_data1"], [62, 1, 1, "c.fp_graddiv_p", "tile_data2"], [62, 1, 1, "c.fp_graddiv_p", "x1"], [62, 1, 1, "c.fp_graddiv_p", "x2"]], "fp_graddiv_s": [[62, 1, 1, "c.fp_graddiv_s", "core_mask"], [62, 1, 1, "c.fp_graddiv_s", "dx1"], [62, 1, 1, "c.fp_graddiv_s", "dx2"], [62, 1, 1, "c.fp_graddiv_s", "dy"], [62, 1, 1, "c.fp_graddiv_s", "indices"], [62, 1, 1, "c.fp_graddiv_s", "large_multiples"], [62, 1, 1, "c.fp_graddiv_s", "large_shape"], [62, 1, 1, "c.fp_graddiv_s", "large_strides"], [62, 1, 1, "c.fp_graddiv_s", "ndims"], [62, 1, 1, "c.fp_graddiv_s", "out_shape"], [62, 1, 1, "c.fp_graddiv_s", "out_strides"], [62, 1, 1, "c.fp_graddiv_s", "small_multiples"], [62, 1, 1, "c.fp_graddiv_s", "small_shape"], [62, 1, 1, "c.fp_graddiv_s", "small_strides"], [62, 1, 1, "c.fp_graddiv_s", "tile_data0"], [62, 1, 1, "c.fp_graddiv_s", "tile_data1"], [62, 1, 1, "c.fp_graddiv_s", "tile_data2"], [62, 1, 1, "c.fp_graddiv_s", "x1"], [62, 1, 1, "c.fp_graddiv_s", "x2"]], "fp_gradmul1l_p": [[131, 1, 1, "c.fp_gradmul1l_p", "dx1"], [131, 1, 1, "c.fp_gradmul1l_p", "dx2"], [131, 1, 1, "c.fp_gradmul1l_p", "dy"], [131, 1, 1, "c.fp_gradmul1l_p", "indices"], [131, 1, 1, "c.fp_gradmul1l_p", "large_multiples"], [131, 1, 1, "c.fp_gradmul1l_p", "large_shape"], [131, 1, 1, "c.fp_gradmul1l_p", "large_strides"], [131, 1, 1, "c.fp_gradmul1l_p", "ndims"], [131, 1, 1, "c.fp_gradmul1l_p", "out_shape"], [131, 1, 1, "c.fp_gradmul1l_p", "out_strides"], [131, 1, 1, "c.fp_gradmul1l_p", "small_multiples"], [131, 1, 1, "c.fp_gradmul1l_p", "small_shape"], [131, 1, 1, "c.fp_gradmul1l_p", "small_strides"], [131, 1, 1, "c.fp_gradmul1l_p", "tile_data0"], [131, 1, 1, "c.fp_gradmul1l_p", "tile_data1"], [131, 1, 1, "c.fp_gradmul1l_p", "x1"], [131, 1, 1, "c.fp_gradmul1l_p", "x2"]], "fp_gradmul1l_s": [[131, 1, 1, "c.fp_gradmul1l_s", "core_mask"], [131, 1, 1, "c.fp_gradmul1l_s", "dx1"], [131, 1, 1, "c.fp_gradmul1l_s", "dx2"], [131, 1, 1, "c.fp_gradmul1l_s", "dy"], [131, 1, 1, "c.fp_gradmul1l_s", "indices"], [131, 1, 1, "c.fp_gradmul1l_s", "large_multiples"], [131, 1, 1, "c.fp_gradmul1l_s", "large_shape"], [131, 1, 1, "c.fp_gradmul1l_s", "large_strides"], [131, 1, 1, "c.fp_gradmul1l_s", "ndims"], [131, 1, 1, "c.fp_gradmul1l_s", "out_shape"], [131, 1, 1, "c.fp_gradmul1l_s", "out_strides"], [131, 1, 1, "c.fp_gradmul1l_s", "small_multiples"], [131, 1, 1, "c.fp_gradmul1l_s", "small_shape"], [131, 1, 1, "c.fp_gradmul1l_s", "small_strides"], [131, 1, 1, "c.fp_gradmul1l_s", "tile_data0"], [131, 1, 1, "c.fp_gradmul1l_s", "tile_data1"], [131, 1, 1, "c.fp_gradmul1l_s", "x1"], [131, 1, 1, "c.fp_gradmul1l_s", "x2"]], "fp_gradmul2l_p": [[131, 1, 1, "c.fp_gradmul2l_p", "dx1"], [131, 1, 1, "c.fp_gradmul2l_p", "dx2"], [131, 1, 1, "c.fp_gradmul2l_p", "dy"], [131, 1, 1, "c.fp_gradmul2l_p", "indices"], [131, 1, 1, "c.fp_gradmul2l_p", "large_multiples"], [131, 1, 1, "c.fp_gradmul2l_p", "large_shape"], [131, 1, 1, "c.fp_gradmul2l_p", "large_strides"], [131, 1, 1, "c.fp_gradmul2l_p", "ndims"], [131, 1, 1, "c.fp_gradmul2l_p", "out_shape"], [131, 1, 1, "c.fp_gradmul2l_p", "out_strides"], [131, 1, 1, "c.fp_gradmul2l_p", "small_multiples"], [131, 1, 1, "c.fp_gradmul2l_p", "small_shape"], [131, 1, 1, "c.fp_gradmul2l_p", "small_strides"], [131, 1, 1, "c.fp_gradmul2l_p", "tile_data0"], [131, 1, 1, "c.fp_gradmul2l_p", "tile_data1"], [131, 1, 1, "c.fp_gradmul2l_p", "x1"], [131, 1, 1, "c.fp_gradmul2l_p", "x2"]], "fp_gradmul2l_s": [[131, 1, 1, "c.fp_gradmul2l_s", "core_mask"], [131, 1, 1, "c.fp_gradmul2l_s", "dx1"], [131, 1, 1, "c.fp_gradmul2l_s", "dx2"], [131, 1, 1, "c.fp_gradmul2l_s", "dy"], [131, 1, 1, "c.fp_gradmul2l_s", "indices"], [131, 1, 1, "c.fp_gradmul2l_s", "large_multiples"], [131, 1, 1, "c.fp_gradmul2l_s", "large_shape"], [131, 1, 1, "c.fp_gradmul2l_s", "large_strides"], [131, 1, 1, "c.fp_gradmul2l_s", "ndims"], [131, 1, 1, "c.fp_gradmul2l_s", "out_shape"], [131, 1, 1, "c.fp_gradmul2l_s", "out_strides"], [131, 1, 1, "c.fp_gradmul2l_s", "small_multiples"], [131, 1, 1, "c.fp_gradmul2l_s", "small_shape"], [131, 1, 1, "c.fp_gradmul2l_s", "small_strides"], [131, 1, 1, "c.fp_gradmul2l_s", "tile_data0"], [131, 1, 1, "c.fp_gradmul2l_s", "tile_data1"], [131, 1, 1, "c.fp_gradmul2l_s", "x1"], [131, 1, 1, "c.fp_gradmul2l_s", "x2"]], "fp_gradmul_p": [[131, 1, 1, "c.fp_gradmul_p", "dx1"], [131, 1, 1, "c.fp_gradmul_p", "dx2"], [131, 1, 1, "c.fp_gradmul_p", "dy"], [131, 1, 1, "c.fp_gradmul_p", "indices"], [131, 1, 1, "c.fp_gradmul_p", "large_multiples"], [131, 1, 1, "c.fp_gradmul_p", "large_shape"], [131, 1, 1, "c.fp_gradmul_p", "large_strides"], [131, 1, 1, "c.fp_gradmul_p", "ndims"], [131, 1, 1, "c.fp_gradmul_p", "out_shape"], [131, 1, 1, "c.fp_gradmul_p", "out_strides"], [131, 1, 1, "c.fp_gradmul_p", "small_multiples"], [131, 1, 1, "c.fp_gradmul_p", "small_shape"], [131, 1, 1, "c.fp_gradmul_p", "small_strides"], [131, 1, 1, "c.fp_gradmul_p", "tile_data0"], [131, 1, 1, "c.fp_gradmul_p", "tile_data1"], [131, 1, 1, "c.fp_gradmul_p", "x1"], [131, 1, 1, "c.fp_gradmul_p", "x2"]], "fp_gradmul_s": [[131, 1, 1, "c.fp_gradmul_s", "core_mask"], [131, 1, 1, "c.fp_gradmul_s", "dx1"], [131, 1, 1, "c.fp_gradmul_s", "dx2"], [131, 1, 1, "c.fp_gradmul_s", "dy"], [131, 1, 1, "c.fp_gradmul_s", "indices"], [131, 1, 1, "c.fp_gradmul_s", "large_multiples"], [131, 1, 1, "c.fp_gradmul_s", "large_shape"], [131, 1, 1, "c.fp_gradmul_s", "large_strides"], [131, 1, 1, "c.fp_gradmul_s", "ndims"], [131, 1, 1, "c.fp_gradmul_s", "out_shape"], [131, 1, 1, "c.fp_gradmul_s", "out_strides"], [131, 1, 1, "c.fp_gradmul_s", "small_multiples"], [131, 1, 1, "c.fp_gradmul_s", "small_shape"], [131, 1, 1, "c.fp_gradmul_s", "small_strides"], [131, 1, 1, "c.fp_gradmul_s", "tile_data0"], [131, 1, 1, "c.fp_gradmul_s", "tile_data1"], [131, 1, 1, "c.fp_gradmul_s", "x1"], [131, 1, 1, "c.fp_gradmul_s", "x2"]], "fp_greater_p": [[92, 1, 1, "c.fp_greater_p", "element_num"], [92, 1, 1, "c.fp_greater_p", "in_elements_num0"], [92, 1, 1, "c.fp_greater_p", "input1"], [92, 1, 1, "c.fp_greater_p", "input2"], [92, 1, 1, "c.fp_greater_p", "optimize"], [92, 1, 1, "c.fp_greater_p", "output"]], "fp_greater_s": [[92, 1, 1, "c.fp_greater_s", "core_mask"], [92, 1, 1, "c.fp_greater_s", "element_num"], [92, 1, 1, "c.fp_greater_s", "in_elements_num0"], [92, 1, 1, "c.fp_greater_s", "input1"], [92, 1, 1, "c.fp_greater_s", "input2"], [92, 1, 1, "c.fp_greater_s", "optimize"], [92, 1, 1, "c.fp_greater_s", "output"]], "fp_greaterequal_p": [[93, 1, 1, "c.fp_greaterequal_p", "element_num"], [93, 1, 1, "c.fp_greaterequal_p", "in_elements_num0"], [93, 1, 1, "c.fp_greaterequal_p", "input1"], [93, 1, 1, "c.fp_greaterequal_p", "input2"], [93, 1, 1, "c.fp_greaterequal_p", "optimize"], [93, 1, 1, "c.fp_greaterequal_p", "output"]], "fp_greaterequal_s": [[93, 1, 1, "c.fp_greaterequal_s", "core_mask"], [93, 1, 1, "c.fp_greaterequal_s", "element_num"], [93, 1, 1, "c.fp_greaterequal_s", "in_elements_num0"], [93, 1, 1, "c.fp_greaterequal_s", "input1"], [93, 1, 1, "c.fp_greaterequal_s", "input2"], [93, 1, 1, "c.fp_greaterequal_s", "optimize"], [93, 1, 1, "c.fp_greaterequal_s", "output"]], "fp_groupnormfusion_p": [[94, 1, 1, "c.fp_groupnormfusion_p", "batch"], [94, 1, 1, "c.fp_groupnormfusion_p", "channel"], [94, 1, 1, "c.fp_groupnormfusion_p", "epsilon"], [94, 1, 1, "c.fp_groupnormfusion_p", "input"], [94, 1, 1, "c.fp_groupnormfusion_p", "mean"], [94, 1, 1, "c.fp_groupnormfusion_p", "num_groups"], [94, 1, 1, "c.fp_groupnormfusion_p", "offset"], [94, 1, 1, "c.fp_groupnormfusion_p", "output"], [94, 1, 1, "c.fp_groupnormfusion_p", "scale"], [94, 1, 1, "c.fp_groupnormfusion_p", "unit"], [94, 1, 1, "c.fp_groupnormfusion_p", "variance"]], "fp_groupnormfusion_s": [[94, 1, 1, "c.fp_groupnormfusion_s", "batch"], [94, 1, 1, "c.fp_groupnormfusion_s", "channel"], [94, 1, 1, "c.fp_groupnormfusion_s", "core_mask"], [94, 1, 1, "c.fp_groupnormfusion_s", "epsilon"], [94, 1, 1, "c.fp_groupnormfusion_s", "input"], [94, 1, 1, "c.fp_groupnormfusion_s", "mean"], [94, 1, 1, "c.fp_groupnormfusion_s", "num_groups"], [94, 1, 1, "c.fp_groupnormfusion_s", "offset"], [94, 1, 1, "c.fp_groupnormfusion_s", "output"], [94, 1, 1, "c.fp_groupnormfusion_s", "scale"], [94, 1, 1, "c.fp_groupnormfusion_s", "unit"], [94, 1, 1, "c.fp_groupnormfusion_s", "variance"]], "fp_h_sigmoid_grad_p": [[13, 1, 1, "c.fp_h_sigmoid_grad_p", "dst"], [13, 1, 1, "c.fp_h_sigmoid_grad_p", "length"], [13, 1, 1, "c.fp_h_sigmoid_grad_p", "src0"], [13, 1, 1, "c.fp_h_sigmoid_grad_p", "src1"]], "fp_h_sigmoid_grad_s": [[13, 1, 1, "c.fp_h_sigmoid_grad_s", "core_mask"], [13, 1, 1, "c.fp_h_sigmoid_grad_s", "dst"], [13, 1, 1, "c.fp_h_sigmoid_grad_s", "length"], [13, 1, 1, "c.fp_h_sigmoid_grad_s", "src0"], [13, 1, 1, "c.fp_h_sigmoid_grad_s", "src1"]], "fp_h_swish_grad_p": [[13, 1, 1, "c.fp_h_swish_grad_p", "dst"], [13, 1, 1, "c.fp_h_swish_grad_p", "length"], [13, 1, 1, "c.fp_h_swish_grad_p", "src0"], [13, 1, 1, "c.fp_h_swish_grad_p", "src1"]], "fp_h_swish_grad_s": [[13, 1, 1, "c.fp_h_swish_grad_s", "core_mask"], [13, 1, 1, "c.fp_h_swish_grad_s", "dst"], [13, 1, 1, "c.fp_h_swish_grad_s", "length"], [13, 1, 1, "c.fp_h_swish_grad_s", "src0"], [13, 1, 1, "c.fp_h_swish_grad_s", "src1"]], "fp_hard_shrink_grad_p": [[13, 1, 1, "c.fp_hard_shrink_grad_p", "dst"], [13, 1, 1, "c.fp_hard_shrink_grad_p", "lambd"], [13, 1, 1, "c.fp_hard_shrink_grad_p", "length"], [13, 1, 1, "c.fp_hard_shrink_grad_p", "src0"], [13, 1, 1, "c.fp_hard_shrink_grad_p", "src1"]], "fp_hard_shrink_grad_s": [[13, 1, 1, "c.fp_hard_shrink_grad_s", "core_mask"], [13, 1, 1, "c.fp_hard_shrink_grad_s", "dst"], [13, 1, 1, "c.fp_hard_shrink_grad_s", "lambd"], [13, 1, 1, "c.fp_hard_shrink_grad_s", "length"], [13, 1, 1, "c.fp_hard_shrink_grad_s", "src0"], [13, 1, 1, "c.fp_hard_shrink_grad_s", "src1"]], "fp_hardshrink_p": [[12, 1, 1, "c.fp_hardshrink_p", "Input0"], [12, 1, 1, "c.fp_hardshrink_p", "lambd"], [12, 1, 1, "c.fp_hardshrink_p", "length"], [12, 1, 1, "c.fp_hardshrink_p", "output"]], "fp_hardshrink_s": [[12, 1, 1, "c.fp_hardshrink_s", "Input0"], [12, 1, 1, "c.fp_hardshrink_s", "core_mask"], [12, 1, 1, "c.fp_hardshrink_s", "lambd"], [12, 1, 1, "c.fp_hardshrink_s", "length"], [12, 1, 1, "c.fp_hardshrink_s", "output"]], "fp_hardtanh_p": [[12, 1, 1, "c.fp_hardtanh_p", "Input0"], [12, 1, 1, "c.fp_hardtanh_p", "length"], [12, 1, 1, "c.fp_hardtanh_p", "max_val"], [12, 1, 1, "c.fp_hardtanh_p", "min_val"], [12, 1, 1, "c.fp_hardtanh_p", "output"]], "fp_hardtanh_s": [[12, 1, 1, "c.fp_hardtanh_s", "Input0"], [12, 1, 1, "c.fp_hardtanh_s", "core_mask"], [12, 1, 1, "c.fp_hardtanh_s", "length"], [12, 1, 1, "c.fp_hardtanh_s", "max_val"], [12, 1, 1, "c.fp_hardtanh_s", "min_val"], [12, 1, 1, "c.fp_hardtanh_s", "output"]], "fp_hsigmoid_p": [[12, 1, 1, "c.fp_hsigmoid_p", "Input0"], [12, 1, 1, "c.fp_hsigmoid_p", "length"], [12, 1, 1, "c.fp_hsigmoid_p", "output"]], "fp_hsigmoid_s": [[12, 1, 1, "c.fp_hsigmoid_s", "Input0"], [12, 1, 1, "c.fp_hsigmoid_s", "core_mask"], [12, 1, 1, "c.fp_hsigmoid_s", "length"], [12, 1, 1, "c.fp_hsigmoid_s", "output"]], "fp_hswish_p": [[12, 1, 1, "c.fp_hswish_p", "Input0"], [12, 1, 1, "c.fp_hswish_p", "length"], [12, 1, 1, "c.fp_hswish_p", "output"]], "fp_hswish_s": [[12, 1, 1, "c.fp_hswish_s", "Input0"], [12, 1, 1, "c.fp_hswish_s", "core_mask"], [12, 1, 1, "c.fp_hswish_s", "length"], [12, 1, 1, "c.fp_hswish_s", "output"]], "fp_instancenorm_p": [[97, 1, 1, "c.fp_instancenorm_p", "batch"], [97, 1, 1, "c.fp_instancenorm_p", "beta"], [97, 1, 1, "c.fp_instancenorm_p", "channel"], [97, 1, 1, "c.fp_instancenorm_p", "epsilon"], [97, 1, 1, "c.fp_instancenorm_p", "gamma"], [97, 1, 1, "c.fp_instancenorm_p", "inner_size"], [97, 1, 1, "c.fp_instancenorm_p", "input"], [97, 1, 1, "c.fp_instancenorm_p", "output"]], "fp_instancenorm_s": [[97, 1, 1, "c.fp_instancenorm_s", "batch"], [97, 1, 1, "c.fp_instancenorm_s", "beta"], [97, 1, 1, "c.fp_instancenorm_s", "channel"], [97, 1, 1, "c.fp_instancenorm_s", "core_mask"], [97, 1, 1, "c.fp_instancenorm_s", "epsilon"], [97, 1, 1, "c.fp_instancenorm_s", "gamma"], [97, 1, 1, "c.fp_instancenorm_s", "inner_size"], [97, 1, 1, "c.fp_instancenorm_s", "input"], [97, 1, 1, "c.fp_instancenorm_s", "output"]], "fp_isfinite_p": [[99, 1, 1, "c.fp_isfinite_p", "Input"], [99, 1, 1, "c.fp_isfinite_p", "length"], [99, 1, 1, "c.fp_isfinite_p", "output"]], "fp_isfinite_s": [[99, 1, 1, "c.fp_isfinite_s", "Input"], [99, 1, 1, "c.fp_isfinite_s", "core_mask"], [99, 1, 1, "c.fp_isfinite_s", "length"], [99, 1, 1, "c.fp_isfinite_s", "output"]], "fp_l2norm_p": [[100, 1, 1, "c.fp_l2norm_p", "Input"], [100, 1, 1, "c.fp_l2norm_p", "is_relu"], [100, 1, 1, "c.fp_l2norm_p", "is_relu6"], [100, 1, 1, "c.fp_l2norm_p", "length"], [100, 1, 1, "c.fp_l2norm_p", "output"], [100, 1, 1, "c.fp_l2norm_p", "sqrt_sum"]], "fp_l2norm_s": [[100, 1, 1, "c.fp_l2norm_s", "Input"], [100, 1, 1, "c.fp_l2norm_s", "core_mask"], [100, 1, 1, "c.fp_l2norm_s", "is_relu"], [100, 1, 1, "c.fp_l2norm_s", "is_relu6"], [100, 1, 1, "c.fp_l2norm_s", "length"], [100, 1, 1, "c.fp_l2norm_s", "output"], [100, 1, 1, "c.fp_l2norm_s", "sqrt_sum"]], "fp_l_relu_grad_p": [[13, 1, 1, "c.fp_l_relu_grad_p", "alpha"], [13, 1, 1, "c.fp_l_relu_grad_p", "dst"], [13, 1, 1, "c.fp_l_relu_grad_p", "length"], [13, 1, 1, "c.fp_l_relu_grad_p", "src0"], [13, 1, 1, "c.fp_l_relu_grad_p", "src1"]], "fp_l_relu_grad_s": [[13, 1, 1, "c.fp_l_relu_grad_s", "alpha"], [13, 1, 1, "c.fp_l_relu_grad_s", "core_mask"], [13, 1, 1, "c.fp_l_relu_grad_s", "dst"], [13, 1, 1, "c.fp_l_relu_grad_s", "length"], [13, 1, 1, "c.fp_l_relu_grad_s", "src0"], [13, 1, 1, "c.fp_l_relu_grad_s", "src1"]], "fp_layernormfusion_p": [[101, 1, 1, "c.fp_layernormfusion_p", "beta_data"], [101, 1, 1, "c.fp_layernormfusion_p", "dst_data"], [101, 1, 1, "c.fp_layernormfusion_p", "epsilon"], [101, 1, 1, "c.fp_layernormfusion_p", "gamma_data"], [101, 1, 1, "c.fp_layernormfusion_p", "norm_inner_size"], [101, 1, 1, "c.fp_layernormfusion_p", "norm_outer_size"], [101, 1, 1, "c.fp_layernormfusion_p", "out_mean"], [101, 1, 1, "c.fp_layernormfusion_p", "out_variance"], [101, 1, 1, "c.fp_layernormfusion_p", "param_inner_size"], [101, 1, 1, "c.fp_layernormfusion_p", "param_outer_size"], [101, 1, 1, "c.fp_layernormfusion_p", "src_data"]], "fp_layernormfusion_s": [[101, 1, 1, "c.fp_layernormfusion_s", "Input0"], [101, 1, 1, "c.fp_layernormfusion_s", "beta_data"], [101, 1, 1, "c.fp_layernormfusion_s", "core_mask"], [101, 1, 1, "c.fp_layernormfusion_s", "epsilon"], [101, 1, 1, "c.fp_layernormfusion_s", "gamma_data"], [101, 1, 1, "c.fp_layernormfusion_s", "length"], [101, 1, 1, "c.fp_layernormfusion_s", "norm_inner_size"], [101, 1, 1, "c.fp_layernormfusion_s", "norm_outer_size"], [101, 1, 1, "c.fp_layernormfusion_s", "out_mean"], [101, 1, 1, "c.fp_layernormfusion_s", "out_variance"], [101, 1, 1, "c.fp_layernormfusion_s", "output"], [101, 1, 1, "c.fp_layernormfusion_s", "param_inner_size"], [101, 1, 1, "c.fp_layernormfusion_s", "param_outer_size"]], "fp_layernormgrad_p": [[102, 1, 1, "c.fp_layernormgrad_p", "block_num"], [102, 1, 1, "c.fp_layernormgrad_p", "block_size"], [102, 1, 1, "c.fp_layernormgrad_p", "db"], [102, 1, 1, "c.fp_layernormgrad_p", "dg"], [102, 1, 1, "c.fp_layernormgrad_p", "dx"], [102, 1, 1, "c.fp_layernormgrad_p", "dy"], [102, 1, 1, "c.fp_layernormgrad_p", "gamma"], [102, 1, 1, "c.fp_layernormgrad_p", "mean"], [102, 1, 1, "c.fp_layernormgrad_p", "param_num"], [102, 1, 1, "c.fp_layernormgrad_p", "param_size"], [102, 1, 1, "c.fp_layernormgrad_p", "var"], [102, 1, 1, "c.fp_layernormgrad_p", "x"]], "fp_layernormgrad_s": [[102, 1, 1, "c.fp_layernormgrad_s", "block_num"], [102, 1, 1, "c.fp_layernormgrad_s", "block_size"], [102, 1, 1, "c.fp_layernormgrad_s", "core_mask"], [102, 1, 1, "c.fp_layernormgrad_s", "db"], [102, 1, 1, "c.fp_layernormgrad_s", "dg"], [102, 1, 1, "c.fp_layernormgrad_s", "dx"], [102, 1, 1, "c.fp_layernormgrad_s", "dy"], [102, 1, 1, "c.fp_layernormgrad_s", "gamma"], [102, 1, 1, "c.fp_layernormgrad_s", "mean"], [102, 1, 1, "c.fp_layernormgrad_s", "param_num"], [102, 1, 1, "c.fp_layernormgrad_s", "param_size"], [102, 1, 1, "c.fp_layernormgrad_s", "var"], [102, 1, 1, "c.fp_layernormgrad_s", "x"]], "fp_leaky_relu_p": [[103, 1, 1, "c.fp_leaky_relu_p", "alpha"], [103, 1, 1, "c.fp_leaky_relu_p", "core_mask"], [103, 1, 1, "c.fp_leaky_relu_p", "elem_cnt"], [103, 1, 1, "c.fp_leaky_relu_p", "input"], [103, 1, 1, "c.fp_leaky_relu_p", "output"]], "fp_leaky_relu_s": [[103, 1, 1, "c.fp_leaky_relu_s", "alpha"], [103, 1, 1, "c.fp_leaky_relu_s", "core_mask"], [103, 1, 1, "c.fp_leaky_relu_s", "elem_cnt"], [103, 1, 1, "c.fp_leaky_relu_s", "input"], [103, 1, 1, "c.fp_leaky_relu_s", "output"]], "fp_less_p": [[104, 1, 1, "c.fp_less_p", "Input0"], [104, 1, 1, "c.fp_less_p", "Input1"], [104, 1, 1, "c.fp_less_p", "in_elements_num0"], [104, 1, 1, "c.fp_less_p", "length"], [104, 1, 1, "c.fp_less_p", "optimize"], [104, 1, 1, "c.fp_less_p", "output"]], "fp_less_s": [[104, 1, 1, "c.fp_less_s", "Input0"], [104, 1, 1, "c.fp_less_s", "Input1"], [104, 1, 1, "c.fp_less_s", "core_mask"], [104, 1, 1, "c.fp_less_s", "in_elements_num0"], [104, 1, 1, "c.fp_less_s", "length"], [104, 1, 1, "c.fp_less_s", "optimize"], [104, 1, 1, "c.fp_less_s", "output"]], "fp_lessequal_p": [[105, 1, 1, "c.fp_lessequal_p", "Input0"], [105, 1, 1, "c.fp_lessequal_p", "Input1"], [105, 1, 1, "c.fp_lessequal_p", "in_elements_num0"], [105, 1, 1, "c.fp_lessequal_p", "length"], [105, 1, 1, "c.fp_lessequal_p", "optimize"], [105, 1, 1, "c.fp_lessequal_p", "output"]], "fp_lessequal_s": [[105, 1, 1, "c.fp_lessequal_s", "Input0"], [105, 1, 1, "c.fp_lessequal_s", "Input1"], [105, 1, 1, "c.fp_lessequal_s", "core_mask"], [105, 1, 1, "c.fp_lessequal_s", "in_elements_num0"], [105, 1, 1, "c.fp_lessequal_s", "length"], [105, 1, 1, "c.fp_lessequal_s", "optimize"], [105, 1, 1, "c.fp_lessequal_s", "output"]], "fp_linspace_p": [[106, 1, 1, "c.fp_linspace_p", "num"], [106, 1, 1, "c.fp_linspace_p", "output"], [106, 1, 1, "c.fp_linspace_p", "start"], [106, 1, 1, "c.fp_linspace_p", "step"]], "fp_linspace_s": [[106, 1, 1, "c.fp_linspace_s", "core_mask"], [106, 1, 1, "c.fp_linspace_s", "end"], [106, 1, 1, "c.fp_linspace_s", "length"], [106, 1, 1, "c.fp_linspace_s", "output"], [106, 1, 1, "c.fp_linspace_s", "start"]], "fp_log1p_p": [[108, 1, 1, "c.fp_log1p_p", "Input"], [108, 1, 1, "c.fp_log1p_p", "length"], [108, 1, 1, "c.fp_log1p_p", "output"]], "fp_log1p_s": [[108, 1, 1, "c.fp_log1p_s", "Input"], [108, 1, 1, "c.fp_log1p_s", "core_mask"], [108, 1, 1, "c.fp_log1p_s", "length"], [108, 1, 1, "c.fp_log1p_s", "output"]], "fp_log_grad_p": [[109, 1, 1, "c.fp_log_grad_p", "input0"], [109, 1, 1, "c.fp_log_grad_p", "input1"], [109, 1, 1, "c.fp_log_grad_p", "length"], [109, 1, 1, "c.fp_log_grad_p", "output"]], "fp_log_grad_s": [[109, 1, 1, "c.fp_log_grad_s", "core_mask"], [109, 1, 1, "c.fp_log_grad_s", "input0"], [109, 1, 1, "c.fp_log_grad_s", "input1"], [109, 1, 1, "c.fp_log_grad_s", "length"], [109, 1, 1, "c.fp_log_grad_s", "output"]], "fp_log_p": [[107, 1, 1, "c.fp_log_p", "input"], [107, 1, 1, "c.fp_log_p", "length"], [107, 1, 1, "c.fp_log_p", "output"]], "fp_log_s": [[107, 1, 1, "c.fp_log_s", "core_mask"], [107, 1, 1, "c.fp_log_s", "input"], [107, 1, 1, "c.fp_log_s", "length"], [107, 1, 1, "c.fp_log_s", "output"]], "fp_logical_not_p": [[110, 1, 1, "c.fp_logical_not_p", "input"], [110, 1, 1, "c.fp_logical_not_p", "length"], [110, 1, 1, "c.fp_logical_not_p", "output"]], "fp_logical_not_s": [[110, 1, 1, "c.fp_logical_not_s", "core_mask"], [110, 1, 1, "c.fp_logical_not_s", "input"], [110, 1, 1, "c.fp_logical_not_s", "length"], [110, 1, 1, "c.fp_logical_not_s", "output"]], "fp_logical_or_p": [[111, 1, 1, "c.fp_logical_or_p", "input0"], [111, 1, 1, "c.fp_logical_or_p", "input1"], [111, 1, 1, "c.fp_logical_or_p", "length"], [111, 1, 1, "c.fp_logical_or_p", "output"]], "fp_logical_or_s": [[111, 1, 1, "c.fp_logical_or_s", "core_mask"], [111, 1, 1, "c.fp_logical_or_s", "input0"], [111, 1, 1, "c.fp_logical_or_s", "input1"], [111, 1, 1, "c.fp_logical_or_s", "length"], [111, 1, 1, "c.fp_logical_or_s", "output"]], "fp_logsoftmax_p": [[113, 1, 1, "c.fp_logsoftmax_p", "axis"], [113, 1, 1, "c.fp_logsoftmax_p", "axis_size"], [113, 1, 1, "c.fp_logsoftmax_p", "inner_size"], [113, 1, 1, "c.fp_logsoftmax_p", "input_ptr"], [113, 1, 1, "c.fp_logsoftmax_p", "n_dim"], [113, 1, 1, "c.fp_logsoftmax_p", "output_ptr"], [113, 1, 1, "c.fp_logsoftmax_p", "outter_size"], [113, 1, 1, "c.fp_logsoftmax_p", "sum_data"]], "fp_logsoftmax_s": [[113, 1, 1, "c.fp_logsoftmax_s", "axis"], [113, 1, 1, "c.fp_logsoftmax_s", "axis_size"], [113, 1, 1, "c.fp_logsoftmax_s", "core_mask"], [113, 1, 1, "c.fp_logsoftmax_s", "inner_size"], [113, 1, 1, "c.fp_logsoftmax_s", "input_ptr"], [113, 1, 1, "c.fp_logsoftmax_s", "n_dim"], [113, 1, 1, "c.fp_logsoftmax_s", "output_ptr"], [113, 1, 1, "c.fp_logsoftmax_s", "outter_size"], [113, 1, 1, "c.fp_logsoftmax_s", "sum_data"]], "fp_lpnorm_p": [[114, 1, 1, "c.fp_lpnorm_p", "batch"], [114, 1, 1, "c.fp_lpnorm_p", "beta"], [114, 1, 1, "c.fp_lpnorm_p", "channel"], [114, 1, 1, "c.fp_lpnorm_p", "epsilon"], [114, 1, 1, "c.fp_lpnorm_p", "gamma"], [114, 1, 1, "c.fp_lpnorm_p", "inner_size"], [114, 1, 1, "c.fp_lpnorm_p", "input"], [114, 1, 1, "c.fp_lpnorm_p", "output"], [114, 1, 1, "c.fp_lpnorm_p", "p"]], "fp_lpnorm_s": [[114, 1, 1, "c.fp_lpnorm_s", "batch"], [114, 1, 1, "c.fp_lpnorm_s", "beta"], [114, 1, 1, "c.fp_lpnorm_s", "channel"], [114, 1, 1, "c.fp_lpnorm_s", "core_mask"], [114, 1, 1, "c.fp_lpnorm_s", "epsilon"], [114, 1, 1, "c.fp_lpnorm_s", "gamma"], [114, 1, 1, "c.fp_lpnorm_s", "inner_size"], [114, 1, 1, "c.fp_lpnorm_s", "input"], [114, 1, 1, "c.fp_lpnorm_s", "output"], [114, 1, 1, "c.fp_lpnorm_s", "p"]], "fp_lrelu_p": [[12, 1, 1, "c.fp_lrelu_p", "Input0"], [12, 1, 1, "c.fp_lrelu_p", "alpha"], [12, 1, 1, "c.fp_lrelu_p", "length"], [12, 1, 1, "c.fp_lrelu_p", "output"]], "fp_lrelu_s": [[12, 1, 1, "c.fp_lrelu_s", "Input0"], [12, 1, 1, "c.fp_lrelu_s", "alpha"], [12, 1, 1, "c.fp_lrelu_s", "core_mask"], [12, 1, 1, "c.fp_lrelu_s", "length"], [12, 1, 1, "c.fp_lrelu_s", "output"]], "fp_lrn_p": [[115, 1, 1, "c.fp_lrn_p", "alpha"], [115, 1, 1, "c.fp_lrn_p", "beta"], [115, 1, 1, "c.fp_lrn_p", "bias"], [115, 1, 1, "c.fp_lrn_p", "channel"], [115, 1, 1, "c.fp_lrn_p", "depth_radius"], [115, 1, 1, "c.fp_lrn_p", "input"], [115, 1, 1, "c.fp_lrn_p", "out_size"], [115, 1, 1, "c.fp_lrn_p", "output"]], "fp_lrn_s": [[115, 1, 1, "c.fp_lrn_s", "alpha"], [115, 1, 1, "c.fp_lrn_s", "beta"], [115, 1, 1, "c.fp_lrn_s", "bias"], [115, 1, 1, "c.fp_lrn_s", "channel"], [115, 1, 1, "c.fp_lrn_s", "core_mask"], [115, 1, 1, "c.fp_lrn_s", "depth_radius"], [115, 1, 1, "c.fp_lrn_s", "input"], [115, 1, 1, "c.fp_lrn_s", "out_size"], [115, 1, 1, "c.fp_lrn_s", "output"]], "fp_lsh_projection_p": [[116, 1, 1, "c.fp_lsh_projection_p", "bits_per_hash"], [116, 1, 1, "c.fp_lsh_projection_p", "feature"], [116, 1, 1, "c.fp_lsh_projection_p", "feature_num"], [116, 1, 1, "c.fp_lsh_projection_p", "hash_group_num"], [116, 1, 1, "c.fp_lsh_projection_p", "hash_seed"], [116, 1, 1, "c.fp_lsh_projection_p", "output"], [116, 1, 1, "c.fp_lsh_projection_p", "weight"]], "fp_lsh_projection_s": [[116, 1, 1, "c.fp_lsh_projection_s", "bits_per_hash"], [116, 1, 1, "c.fp_lsh_projection_s", "core_mask"], [116, 1, 1, "c.fp_lsh_projection_s", "feature"], [116, 1, 1, "c.fp_lsh_projection_s", "feature_num"], [116, 1, 1, "c.fp_lsh_projection_s", "hash_group_num"], [116, 1, 1, "c.fp_lsh_projection_s", "hash_seed"], [116, 1, 1, "c.fp_lsh_projection_s", "output"], [116, 1, 1, "c.fp_lsh_projection_s", "weight"]], "fp_lstmgrad_p": [[118, 1, 1, "c.fp_lstmgrad_p", "dynamic_params"], [118, 1, 1, "c.fp_lstmgrad_p", "params"]], "fp_lstmgrad_s": [[118, 1, 1, "c.fp_lstmgrad_s", "core_mask"], [118, 1, 1, "c.fp_lstmgrad_s", "dynamic_params"], [118, 1, 1, "c.fp_lstmgrad_s", "params"]], "fp_lstmgraddata_p": [[119, 1, 1, "c.fp_lstmgraddata_p", "dynamic_params"], [119, 1, 1, "c.fp_lstmgraddata_p", "params"]], "fp_lstmgraddata_s": [[119, 1, 1, "c.fp_lstmgraddata_s", "core_mask"], [119, 1, 1, "c.fp_lstmgraddata_s", "dynamic_params"], [119, 1, 1, "c.fp_lstmgraddata_s", "params"]], "fp_lstmgradweight_p": [[120, 1, 1, "c.fp_lstmgradweight_p", "dynamic_params"], [120, 1, 1, "c.fp_lstmgradweight_p", "params"]], "fp_lstmgradweight_s": [[120, 1, 1, "c.fp_lstmgradweight_s", "core_mask"], [120, 1, 1, "c.fp_lstmgradweight_s", "dynamic_params"], [120, 1, 1, "c.fp_lstmgradweight_s", "params"]], "fp_matmulfusion_p": [[121, 1, 1, "c.fp_matmulfusion_p", "A"], [121, 1, 1, "c.fp_matmulfusion_p", "B"], [121, 1, 1, "c.fp_matmulfusion_p", "C"], [121, 1, 1, "c.fp_matmulfusion_p", "K"], [121, 1, 1, "c.fp_matmulfusion_p", "M"], [121, 1, 1, "c.fp_matmulfusion_p", "N"], [121, 1, 1, "c.fp_matmulfusion_p", "activation_type"], [121, 1, 1, "c.fp_matmulfusion_p", "bias"]], "fp_matmulfusion_s": [[121, 1, 1, "c.fp_matmulfusion_s", "A"], [121, 1, 1, "c.fp_matmulfusion_s", "B"], [121, 1, 1, "c.fp_matmulfusion_s", "C"], [121, 1, 1, "c.fp_matmulfusion_s", "K"], [121, 1, 1, "c.fp_matmulfusion_s", "M"], [121, 1, 1, "c.fp_matmulfusion_s", "N"], [121, 1, 1, "c.fp_matmulfusion_s", "activation_type"], [121, 1, 1, "c.fp_matmulfusion_s", "bias"], [121, 1, 1, "c.fp_matmulfusion_s", "core_mask"]], "fp_maximum_p": [[122, 1, 1, "c.fp_maximum_p", "input0"], [122, 1, 1, "c.fp_maximum_p", "input1"], [122, 1, 1, "c.fp_maximum_p", "length"], [122, 1, 1, "c.fp_maximum_p", "output"]], "fp_maximum_s": [[122, 1, 1, "c.fp_maximum_s", "core_mask"], [122, 1, 1, "c.fp_maximum_s", "input0"], [122, 1, 1, "c.fp_maximum_s", "input1"], [122, 1, 1, "c.fp_maximum_s", "length"], [122, 1, 1, "c.fp_maximum_s", "output"]], "fp_maximumgrad_p": [[123, 1, 1, "c.fp_maximumgrad_p", "Input0"], [123, 1, 1, "c.fp_maximumgrad_p", "Input0_dims"], [123, 1, 1, "c.fp_maximumgrad_p", "Input1"], [123, 1, 1, "c.fp_maximumgrad_p", "Input1_dims"], [123, 1, 1, "c.fp_maximumgrad_p", "dx0"], [123, 1, 1, "c.fp_maximumgrad_p", "dx1"], [123, 1, 1, "c.fp_maximumgrad_p", "dy"], [123, 1, 1, "c.fp_maximumgrad_p", "num_dims"]], "fp_maximumgrad_s": [[123, 1, 1, "c.fp_maximumgrad_s", "Input0"], [123, 1, 1, "c.fp_maximumgrad_s", "Input0_dims"], [123, 1, 1, "c.fp_maximumgrad_s", "Input1"], [123, 1, 1, "c.fp_maximumgrad_s", "Input1_dims"], [123, 1, 1, "c.fp_maximumgrad_s", "core_mask"], [123, 1, 1, "c.fp_maximumgrad_s", "dx0"], [123, 1, 1, "c.fp_maximumgrad_s", "dx1"], [123, 1, 1, "c.fp_maximumgrad_s", "dy"], [123, 1, 1, "c.fp_maximumgrad_s", "num_dims"]], "fp_maxpool_fusion_p": [[124, 1, 1, "c.fp_maxpool_fusion_p", "params"]], "fp_maxpool_fusion_s": [[124, 1, 1, "c.fp_maxpool_fusion_s", "core_mask"], [124, 1, 1, "c.fp_maxpool_fusion_s", "params"]], "fp_maxpool_grad_p": [[125, 1, 1, "c.fp_maxpool_grad_p", "params"]], "fp_maxpool_grad_s": [[125, 1, 1, "c.fp_maxpool_grad_s", "core_mask"], [125, 1, 1, "c.fp_maxpool_grad_s", "params"]], "fp_mfcc_p": [[126, 1, 1, "c.fp_mfcc_p", "mfcc_params"], [126, 1, 1, "c.fp_mfcc_p", "mfcc_workspace"], [126, 1, 1, "c.fp_mfcc_p", "spec_params"], [126, 1, 1, "c.fp_mfcc_p", "spec_workspace"]], "fp_mfcc_s": [[126, 1, 1, "c.fp_mfcc_s", "core_mask"], [126, 1, 1, "c.fp_mfcc_s", "mfcc_params"], [126, 1, 1, "c.fp_mfcc_s", "mfcc_workspace"], [126, 1, 1, "c.fp_mfcc_s", "spec_params"], [126, 1, 1, "c.fp_mfcc_s", "spec_workspace"]], "fp_minimum_p": [[127, 1, 1, "c.fp_minimum_p", "input0"], [127, 1, 1, "c.fp_minimum_p", "input1"], [127, 1, 1, "c.fp_minimum_p", "length"], [127, 1, 1, "c.fp_minimum_p", "output"]], "fp_minimum_s": [[127, 1, 1, "c.fp_minimum_s", "core_mask"], [127, 1, 1, "c.fp_minimum_s", "input0"], [127, 1, 1, "c.fp_minimum_s", "input1"], [127, 1, 1, "c.fp_minimum_s", "length"], [127, 1, 1, "c.fp_minimum_s", "output"]], "fp_minimumgrad_p": [[128, 1, 1, "c.fp_minimumgrad_p", "Input0"], [128, 1, 1, "c.fp_minimumgrad_p", "Input0_dims"], [128, 1, 1, "c.fp_minimumgrad_p", "Input1"], [128, 1, 1, "c.fp_minimumgrad_p", "Input1_dims"], [128, 1, 1, "c.fp_minimumgrad_p", "dx0"], [128, 1, 1, "c.fp_minimumgrad_p", "dx1"], [128, 1, 1, "c.fp_minimumgrad_p", "dy"], [128, 1, 1, "c.fp_minimumgrad_p", "num_dims"]], "fp_minimumgrad_s": [[128, 1, 1, "c.fp_minimumgrad_s", "Input0"], [128, 1, 1, "c.fp_minimumgrad_s", "Input0_dims"], [128, 1, 1, "c.fp_minimumgrad_s", "Input1"], [128, 1, 1, "c.fp_minimumgrad_s", "Input1_dims"], [128, 1, 1, "c.fp_minimumgrad_s", "core_mask"], [128, 1, 1, "c.fp_minimumgrad_s", "dx0"], [128, 1, 1, "c.fp_minimumgrad_s", "dx1"], [128, 1, 1, "c.fp_minimumgrad_s", "dy"], [128, 1, 1, "c.fp_minimumgrad_s", "num_dims"]], "fp_mod_p": [[129, 1, 1, "c.fp_mod_p", "input0"], [129, 1, 1, "c.fp_mod_p", "input1"], [129, 1, 1, "c.fp_mod_p", "length"], [129, 1, 1, "c.fp_mod_p", "output"]], "fp_mod_s": [[129, 1, 1, "c.fp_mod_s", "core_mask"], [129, 1, 1, "c.fp_mod_s", "input0"], [129, 1, 1, "c.fp_mod_s", "input1"], [129, 1, 1, "c.fp_mod_s", "length"], [129, 1, 1, "c.fp_mod_s", "output"]], "fp_mul_p": [[130, 1, 1, "c.fp_mul_p", "input0"], [130, 1, 1, "c.fp_mul_p", "input1"], [130, 1, 1, "c.fp_mul_p", "length"], [130, 1, 1, "c.fp_mul_p", "output"]], "fp_mul_s": [[130, 1, 1, "c.fp_mul_s", "core_mask"], [130, 1, 1, "c.fp_mul_s", "input0"], [130, 1, 1, "c.fp_mul_s", "input1"], [130, 1, 1, "c.fp_mul_s", "length"], [130, 1, 1, "c.fp_mul_s", "output"]], "fp_neg_grad_p": [[133, 1, 1, "c.fp_neg_grad_p", "Input"], [133, 1, 1, "c.fp_neg_grad_p", "length"], [133, 1, 1, "c.fp_neg_grad_p", "output"]], "fp_neg_grad_s": [[133, 1, 1, "c.fp_neg_grad_s", "Input"], [133, 1, 1, "c.fp_neg_grad_s", "core_mask"], [133, 1, 1, "c.fp_neg_grad_s", "length"], [133, 1, 1, "c.fp_neg_grad_s", "output"]], "fp_neg_p": [[132, 1, 1, "c.fp_neg_p", "Input"], [132, 1, 1, "c.fp_neg_p", "length"], [132, 1, 1, "c.fp_neg_p", "output"]], "fp_neg_s": [[132, 1, 1, "c.fp_neg_s", "Input"], [132, 1, 1, "c.fp_neg_s", "core_mask"], [132, 1, 1, "c.fp_neg_s", "length"], [132, 1, 1, "c.fp_neg_s", "output"]], "fp_nllloss_p": [[134, 1, 1, "c.fp_nllloss_p", "batch_size"], [134, 1, 1, "c.fp_nllloss_p", "class_num"], [134, 1, 1, "c.fp_nllloss_p", "labels"], [134, 1, 1, "c.fp_nllloss_p", "log_probs"], [134, 1, 1, "c.fp_nllloss_p", "loss"], [134, 1, 1, "c.fp_nllloss_p", "reduction_type"], [134, 1, 1, "c.fp_nllloss_p", "total_weight"], [134, 1, 1, "c.fp_nllloss_p", "weight"]], "fp_nllloss_s": [[134, 1, 1, "c.fp_nllloss_s", "batch_size"], [134, 1, 1, "c.fp_nllloss_s", "class_num"], [134, 1, 1, "c.fp_nllloss_s", "core_mask"], [134, 1, 1, "c.fp_nllloss_s", "labels"], [134, 1, 1, "c.fp_nllloss_s", "log_probs"], [134, 1, 1, "c.fp_nllloss_s", "loss"], [134, 1, 1, "c.fp_nllloss_s", "reduction_type"], [134, 1, 1, "c.fp_nllloss_s", "total_weight"], [134, 1, 1, "c.fp_nllloss_s", "weight"]], "fp_nlllossgrad_p": [[135, 1, 1, "c.fp_nlllossgrad_p", "batch"], [135, 1, 1, "c.fp_nlllossgrad_p", "class_num"], [135, 1, 1, "c.fp_nlllossgrad_p", "labels"], [135, 1, 1, "c.fp_nlllossgrad_p", "logits"], [135, 1, 1, "c.fp_nlllossgrad_p", "logits_grad"], [135, 1, 1, "c.fp_nlllossgrad_p", "loss_grad"], [135, 1, 1, "c.fp_nlllossgrad_p", "reduction_type"], [135, 1, 1, "c.fp_nlllossgrad_p", "total_weight"], [135, 1, 1, "c.fp_nlllossgrad_p", "weight"]], "fp_nlllossgrad_s": [[135, 1, 1, "c.fp_nlllossgrad_s", "batch"], [135, 1, 1, "c.fp_nlllossgrad_s", "class_num"], [135, 1, 1, "c.fp_nlllossgrad_s", "core_mask"], [135, 1, 1, "c.fp_nlllossgrad_s", "labels"], [135, 1, 1, "c.fp_nlllossgrad_s", "logits"], [135, 1, 1, "c.fp_nlllossgrad_s", "logits_grad"], [135, 1, 1, "c.fp_nlllossgrad_s", "loss_grad"], [135, 1, 1, "c.fp_nlllossgrad_s", "reduction_type"], [135, 1, 1, "c.fp_nlllossgrad_s", "total_weight"], [135, 1, 1, "c.fp_nlllossgrad_s", "weight"]], "fp_non_max_suppression_p": [[136, 1, 1, "c.fp_non_max_suppression_p", "param"]], "fp_non_max_suppression_s": [[136, 1, 1, "c.fp_non_max_suppression_s", "core_mask"], [136, 1, 1, "c.fp_non_max_suppression_s", "param"]], "fp_nonzero_p": [[137, 1, 1, "c.fp_nonzero_p", "dim_strides"], [137, 1, 1, "c.fp_nonzero_p", "input"], [137, 1, 1, "c.fp_nonzero_p", "input_rank"], [137, 1, 1, "c.fp_nonzero_p", "length"], [137, 1, 1, "c.fp_nonzero_p", "non_zero_num"], [137, 1, 1, "c.fp_nonzero_p", "output"], [137, 1, 1, "c.fp_nonzero_p", "shape"]], "fp_nonzero_s": [[137, 1, 1, "c.fp_nonzero_s", "core_mask"], [137, 1, 1, "c.fp_nonzero_s", "dim_strides"], [137, 1, 1, "c.fp_nonzero_s", "input"], [137, 1, 1, "c.fp_nonzero_s", "input_rank"], [137, 1, 1, "c.fp_nonzero_s", "length"], [137, 1, 1, "c.fp_nonzero_s", "non_zero_num"], [137, 1, 1, "c.fp_nonzero_s", "output"], [137, 1, 1, "c.fp_nonzero_s", "shape"]], "fp_not_equal_p": [[138, 1, 1, "c.fp_not_equal_p", "Input0"], [138, 1, 1, "c.fp_not_equal_p", "Input1"], [138, 1, 1, "c.fp_not_equal_p", "length"], [138, 1, 1, "c.fp_not_equal_p", "output"]], "fp_not_equal_s": [[138, 1, 1, "c.fp_not_equal_s", "Input0"], [138, 1, 1, "c.fp_not_equal_s", "Input1"], [138, 1, 1, "c.fp_not_equal_s", "core_mask"], [138, 1, 1, "c.fp_not_equal_s", "length"], [138, 1, 1, "c.fp_not_equal_s", "output"]], "fp_onehot_p": [[139, 1, 1, "c.fp_onehot_p", "axis"], [139, 1, 1, "c.fp_onehot_p", "depth"], [139, 1, 1, "c.fp_onehot_p", "indices"], [139, 1, 1, "c.fp_onehot_p", "indices_shape"], [139, 1, 1, "c.fp_onehot_p", "indices_shape_size"], [139, 1, 1, "c.fp_onehot_p", "on_off"], [139, 1, 1, "c.fp_onehot_p", "output"], [139, 1, 1, "c.fp_onehot_p", "support_neg_index"]], "fp_onehot_s": [[139, 1, 1, "c.fp_onehot_s", "axis"], [139, 1, 1, "c.fp_onehot_s", "core_mask"], [139, 1, 1, "c.fp_onehot_s", "depth"], [139, 1, 1, "c.fp_onehot_s", "indices"], [139, 1, 1, "c.fp_onehot_s", "indices_shape"], [139, 1, 1, "c.fp_onehot_s", "indices_shape_size"], [139, 1, 1, "c.fp_onehot_s", "on_off"], [139, 1, 1, "c.fp_onehot_s", "output"], [139, 1, 1, "c.fp_onehot_s", "support_neg_index"]], "fp_ones_like_p": [[140, 1, 1, "c.fp_ones_like_p", "length"], [140, 1, 1, "c.fp_ones_like_p", "output"]], "fp_ones_like_s": [[140, 1, 1, "c.fp_ones_like_s", "core_mask"], [140, 1, 1, "c.fp_ones_like_s", "length"], [140, 1, 1, "c.fp_ones_like_s", "output"]], "fp_padfusion_p": [[141, 1, 1, "c.fp_padfusion_p", "params"]], "fp_padfusion_s": [[141, 1, 1, "c.fp_padfusion_s", "core_mask"], [141, 1, 1, "c.fp_padfusion_s", "params"]], "fp_pow_fusion_p": [[142, 1, 1, "c.fp_pow_fusion_p", "Input"], [142, 1, 1, "c.fp_pow_fusion_p", "broadcast"], [142, 1, 1, "c.fp_pow_fusion_p", "exponent"], [142, 1, 1, "c.fp_pow_fusion_p", "length_in"], [142, 1, 1, "c.fp_pow_fusion_p", "output"], [142, 1, 1, "c.fp_pow_fusion_p", "scale"], [142, 1, 1, "c.fp_pow_fusion_p", "shift"]], "fp_pow_fusion_s": [[142, 1, 1, "c.fp_pow_fusion_s", "Input"], [142, 1, 1, "c.fp_pow_fusion_s", "broadcast"], [142, 1, 1, "c.fp_pow_fusion_s", "core_mask"], [142, 1, 1, "c.fp_pow_fusion_s", "exponent"], [142, 1, 1, "c.fp_pow_fusion_s", "length_in"], [142, 1, 1, "c.fp_pow_fusion_s", "output"], [142, 1, 1, "c.fp_pow_fusion_s", "scale"], [142, 1, 1, "c.fp_pow_fusion_s", "shift"]], "fp_power_grad_p": [[143, 1, 1, "c.fp_power_grad_p", "Input1"], [143, 1, 1, "c.fp_power_grad_p", "Input2"], [143, 1, 1, "c.fp_power_grad_p", "length"], [143, 1, 1, "c.fp_power_grad_p", "output"], [143, 1, 1, "c.fp_power_grad_p", "power"], [143, 1, 1, "c.fp_power_grad_p", "scale"], [143, 1, 1, "c.fp_power_grad_p", "shift"]], "fp_power_grad_s": [[143, 1, 1, "c.fp_power_grad_s", "Input1"], [143, 1, 1, "c.fp_power_grad_s", "Input2"], [143, 1, 1, "c.fp_power_grad_s", "core_mask"], [143, 1, 1, "c.fp_power_grad_s", "length"], [143, 1, 1, "c.fp_power_grad_s", "output"], [143, 1, 1, "c.fp_power_grad_s", "power"], [143, 1, 1, "c.fp_power_grad_s", "scale"], [143, 1, 1, "c.fp_power_grad_s", "shift"]], "fp_prelufusion_p": [[144, 1, 1, "c.fp_prelufusion_p", "dst_data"], [144, 1, 1, "c.fp_prelufusion_p", "end"], [144, 1, 1, "c.fp_prelufusion_p", "slope"], [144, 1, 1, "c.fp_prelufusion_p", "src_data"], [144, 1, 1, "c.fp_prelufusion_p", "start"]], "fp_prelufusion_s": [[144, 1, 1, "c.fp_prelufusion_s", "core_mask"], [144, 1, 1, "c.fp_prelufusion_s", "dst_data"], [144, 1, 1, "c.fp_prelufusion_s", "end"], [144, 1, 1, "c.fp_prelufusion_s", "slope"], [144, 1, 1, "c.fp_prelufusion_s", "src_data"], [144, 1, 1, "c.fp_prelufusion_s", "start"]], "fp_priorbox_p": [[145, 1, 1, "c.fp_priorbox_p", "different_aspect_ratios"], [145, 1, 1, "c.fp_priorbox_p", "different_aspect_ratios_size"], [145, 1, 1, "c.fp_priorbox_p", "fmap_h"], [145, 1, 1, "c.fp_priorbox_p", "fmap_w"], [145, 1, 1, "c.fp_priorbox_p", "max_sizes"], [145, 1, 1, "c.fp_priorbox_p", "max_sizes_size"], [145, 1, 1, "c.fp_priorbox_p", "min_sizes"], [145, 1, 1, "c.fp_priorbox_p", "min_sizes_size"], [145, 1, 1, "c.fp_priorbox_p", "offset"], [145, 1, 1, "c.fp_priorbox_p", "output"], [145, 1, 1, "c.fp_priorbox_p", "output_size"], [145, 1, 1, "c.fp_priorbox_p", "step_h"], [145, 1, 1, "c.fp_priorbox_p", "step_w"]], "fp_priorbox_s": [[145, 1, 1, "c.fp_priorbox_s", "core_mask"], [145, 1, 1, "c.fp_priorbox_s", "different_aspect_ratios"], [145, 1, 1, "c.fp_priorbox_s", "different_aspect_ratios_size"], [145, 1, 1, "c.fp_priorbox_s", "fmap_h"], [145, 1, 1, "c.fp_priorbox_s", "fmap_w"], [145, 1, 1, "c.fp_priorbox_s", "max_sizes"], [145, 1, 1, "c.fp_priorbox_s", "max_sizes_size"], [145, 1, 1, "c.fp_priorbox_s", "min_sizes"], [145, 1, 1, "c.fp_priorbox_s", "min_sizes_size"], [145, 1, 1, "c.fp_priorbox_s", "offset"], [145, 1, 1, "c.fp_priorbox_s", "output"], [145, 1, 1, "c.fp_priorbox_s", "output_size"], [145, 1, 1, "c.fp_priorbox_s", "step_h"], [145, 1, 1, "c.fp_priorbox_s", "step_w"]], "fp_raggedrange_p": [[147, 1, 1, "c.fp_raggedrange_p", "deltas"], [147, 1, 1, "c.fp_raggedrange_p", "limits"], [147, 1, 1, "c.fp_raggedrange_p", "range_count"], [147, 1, 1, "c.fp_raggedrange_p", "splits"], [147, 1, 1, "c.fp_raggedrange_p", "starts"], [147, 1, 1, "c.fp_raggedrange_p", "values"]], "fp_raggedrange_s": [[147, 1, 1, "c.fp_raggedrange_s", "core_mask"], [147, 1, 1, "c.fp_raggedrange_s", "deltas"], [147, 1, 1, "c.fp_raggedrange_s", "limits"], [147, 1, 1, "c.fp_raggedrange_s", "range_count"], [147, 1, 1, "c.fp_raggedrange_s", "splits"], [147, 1, 1, "c.fp_raggedrange_s", "starts"], [147, 1, 1, "c.fp_raggedrange_s", "values"]], "fp_random_normal_p": [[148, 1, 1, "c.fp_random_normal_p", "length"], [148, 1, 1, "c.fp_random_normal_p", "mean"], [148, 1, 1, "c.fp_random_normal_p", "output"], [148, 1, 1, "c.fp_random_normal_p", "scale"], [148, 1, 1, "c.fp_random_normal_p", "seed"]], "fp_random_normal_s": [[148, 1, 1, "c.fp_random_normal_s", "core_mask"], [148, 1, 1, "c.fp_random_normal_s", "length"], [148, 1, 1, "c.fp_random_normal_s", "mean"], [148, 1, 1, "c.fp_random_normal_s", "output"], [148, 1, 1, "c.fp_random_normal_s", "scale"], [148, 1, 1, "c.fp_random_normal_s", "seed"]], "fp_random_standard_normal_p": [[149, 1, 1, "c.fp_random_standard_normal_p", "length"], [149, 1, 1, "c.fp_random_standard_normal_p", "output"], [149, 1, 1, "c.fp_random_standard_normal_p", "seed"]], "fp_random_standard_normal_s": [[149, 1, 1, "c.fp_random_standard_normal_s", "core_mask"], [149, 1, 1, "c.fp_random_standard_normal_s", "length"], [149, 1, 1, "c.fp_random_standard_normal_s", "output"], [149, 1, 1, "c.fp_random_standard_normal_s", "seed"]], "fp_range_p": [[150, 1, 1, "c.fp_range_p", "delta"], [150, 1, 1, "c.fp_range_p", "length"], [150, 1, 1, "c.fp_range_p", "output"], [150, 1, 1, "c.fp_range_p", "start"]], "fp_range_s": [[150, 1, 1, "c.fp_range_s", "core_mask"], [150, 1, 1, "c.fp_range_s", "delta"], [150, 1, 1, "c.fp_range_s", "length"], [150, 1, 1, "c.fp_range_s", "output"], [150, 1, 1, "c.fp_range_s", "start"]], "fp_real_div_p": [[152, 1, 1, "c.fp_real_div_p", "input0"], [152, 1, 1, "c.fp_real_div_p", "input1"], [152, 1, 1, "c.fp_real_div_p", "length"], [152, 1, 1, "c.fp_real_div_p", "output"]], "fp_real_div_s": [[152, 1, 1, "c.fp_real_div_s", "core_mask"], [152, 1, 1, "c.fp_real_div_s", "input0"], [152, 1, 1, "c.fp_real_div_s", "input1"], [152, 1, 1, "c.fp_real_div_s", "length"], [152, 1, 1, "c.fp_real_div_s", "output"]], "fp_reciprocal_p": [[153, 1, 1, "c.fp_reciprocal_p", "Input"], [153, 1, 1, "c.fp_reciprocal_p", "length"], [153, 1, 1, "c.fp_reciprocal_p", "output"]], "fp_reciprocal_s": [[153, 1, 1, "c.fp_reciprocal_s", "Input"], [153, 1, 1, "c.fp_reciprocal_s", "core_mask"], [153, 1, 1, "c.fp_reciprocal_s", "length"], [153, 1, 1, "c.fp_reciprocal_s", "output"]], "fp_reduce_p": [[154, 1, 1, "c.fp_reduce_p", "core_mask"], [154, 1, 1, "c.fp_reduce_p", "dst_data"], [154, 1, 1, "c.fp_reduce_p", "param"], [154, 1, 1, "c.fp_reduce_p", "src_data"], [154, 1, 1, "c.fp_reduce_p", "tmp_dst_data"], [154, 1, 1, "c.fp_reduce_p", "tmp_src_data"]], "fp_reduce_s": [[154, 1, 1, "c.fp_reduce_s", "core_mask"], [154, 1, 1, "c.fp_reduce_s", "dst_data"], [154, 1, 1, "c.fp_reduce_s", "param"], [154, 1, 1, "c.fp_reduce_s", "src_data"]], "fp_reduceall_p": [[21, 1, 1, "c.fp_reduceall_p", "axis_size"], [21, 1, 1, "c.fp_reduceall_p", "dst_data"], [21, 1, 1, "c.fp_reduceall_p", "inner_size"], [21, 1, 1, "c.fp_reduceall_p", "outer_size"], [21, 1, 1, "c.fp_reduceall_p", "src_data"]], "fp_reduceall_s": [[21, 1, 1, "c.fp_reduceall_s", "axis_size"], [21, 1, 1, "c.fp_reduceall_s", "core_mask"], [21, 1, 1, "c.fp_reduceall_s", "dst_data"], [21, 1, 1, "c.fp_reduceall_s", "inner_size"], [21, 1, 1, "c.fp_reduceall_s", "outer_size"], [21, 1, 1, "c.fp_reduceall_s", "src_data"]], "fp_reducescatter_p": [[155, 1, 1, "c.fp_reducescatter_p", "data_size"], [155, 1, 1, "c.fp_reducescatter_p", "input_data"], [155, 1, 1, "c.fp_reducescatter_p", "output_data"], [155, 1, 1, "c.fp_reducescatter_p", "reduce_type"]], "fp_reducescatter_s": [[155, 1, 1, "c.fp_reducescatter_s", "core_mask"], [155, 1, 1, "c.fp_reducescatter_s", "data_size"], [155, 1, 1, "c.fp_reducescatter_s", "input_data"], [155, 1, 1, "c.fp_reducescatter_s", "output_data"], [155, 1, 1, "c.fp_reducescatter_s", "reduce_type"]], "fp_relu6_grad_p": [[13, 1, 1, "c.fp_relu6_grad_p", "dst"], [13, 1, 1, "c.fp_relu6_grad_p", "length"], [13, 1, 1, "c.fp_relu6_grad_p", "src0"], [13, 1, 1, "c.fp_relu6_grad_p", "src1"]], "fp_relu6_grad_s": [[13, 1, 1, "c.fp_relu6_grad_s", "core_mask"], [13, 1, 1, "c.fp_relu6_grad_s", "dst"], [13, 1, 1, "c.fp_relu6_grad_s", "length"], [13, 1, 1, "c.fp_relu6_grad_s", "src0"], [13, 1, 1, "c.fp_relu6_grad_s", "src1"]], "fp_relu6_p": [[12, 1, 1, "c.fp_relu6_p", "Input0"], [12, 1, 1, "c.fp_relu6_p", "length"], [12, 1, 1, "c.fp_relu6_p", "output"]], "fp_relu6_s": [[12, 1, 1, "c.fp_relu6_s", "Input0"], [12, 1, 1, "c.fp_relu6_s", "core_mask"], [12, 1, 1, "c.fp_relu6_s", "length"], [12, 1, 1, "c.fp_relu6_s", "output"]], "fp_relu_grad_p": [[13, 1, 1, "c.fp_relu_grad_p", "dst"], [13, 1, 1, "c.fp_relu_grad_p", "length"], [13, 1, 1, "c.fp_relu_grad_p", "src0"], [13, 1, 1, "c.fp_relu_grad_p", "src1"]], "fp_relu_grad_s": [[13, 1, 1, "c.fp_relu_grad_s", "core_mask"], [13, 1, 1, "c.fp_relu_grad_s", "dst"], [13, 1, 1, "c.fp_relu_grad_s", "length"], [13, 1, 1, "c.fp_relu_grad_s", "src0"], [13, 1, 1, "c.fp_relu_grad_s", "src1"]], "fp_relu_p": [[12, 1, 1, "c.fp_relu_p", "Input0"], [12, 1, 1, "c.fp_relu_p", "length"], [12, 1, 1, "c.fp_relu_p", "output"]], "fp_relu_s": [[12, 1, 1, "c.fp_relu_s", "Input0"], [12, 1, 1, "c.fp_relu_s", "core_mask"], [12, 1, 1, "c.fp_relu_s", "length"], [12, 1, 1, "c.fp_relu_s", "output"]], "fp_reshape_p": [[156, 1, 1, "c.fp_reshape_p", "input"], [156, 1, 1, "c.fp_reshape_p", "length"], [156, 1, 1, "c.fp_reshape_p", "output"]], "fp_reshape_s": [[156, 1, 1, "c.fp_reshape_s", "core_mask"], [156, 1, 1, "c.fp_reshape_s", "input"], [156, 1, 1, "c.fp_reshape_s", "length"], [156, 1, 1, "c.fp_reshape_s", "output"]], "fp_resize_anycore": [[157, 1, 1, "c.fp_resize_anycore", "core_mask"], [157, 1, 1, "c.fp_resize_anycore", "input"], [157, 1, 1, "c.fp_resize_anycore", "output"], [157, 1, 1, "c.fp_resize_anycore", "param"]], "fp_resizebilineargrad_p": [[158, 1, 1, "c.fp_resizebilineargrad_p", "align_corners"], [158, 1, 1, "c.fp_resizebilineargrad_p", "batch_size"], [158, 1, 1, "c.fp_resizebilineargrad_p", "channel"], [158, 1, 1, "c.fp_resizebilineargrad_p", "format"], [158, 1, 1, "c.fp_resizebilineargrad_p", "height_scale"], [158, 1, 1, "c.fp_resizebilineargrad_p", "in_addr"], [158, 1, 1, "c.fp_resizebilineargrad_p", "in_height"], [158, 1, 1, "c.fp_resizebilineargrad_p", "in_width"], [158, 1, 1, "c.fp_resizebilineargrad_p", "out_addr"], [158, 1, 1, "c.fp_resizebilineargrad_p", "out_height"], [158, 1, 1, "c.fp_resizebilineargrad_p", "out_width"], [158, 1, 1, "c.fp_resizebilineargrad_p", "width_scale"]], "fp_resizebilineargrad_s": [[158, 1, 1, "c.fp_resizebilineargrad_s", "align_corners"], [158, 1, 1, "c.fp_resizebilineargrad_s", "batch_size"], [158, 1, 1, "c.fp_resizebilineargrad_s", "channel"], [158, 1, 1, "c.fp_resizebilineargrad_s", "core_mask"], [158, 1, 1, "c.fp_resizebilineargrad_s", "format"], [158, 1, 1, "c.fp_resizebilineargrad_s", "height_scale"], [158, 1, 1, "c.fp_resizebilineargrad_s", "in_addr"], [158, 1, 1, "c.fp_resizebilineargrad_s", "in_height"], [158, 1, 1, "c.fp_resizebilineargrad_s", "in_width"], [158, 1, 1, "c.fp_resizebilineargrad_s", "out_addr"], [158, 1, 1, "c.fp_resizebilineargrad_s", "out_height"], [158, 1, 1, "c.fp_resizebilineargrad_s", "out_width"], [158, 1, 1, "c.fp_resizebilineargrad_s", "width_scale"]], "fp_resizenearestneighborgrad_p": [[158, 1, 1, "c.fp_resizenearestneighborgrad_p", "align_corners"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "batch_size"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "channel"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "format"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "height_scale"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "in_addr"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "in_height"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "in_width"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "out_addr"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "out_height"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "out_width"], [158, 1, 1, "c.fp_resizenearestneighborgrad_p", "width_scale"]], "fp_resizenearestneighborgrad_s": [[158, 1, 1, "c.fp_resizenearestneighborgrad_s", "align_corners"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "batch_size"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "channel"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "core_mask"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "format"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "height_scale"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "in_addr"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "in_height"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "in_width"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "out_addr"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "out_height"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "out_width"], [158, 1, 1, "c.fp_resizenearestneighborgrad_s", "width_scale"]], "fp_roipooling_p": [[162, 1, 1, "c.fp_roipooling_p", "in_ptr"], [162, 1, 1, "c.fp_roipooling_p", "input_c"], [162, 1, 1, "c.fp_roipooling_p", "input_h"], [162, 1, 1, "c.fp_roipooling_p", "input_n"], [162, 1, 1, "c.fp_roipooling_p", "input_w"], [162, 1, 1, "c.fp_roipooling_p", "max_c"], [162, 1, 1, "c.fp_roipooling_p", "num_rois"], [162, 1, 1, "c.fp_roipooling_p", "out_ptr"], [162, 1, 1, "c.fp_roipooling_p", "pooled_height"], [162, 1, 1, "c.fp_roipooling_p", "pooled_width"], [162, 1, 1, "c.fp_roipooling_p", "roi"], [162, 1, 1, "c.fp_roipooling_p", "scale"]], "fp_roipooling_s": [[162, 1, 1, "c.fp_roipooling_s", "core_mask"], [162, 1, 1, "c.fp_roipooling_s", "in_ptr"], [162, 1, 1, "c.fp_roipooling_s", "input_c"], [162, 1, 1, "c.fp_roipooling_s", "input_h"], [162, 1, 1, "c.fp_roipooling_s", "input_n"], [162, 1, 1, "c.fp_roipooling_s", "input_w"], [162, 1, 1, "c.fp_roipooling_s", "max_c"], [162, 1, 1, "c.fp_roipooling_s", "num_rois"], [162, 1, 1, "c.fp_roipooling_s", "out_ptr"], [162, 1, 1, "c.fp_roipooling_s", "pooled_height"], [162, 1, 1, "c.fp_roipooling_s", "pooled_width"], [162, 1, 1, "c.fp_roipooling_s", "roi"], [162, 1, 1, "c.fp_roipooling_s", "scale"]], "fp_round_p": [[163, 1, 1, "c.fp_round_p", "input"], [163, 1, 1, "c.fp_round_p", "length"], [163, 1, 1, "c.fp_round_p", "output"]], "fp_round_s": [[163, 1, 1, "c.fp_round_s", "core_mask"], [163, 1, 1, "c.fp_round_s", "input"], [163, 1, 1, "c.fp_round_s", "length"], [163, 1, 1, "c.fp_round_s", "output"]], "fp_rsqrt_p": [[164, 1, 1, "c.fp_rsqrt_p", "dst"], [164, 1, 1, "c.fp_rsqrt_p", "length"], [164, 1, 1, "c.fp_rsqrt_p", "src"]], "fp_rsqrt_s": [[164, 1, 1, "c.fp_rsqrt_s", "core_mask"], [164, 1, 1, "c.fp_rsqrt_s", "dst"], [164, 1, 1, "c.fp_rsqrt_s", "length"], [164, 1, 1, "c.fp_rsqrt_s", "src"]], "fp_rsqrtgrad_p": [[165, 1, 1, "c.fp_rsqrtgrad_p", "input1"], [165, 1, 1, "c.fp_rsqrtgrad_p", "input2"], [165, 1, 1, "c.fp_rsqrtgrad_p", "output"], [165, 1, 1, "c.fp_rsqrtgrad_p", "size"]], "fp_rsqrtgrad_s": [[165, 1, 1, "c.fp_rsqrtgrad_s", "core_mask"], [165, 1, 1, "c.fp_rsqrtgrad_s", "input1"], [165, 1, 1, "c.fp_rsqrtgrad_s", "input2"], [165, 1, 1, "c.fp_rsqrtgrad_s", "output"], [165, 1, 1, "c.fp_rsqrtgrad_s", "size"]], "fp_scalefusion_p": [[166, 1, 1, "c.fp_scalefusion_p", "bias"], [166, 1, 1, "c.fp_scalefusion_p", "dst_data"], [166, 1, 1, "c.fp_scalefusion_p", "length"], [166, 1, 1, "c.fp_scalefusion_p", "scale"], [166, 1, 1, "c.fp_scalefusion_p", "src_data"]], "fp_scalefusion_s": [[166, 1, 1, "c.fp_scalefusion_s", "bias"], [166, 1, 1, "c.fp_scalefusion_s", "core_mask"], [166, 1, 1, "c.fp_scalefusion_s", "dst_data"], [166, 1, 1, "c.fp_scalefusion_s", "length"], [166, 1, 1, "c.fp_scalefusion_s", "scale"], [166, 1, 1, "c.fp_scalefusion_s", "src_data"]], "fp_scatter_elements_p": [[167, 1, 1, "c.fp_scatter_elements_p", "core_mask"], [167, 1, 1, "c.fp_scatter_elements_p", "indices"], [167, 1, 1, "c.fp_scatter_elements_p", "input"], [167, 1, 1, "c.fp_scatter_elements_p", "output"], [167, 1, 1, "c.fp_scatter_elements_p", "param"], [167, 1, 1, "c.fp_scatter_elements_p", "updates"]], "fp_scatter_elements_s": [[167, 1, 1, "c.fp_scatter_elements_s", "core_mask"], [167, 1, 1, "c.fp_scatter_elements_s", "indices"], [167, 1, 1, "c.fp_scatter_elements_s", "input"], [167, 1, 1, "c.fp_scatter_elements_s", "output"], [167, 1, 1, "c.fp_scatter_elements_s", "param"], [167, 1, 1, "c.fp_scatter_elements_s", "updates"]], "fp_scatter_nd_p": [[168, 1, 1, "c.fp_scatter_nd_p", "indices"], [168, 1, 1, "c.fp_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.fp_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.fp_scatter_nd_p", "output"], [168, 1, 1, "c.fp_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.fp_scatter_nd_p", "output_shape"], [168, 1, 1, "c.fp_scatter_nd_p", "updates"]], "fp_scatter_nd_s": [[168, 1, 1, "c.fp_scatter_nd_s", "core_mask"], [168, 1, 1, "c.fp_scatter_nd_s", "indices"], [168, 1, 1, "c.fp_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.fp_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.fp_scatter_nd_s", "output"], [168, 1, 1, "c.fp_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.fp_scatter_nd_s", "output_shape"], [168, 1, 1, "c.fp_scatter_nd_s", "updates"]], "fp_scatter_nd_update_p": [[169, 1, 1, "c.fp_scatter_nd_update_p", "indices"], [169, 1, 1, "c.fp_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.fp_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.fp_scatter_nd_update_p", "output"], [169, 1, 1, "c.fp_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.fp_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.fp_scatter_nd_update_p", "updates"]], "fp_scatter_nd_update_s": [[169, 1, 1, "c.fp_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.fp_scatter_nd_update_s", "indices"], [169, 1, 1, "c.fp_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.fp_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.fp_scatter_nd_update_s", "output"], [169, 1, 1, "c.fp_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.fp_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.fp_scatter_nd_update_s", "updates"]], "fp_select_p": [[170, 1, 1, "c.fp_select_p", "condition"], [170, 1, 1, "c.fp_select_p", "index_list1"], [170, 1, 1, "c.fp_select_p", "index_list2"], [170, 1, 1, "c.fp_select_p", "index_list3"], [170, 1, 1, "c.fp_select_p", "input0"], [170, 1, 1, "c.fp_select_p", "input1"], [170, 1, 1, "c.fp_select_p", "is_broadcast"], [170, 1, 1, "c.fp_select_p", "output"], [170, 1, 1, "c.fp_select_p", "output_dims"], [170, 1, 1, "c.fp_select_p", "output_dims_num"]], "fp_select_s": [[170, 1, 1, "c.fp_select_s", "condition"], [170, 1, 1, "c.fp_select_s", "core_mask"], [170, 1, 1, "c.fp_select_s", "index_list1"], [170, 1, 1, "c.fp_select_s", "index_list2"], [170, 1, 1, "c.fp_select_s", "index_list3"], [170, 1, 1, "c.fp_select_s", "input0"], [170, 1, 1, "c.fp_select_s", "input1"], [170, 1, 1, "c.fp_select_s", "is_broadcast"], [170, 1, 1, "c.fp_select_s", "output"], [170, 1, 1, "c.fp_select_s", "output_dims"], [170, 1, 1, "c.fp_select_s", "output_dims_num"]], "fp_sgd_p": [[171, 1, 1, "c.fp_sgd_p", "accumulate"], [171, 1, 1, "c.fp_sgd_p", "dampening"], [171, 1, 1, "c.fp_sgd_p", "gradient"], [171, 1, 1, "c.fp_sgd_p", "learning_rate"], [171, 1, 1, "c.fp_sgd_p", "length"], [171, 1, 1, "c.fp_sgd_p", "moment"], [171, 1, 1, "c.fp_sgd_p", "nesterov"], [171, 1, 1, "c.fp_sgd_p", "weight"], [171, 1, 1, "c.fp_sgd_p", "weight_decay"]], "fp_sgd_s": [[171, 1, 1, "c.fp_sgd_s", "accumulate"], [171, 1, 1, "c.fp_sgd_s", "core_mask"], [171, 1, 1, "c.fp_sgd_s", "dampening"], [171, 1, 1, "c.fp_sgd_s", "end"], [171, 1, 1, "c.fp_sgd_s", "gradient"], [171, 1, 1, "c.fp_sgd_s", "learning_rate"], [171, 1, 1, "c.fp_sgd_s", "moment"], [171, 1, 1, "c.fp_sgd_s", "nesterov"], [171, 1, 1, "c.fp_sgd_s", "start"], [171, 1, 1, "c.fp_sgd_s", "weight"], [171, 1, 1, "c.fp_sgd_s", "weight_decay"]], "fp_sigmoid_grad_p": [[13, 1, 1, "c.fp_sigmoid_grad_p", "dst"], [13, 1, 1, "c.fp_sigmoid_grad_p", "length"], [13, 1, 1, "c.fp_sigmoid_grad_p", "src0"], [13, 1, 1, "c.fp_sigmoid_grad_p", "src1"]], "fp_sigmoid_grad_s": [[13, 1, 1, "c.fp_sigmoid_grad_s", "core_mask"], [13, 1, 1, "c.fp_sigmoid_grad_s", "dst"], [13, 1, 1, "c.fp_sigmoid_grad_s", "length"], [13, 1, 1, "c.fp_sigmoid_grad_s", "src0"], [13, 1, 1, "c.fp_sigmoid_grad_s", "src1"]], "fp_sigmoid_p": [[12, 1, 1, "c.fp_sigmoid_p", "Input0"], [12, 1, 1, "c.fp_sigmoid_p", "length"], [12, 1, 1, "c.fp_sigmoid_p", "output"]], "fp_sigmoid_s": [[12, 1, 1, "c.fp_sigmoid_s", "Input0"], [12, 1, 1, "c.fp_sigmoid_s", "core_mask"], [12, 1, 1, "c.fp_sigmoid_s", "length"], [12, 1, 1, "c.fp_sigmoid_s", "output"]], "fp_sigmoidcrossentropywithlogits_p": [[174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_p", "input0"], [174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_p", "input1"], [174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_p", "length"], [174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_p", "output"]], "fp_sigmoidcrossentropywithlogits_s": [[174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_s", "core_mask"], [174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_s", "input0"], [174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_s", "input1"], [174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_s", "length"], [174, 1, 1, "c.fp_sigmoidcrossentropywithlogits_s", "output"]], "fp_sigmoidcrossentropywithlogitsgrad_p": [[173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_p", "Input0"], [173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_p", "Input1"], [173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_p", "length"], [173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_p", "output"]], "fp_sigmoidcrossentropywithlogitsgrad_s": [[173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_s", "Input0"], [173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_s", "Input1"], [173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_s", "core_mask"], [173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_s", "length"], [173, 1, 1, "c.fp_sigmoidcrossentropywithlogitsgrad_s", "output"]], "fp_sin_p": [[175, 1, 1, "c.fp_sin_p", "dst_data"], [175, 1, 1, "c.fp_sin_p", "length"], [175, 1, 1, "c.fp_sin_p", "src_data"]], "fp_sin_s": [[175, 1, 1, "c.fp_sin_s", "core_mask"], [175, 1, 1, "c.fp_sin_s", "dst_data"], [175, 1, 1, "c.fp_sin_s", "length"], [175, 1, 1, "c.fp_sin_s", "src_data"]], "fp_slice_p": [[178, 1, 1, "c.fp_slice_p", "begin"], [178, 1, 1, "c.fp_slice_p", "input"], [178, 1, 1, "c.fp_slice_p", "input_shape"], [178, 1, 1, "c.fp_slice_p", "ndim"], [178, 1, 1, "c.fp_slice_p", "output"], [178, 1, 1, "c.fp_slice_p", "size"]], "fp_slice_s": [[178, 1, 1, "c.fp_slice_s", "begin"], [178, 1, 1, "c.fp_slice_s", "core_mask"], [178, 1, 1, "c.fp_slice_s", "input"], [178, 1, 1, "c.fp_slice_s", "input_shape"], [178, 1, 1, "c.fp_slice_s", "ndim"], [178, 1, 1, "c.fp_slice_s", "output"], [178, 1, 1, "c.fp_slice_s", "size"]], "fp_smoothl1loss_p": [[179, 1, 1, "c.fp_smoothl1loss_p", "beta"], [179, 1, 1, "c.fp_smoothl1loss_p", "length"], [179, 1, 1, "c.fp_smoothl1loss_p", "out"], [179, 1, 1, "c.fp_smoothl1loss_p", "predict"], [179, 1, 1, "c.fp_smoothl1loss_p", "target"]], "fp_smoothl1loss_s": [[179, 1, 1, "c.fp_smoothl1loss_s", "beta"], [179, 1, 1, "c.fp_smoothl1loss_s", "core_mask"], [179, 1, 1, "c.fp_smoothl1loss_s", "length"], [179, 1, 1, "c.fp_smoothl1loss_s", "out"], [179, 1, 1, "c.fp_smoothl1loss_s", "predict"], [179, 1, 1, "c.fp_smoothl1loss_s", "target"]], "fp_smoothl1lossgrad_p": [[180, 1, 1, "c.fp_smoothl1lossgrad_p", "beta"], [180, 1, 1, "c.fp_smoothl1lossgrad_p", "dx1"], [180, 1, 1, "c.fp_smoothl1lossgrad_p", "dy"], [180, 1, 1, "c.fp_smoothl1lossgrad_p", "length"], [180, 1, 1, "c.fp_smoothl1lossgrad_p", "x1"], [180, 1, 1, "c.fp_smoothl1lossgrad_p", "x2"]], "fp_smoothl1lossgrad_s": [[180, 1, 1, "c.fp_smoothl1lossgrad_s", "beta"], [180, 1, 1, "c.fp_smoothl1lossgrad_s", "core_mask"], [180, 1, 1, "c.fp_smoothl1lossgrad_s", "dx1"], [180, 1, 1, "c.fp_smoothl1lossgrad_s", "dy"], [180, 1, 1, "c.fp_smoothl1lossgrad_s", "length"], [180, 1, 1, "c.fp_smoothl1lossgrad_s", "x1"], [180, 1, 1, "c.fp_smoothl1lossgrad_s", "x2"]], "fp_softmax_cross_entropy_with_logits_p": [[182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "batch_size"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "grads"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "labels"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "logits"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "need_grads"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "num_of_classes"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "output"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "probs"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_p", "sum_data"]], "fp_softmax_cross_entropy_with_logits_s": [[182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "batch_size"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "core_mask"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "grads"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "labels"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "logits"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "need_grads"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "num_of_classes"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "output"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "probs"], [182, 1, 1, "c.fp_softmax_cross_entropy_with_logits_s", "sum_data"]], "fp_softmax_p": [[181, 1, 1, "c.fp_softmax_p", "axis"], [181, 1, 1, "c.fp_softmax_p", "axis_size"], [181, 1, 1, "c.fp_softmax_p", "inner_size"], [181, 1, 1, "c.fp_softmax_p", "input_ptr"], [181, 1, 1, "c.fp_softmax_p", "n_dim"], [181, 1, 1, "c.fp_softmax_p", "output_ptr"], [181, 1, 1, "c.fp_softmax_p", "outter_size"], [181, 1, 1, "c.fp_softmax_p", "sum_data"]], "fp_softmax_s": [[181, 1, 1, "c.fp_softmax_s", "axis"], [181, 1, 1, "c.fp_softmax_s", "axis_size"], [181, 1, 1, "c.fp_softmax_s", "core_mask"], [181, 1, 1, "c.fp_softmax_s", "inner_size"], [181, 1, 1, "c.fp_softmax_s", "input_ptr"], [181, 1, 1, "c.fp_softmax_s", "n_dim"], [181, 1, 1, "c.fp_softmax_s", "output_ptr"], [181, 1, 1, "c.fp_softmax_s", "outter_size"], [181, 1, 1, "c.fp_softmax_s", "sum_data"]], "fp_softplus_grad_p": [[13, 1, 1, "c.fp_softplus_grad_p", "dst"], [13, 1, 1, "c.fp_softplus_grad_p", "length"], [13, 1, 1, "c.fp_softplus_grad_p", "src0"], [13, 1, 1, "c.fp_softplus_grad_p", "src1"]], "fp_softplus_grad_s": [[13, 1, 1, "c.fp_softplus_grad_s", "core_mask"], [13, 1, 1, "c.fp_softplus_grad_s", "dst"], [13, 1, 1, "c.fp_softplus_grad_s", "length"], [13, 1, 1, "c.fp_softplus_grad_s", "src0"], [13, 1, 1, "c.fp_softplus_grad_s", "src1"]], "fp_softplus_p": [[12, 1, 1, "c.fp_softplus_p", "Input0"], [12, 1, 1, "c.fp_softplus_p", "length"], [12, 1, 1, "c.fp_softplus_p", "output"]], "fp_softplus_s": [[12, 1, 1, "c.fp_softplus_s", "Input0"], [12, 1, 1, "c.fp_softplus_s", "core_mask"], [12, 1, 1, "c.fp_softplus_s", "length"], [12, 1, 1, "c.fp_softplus_s", "output"]], "fp_softshrink_grad_p": [[13, 1, 1, "c.fp_softshrink_grad_p", "dst"], [13, 1, 1, "c.fp_softshrink_grad_p", "lambd"], [13, 1, 1, "c.fp_softshrink_grad_p", "length"], [13, 1, 1, "c.fp_softshrink_grad_p", "src0"], [13, 1, 1, "c.fp_softshrink_grad_p", "src1"]], "fp_softshrink_grad_s": [[13, 1, 1, "c.fp_softshrink_grad_s", "core_mask"], [13, 1, 1, "c.fp_softshrink_grad_s", "dst"], [13, 1, 1, "c.fp_softshrink_grad_s", "lambd"], [13, 1, 1, "c.fp_softshrink_grad_s", "length"], [13, 1, 1, "c.fp_softshrink_grad_s", "src0"], [13, 1, 1, "c.fp_softshrink_grad_s", "src1"]], "fp_softshrink_p": [[12, 1, 1, "c.fp_softshrink_p", "Input0"], [12, 1, 1, "c.fp_softshrink_p", "lambd"], [12, 1, 1, "c.fp_softshrink_p", "length"], [12, 1, 1, "c.fp_softshrink_p", "output"]], "fp_softshrink_s": [[12, 1, 1, "c.fp_softshrink_s", "Input0"], [12, 1, 1, "c.fp_softshrink_s", "core_mask"], [12, 1, 1, "c.fp_softshrink_s", "lambd"], [12, 1, 1, "c.fp_softshrink_s", "length"], [12, 1, 1, "c.fp_softshrink_s", "output"]], "fp_softsignopt_p": [[12, 1, 1, "c.fp_softsignopt_p", "Input0"], [12, 1, 1, "c.fp_softsignopt_p", "length"], [12, 1, 1, "c.fp_softsignopt_p", "output"]], "fp_softsignopt_s": [[12, 1, 1, "c.fp_softsignopt_s", "Input0"], [12, 1, 1, "c.fp_softsignopt_s", "core_mask"], [12, 1, 1, "c.fp_softsignopt_s", "length"], [12, 1, 1, "c.fp_softsignopt_s", "output"]], "fp_spacetobatch_p": [[183, 1, 1, "c.fp_spacetobatch_p", "block_size"], [183, 1, 1, "c.fp_spacetobatch_p", "data_size"], [183, 1, 1, "c.fp_spacetobatch_p", "input"], [183, 1, 1, "c.fp_spacetobatch_p", "input_shape"], [183, 1, 1, "c.fp_spacetobatch_p", "output"], [183, 1, 1, "c.fp_spacetobatch_p", "paddings"]], "fp_spacetobatch_s": [[183, 1, 1, "c.fp_spacetobatch_s", "block_size"], [183, 1, 1, "c.fp_spacetobatch_s", "core_mask"], [183, 1, 1, "c.fp_spacetobatch_s", "data_size"], [183, 1, 1, "c.fp_spacetobatch_s", "input"], [183, 1, 1, "c.fp_spacetobatch_s", "input_shape"], [183, 1, 1, "c.fp_spacetobatch_s", "output"], [183, 1, 1, "c.fp_spacetobatch_s", "paddings"]], "fp_spacetobatchnd_p": [[184, 1, 1, "c.fp_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.fp_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.fp_spacetobatchnd_p", "input"], [184, 1, 1, "c.fp_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.fp_spacetobatchnd_p", "output"], [184, 1, 1, "c.fp_spacetobatchnd_p", "paddings"]], "fp_spacetobatchnd_s": [[184, 1, 1, "c.fp_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.fp_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.fp_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.fp_spacetobatchnd_s", "input"], [184, 1, 1, "c.fp_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.fp_spacetobatchnd_s", "output"], [184, 1, 1, "c.fp_spacetobatchnd_s", "paddings"]], "fp_spacetodepth_p": [[185, 1, 1, "c.fp_spacetodepth_p", "block"], [185, 1, 1, "c.fp_spacetodepth_p", "data_size"], [185, 1, 1, "c.fp_spacetodepth_p", "in_shape"], [185, 1, 1, "c.fp_spacetodepth_p", "input"], [185, 1, 1, "c.fp_spacetodepth_p", "output"]], "fp_spacetodepth_s": [[185, 1, 1, "c.fp_spacetodepth_s", "block"], [185, 1, 1, "c.fp_spacetodepth_s", "core_mask"], [185, 1, 1, "c.fp_spacetodepth_s", "data_size"], [185, 1, 1, "c.fp_spacetodepth_s", "in_shape"], [185, 1, 1, "c.fp_spacetodepth_s", "input"], [185, 1, 1, "c.fp_spacetodepth_s", "output"]], "fp_sparse_softmax_cross_entropy_with_logits_p": [[186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "axis_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "batch_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "inner_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "input"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "is_grad"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "labels"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "losses"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "number_of_classes"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "output"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "outter_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "partial_losses"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_p", "sum_data"]], "fp_sparse_softmax_cross_entropy_with_logits_s": [[186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "axis_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "batch_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "core_mask"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "inner_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "input"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "is_grad"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "labels"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "losses"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "number_of_classes"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "output"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "outter_size"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "partial_losses"], [186, 1, 1, "c.fp_sparse_softmax_cross_entropy_with_logits_s", "sum_data"]], "fp_sparsefillemptyrows_p": [[187, 1, 1, "c.fp_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_p", "values_ptr"]], "fp_sparsefillemptyrows_s": [[187, 1, 1, "c.fp_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.fp_sparsefillemptyrows_s", "values_ptr"]], "fp_sparsesegmentsum_p": [[189, 1, 1, "c.fp_sparsesegmentsum_p", "in_data"], [189, 1, 1, "c.fp_sparsesegmentsum_p", "in_data_shape"], [189, 1, 1, "c.fp_sparsesegmentsum_p", "in_data_shape_size"], [189, 1, 1, "c.fp_sparsesegmentsum_p", "in_indices"], [189, 1, 1, "c.fp_sparsesegmentsum_p", "in_indices_size"], [189, 1, 1, "c.fp_sparsesegmentsum_p", "in_segment_ids"], [189, 1, 1, "c.fp_sparsesegmentsum_p", "out_data"], [189, 1, 1, "c.fp_sparsesegmentsum_p", "out_data_shape"]], "fp_sparsesegmentsum_s": [[189, 1, 1, "c.fp_sparsesegmentsum_s", "core_mask"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "in_data"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "in_data_shape"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "in_data_shape_size"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "in_indices"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "in_indices_size"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "in_segment_ids"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "out_data"], [189, 1, 1, "c.fp_sparsesegmentsum_s", "out_data_shape"]], "fp_sparsetodense_p": [[190, 1, 1, "c.fp_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.fp_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.fp_sparsetodense_p", "output"], [190, 1, 1, "c.fp_sparsetodense_p", "output_strides"], [190, 1, 1, "c.fp_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.fp_sparsetodense_p", "sparse_values"]], "fp_sparsetodense_s": [[190, 1, 1, "c.fp_sparsetodense_s", "core_mask"], [190, 1, 1, "c.fp_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.fp_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.fp_sparsetodense_s", "output"], [190, 1, 1, "c.fp_sparsetodense_s", "output_strides"], [190, 1, 1, "c.fp_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.fp_sparsetodense_s", "sparse_values"]], "fp_splice_p": [[191, 1, 1, "c.fp_splice_p", "context_dim"], [191, 1, 1, "c.fp_splice_p", "dst_col"], [191, 1, 1, "c.fp_splice_p", "dst_data"], [191, 1, 1, "c.fp_splice_p", "dst_row"], [191, 1, 1, "c.fp_splice_p", "forward_indexes"], [191, 1, 1, "c.fp_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.fp_splice_p", "src_col"], [191, 1, 1, "c.fp_splice_p", "src_data"], [191, 1, 1, "c.fp_splice_p", "src_row"]], "fp_splice_s": [[191, 1, 1, "c.fp_splice_s", "context_dim"], [191, 1, 1, "c.fp_splice_s", "core_mask"], [191, 1, 1, "c.fp_splice_s", "dst_col"], [191, 1, 1, "c.fp_splice_s", "dst_data"], [191, 1, 1, "c.fp_splice_s", "dst_row"], [191, 1, 1, "c.fp_splice_s", "forward_indexes"], [191, 1, 1, "c.fp_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.fp_splice_s", "src_col"], [191, 1, 1, "c.fp_splice_s", "src_data"], [191, 1, 1, "c.fp_splice_s", "src_row"]], "fp_split_p": [[192, 1, 1, "c.fp_split_p", "axis"], [192, 1, 1, "c.fp_split_p", "input"], [192, 1, 1, "c.fp_split_p", "input_ndim"], [192, 1, 1, "c.fp_split_p", "input_shape"], [192, 1, 1, "c.fp_split_p", "num_split"], [192, 1, 1, "c.fp_split_p", "outputs"], [192, 1, 1, "c.fp_split_p", "split_sizes"]], "fp_split_s": [[192, 1, 1, "c.fp_split_s", "axis"], [192, 1, 1, "c.fp_split_s", "core_mask"], [192, 1, 1, "c.fp_split_s", "input"], [192, 1, 1, "c.fp_split_s", "input_ndim"], [192, 1, 1, "c.fp_split_s", "input_shape"], [192, 1, 1, "c.fp_split_s", "num_split"], [192, 1, 1, "c.fp_split_s", "outputs"], [192, 1, 1, "c.fp_split_s", "split_sizes"]], "fp_split_with_overlap_p": [[193, 1, 1, "c.fp_split_with_overlap_p", "axis"], [193, 1, 1, "c.fp_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.fp_split_with_overlap_p", "input"], [193, 1, 1, "c.fp_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.fp_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.fp_split_with_overlap_p", "num_split"], [193, 1, 1, "c.fp_split_with_overlap_p", "outputs"], [193, 1, 1, "c.fp_split_with_overlap_p", "start_indices"]], "fp_split_with_overlap_s": [[193, 1, 1, "c.fp_split_with_overlap_s", "axis"], [193, 1, 1, "c.fp_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.fp_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.fp_split_with_overlap_s", "input"], [193, 1, 1, "c.fp_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.fp_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.fp_split_with_overlap_s", "num_split"], [193, 1, 1, "c.fp_split_with_overlap_s", "outputs"], [193, 1, 1, "c.fp_split_with_overlap_s", "start_indices"]], "fp_sqrt_p": [[194, 1, 1, "c.fp_sqrt_p", "dst_data"], [194, 1, 1, "c.fp_sqrt_p", "length"], [194, 1, 1, "c.fp_sqrt_p", "src_data"]], "fp_sqrt_s": [[194, 1, 1, "c.fp_sqrt_s", "core_mask"], [194, 1, 1, "c.fp_sqrt_s", "dst_data"], [194, 1, 1, "c.fp_sqrt_s", "length"], [194, 1, 1, "c.fp_sqrt_s", "src_data"]], "fp_sqrtgrad_p": [[195, 1, 1, "c.fp_sqrtgrad_p", "input1"], [195, 1, 1, "c.fp_sqrtgrad_p", "input2"], [195, 1, 1, "c.fp_sqrtgrad_p", "output"], [195, 1, 1, "c.fp_sqrtgrad_p", "size"]], "fp_sqrtgrad_s": [[195, 1, 1, "c.fp_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.fp_sqrtgrad_s", "input1"], [195, 1, 1, "c.fp_sqrtgrad_s", "input2"], [195, 1, 1, "c.fp_sqrtgrad_s", "output"], [195, 1, 1, "c.fp_sqrtgrad_s", "size"]], "fp_square_p": [[196, 1, 1, "c.fp_square_p", "dst"], [196, 1, 1, "c.fp_square_p", "length"], [196, 1, 1, "c.fp_square_p", "src"]], "fp_square_s": [[196, 1, 1, "c.fp_square_s", "core_mask"], [196, 1, 1, "c.fp_square_s", "dst"], [196, 1, 1, "c.fp_square_s", "length"], [196, 1, 1, "c.fp_square_s", "src"]], "fp_squaredifference_p": [[197, 1, 1, "c.fp_squaredifference_p", "input0"], [197, 1, 1, "c.fp_squaredifference_p", "input1"], [197, 1, 1, "c.fp_squaredifference_p", "length"], [197, 1, 1, "c.fp_squaredifference_p", "output"]], "fp_squaredifference_s": [[197, 1, 1, "c.fp_squaredifference_s", "core_mask"], [197, 1, 1, "c.fp_squaredifference_s", "input0"], [197, 1, 1, "c.fp_squaredifference_s", "input1"], [197, 1, 1, "c.fp_squaredifference_s", "length"], [197, 1, 1, "c.fp_squaredifference_s", "output"]], "fp_stack_p": [[199, 1, 1, "c.fp_stack_p", "axis"], [199, 1, 1, "c.fp_stack_p", "input_ndim"], [199, 1, 1, "c.fp_stack_p", "input_shape"], [199, 1, 1, "c.fp_stack_p", "inputs"], [199, 1, 1, "c.fp_stack_p", "num_inputs"], [199, 1, 1, "c.fp_stack_p", "output"]], "fp_stack_s": [[199, 1, 1, "c.fp_stack_s", "axis"], [199, 1, 1, "c.fp_stack_s", "core_mask"], [199, 1, 1, "c.fp_stack_s", "input_ndim"], [199, 1, 1, "c.fp_stack_s", "input_shape"], [199, 1, 1, "c.fp_stack_s", "inputs"], [199, 1, 1, "c.fp_stack_s", "num_inputs"], [199, 1, 1, "c.fp_stack_s", "output"]], "fp_stridedslicegrad_p": [[201, 1, 1, "c.fp_stridedslicegrad_p", "begins"], [201, 1, 1, "c.fp_stridedslicegrad_p", "dx_shape"], [201, 1, 1, "c.fp_stridedslicegrad_p", "in_shape"], [201, 1, 1, "c.fp_stridedslicegrad_p", "inputs"], [201, 1, 1, "c.fp_stridedslicegrad_p", "output"], [201, 1, 1, "c.fp_stridedslicegrad_p", "strides"]], "fp_stridedslicegrad_s": [[201, 1, 1, "c.fp_stridedslicegrad_s", "begins"], [201, 1, 1, "c.fp_stridedslicegrad_s", "core_mask"], [201, 1, 1, "c.fp_stridedslicegrad_s", "dx_shape"], [201, 1, 1, "c.fp_stridedslicegrad_s", "in_shape"], [201, 1, 1, "c.fp_stridedslicegrad_s", "inputs"], [201, 1, 1, "c.fp_stridedslicegrad_s", "output"], [201, 1, 1, "c.fp_stridedslicegrad_s", "strides"]], "fp_subext_p": [[202, 1, 1, "c.fp_subext_p", "alpha"], [202, 1, 1, "c.fp_subext_p", "input0"], [202, 1, 1, "c.fp_subext_p", "input1"], [202, 1, 1, "c.fp_subext_p", "output"], [202, 1, 1, "c.fp_subext_p", "size"]], "fp_subext_s": [[202, 1, 1, "c.fp_subext_s", "alpha"], [202, 1, 1, "c.fp_subext_s", "core_mask"], [202, 1, 1, "c.fp_subext_s", "input0"], [202, 1, 1, "c.fp_subext_s", "input1"], [202, 1, 1, "c.fp_subext_s", "output"], [202, 1, 1, "c.fp_subext_s", "size"]], "fp_subgrad_p": [[203, 1, 1, "c.fp_subgrad_p", "dx1"], [203, 1, 1, "c.fp_subgrad_p", "dx2"], [203, 1, 1, "c.fp_subgrad_p", "dy"], [203, 1, 1, "c.fp_subgrad_p", "dy_dims"], [203, 1, 1, "c.fp_subgrad_p", "num_dims"], [203, 1, 1, "c.fp_subgrad_p", "x1_dims"], [203, 1, 1, "c.fp_subgrad_p", "x2_dims"]], "fp_subgrad_s": [[203, 1, 1, "c.fp_subgrad_s", "core_mask"], [203, 1, 1, "c.fp_subgrad_s", "dx1"], [203, 1, 1, "c.fp_subgrad_s", "dx2"], [203, 1, 1, "c.fp_subgrad_s", "dy"], [203, 1, 1, "c.fp_subgrad_s", "dy_dims"], [203, 1, 1, "c.fp_subgrad_s", "num_dims"], [203, 1, 1, "c.fp_subgrad_s", "x1_dims"], [203, 1, 1, "c.fp_subgrad_s", "x2_dims"]], "fp_subrelu6_p": [[202, 1, 1, "c.fp_subrelu6_p", "input0"], [202, 1, 1, "c.fp_subrelu6_p", "input1"], [202, 1, 1, "c.fp_subrelu6_p", "output"], [202, 1, 1, "c.fp_subrelu6_p", "size"]], "fp_subrelu6_s": [[202, 1, 1, "c.fp_subrelu6_s", "core_mask"], [202, 1, 1, "c.fp_subrelu6_s", "input0"], [202, 1, 1, "c.fp_subrelu6_s", "input1"], [202, 1, 1, "c.fp_subrelu6_s", "output"], [202, 1, 1, "c.fp_subrelu6_s", "size"]], "fp_subrelu_p": [[202, 1, 1, "c.fp_subrelu_p", "input0"], [202, 1, 1, "c.fp_subrelu_p", "input1"], [202, 1, 1, "c.fp_subrelu_p", "output"], [202, 1, 1, "c.fp_subrelu_p", "size"]], "fp_subrelu_s": [[202, 1, 1, "c.fp_subrelu_s", "core_mask"], [202, 1, 1, "c.fp_subrelu_s", "input0"], [202, 1, 1, "c.fp_subrelu_s", "input1"], [202, 1, 1, "c.fp_subrelu_s", "output"], [202, 1, 1, "c.fp_subrelu_s", "size"]], "fp_swish_p": [[12, 1, 1, "c.fp_swish_p", "Input0"], [12, 1, 1, "c.fp_swish_p", "length"], [12, 1, 1, "c.fp_swish_p", "output"]], "fp_swish_s": [[12, 1, 1, "c.fp_swish_s", "Input0"], [12, 1, 1, "c.fp_swish_s", "core_mask"], [12, 1, 1, "c.fp_swish_s", "length"], [12, 1, 1, "c.fp_swish_s", "output"]], "fp_tanh_grad_p": [[13, 1, 1, "c.fp_tanh_grad_p", "dst"], [13, 1, 1, "c.fp_tanh_grad_p", "length"], [13, 1, 1, "c.fp_tanh_grad_p", "src0"], [13, 1, 1, "c.fp_tanh_grad_p", "src1"]], "fp_tanh_grad_s": [[13, 1, 1, "c.fp_tanh_grad_s", "core_mask"], [13, 1, 1, "c.fp_tanh_grad_s", "dst"], [13, 1, 1, "c.fp_tanh_grad_s", "length"], [13, 1, 1, "c.fp_tanh_grad_s", "src0"], [13, 1, 1, "c.fp_tanh_grad_s", "src1"]], "fp_tanh_p": [[12, 1, 1, "c.fp_tanh_p", "Input0"], [12, 1, 1, "c.fp_tanh_p", "length"], [12, 1, 1, "c.fp_tanh_p", "output"]], "fp_tanh_s": [[12, 1, 1, "c.fp_tanh_s", "Input0"], [12, 1, 1, "c.fp_tanh_s", "core_mask"], [12, 1, 1, "c.fp_tanh_s", "length"], [12, 1, 1, "c.fp_tanh_s", "output"]], "fp_tensor_scatter_add_p": [[206, 1, 1, "c.fp_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "input"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "output"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.fp_tensor_scatter_add_p", "updates"]], "fp_tensor_scatter_add_s": [[206, 1, 1, "c.fp_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "input"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "output"], [206, 1, 1, "c.fp_tensor_scatter_add_s", "updates"]], "fp_tensorarrayread_p": [[208, 1, 1, "c.fp_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.fp_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.fp_tensorarrayread_p", "index"], [208, 1, 1, "c.fp_tensorarrayread_p", "output_data"], [208, 1, 1, "c.fp_tensorarrayread_p", "output_size"]], "fp_tensorarrayread_s": [[208, 1, 1, "c.fp_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.fp_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.fp_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.fp_tensorarrayread_s", "index"], [208, 1, 1, "c.fp_tensorarrayread_s", "output_data"], [208, 1, 1, "c.fp_tensorarrayread_s", "output_size"]], "fp_tensorlistfromtensor_p": [[210, 1, 1, "c.fp_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.fp_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.fp_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.fp_tensorlistfromtensor_p", "output_tensors"]], "fp_tensorlistfromtensor_s": [[210, 1, 1, "c.fp_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.fp_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.fp_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.fp_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.fp_tensorlistfromtensor_s", "output_tensors"]], "fp_tile_p": [[215, 1, 1, "c.fp_tile_p", "input"], [215, 1, 1, "c.fp_tile_p", "input_shape"], [215, 1, 1, "c.fp_tile_p", "output"], [215, 1, 1, "c.fp_tile_p", "stride"], [215, 1, 1, "c.fp_tile_p", "tile_dim"], [215, 1, 1, "c.fp_tile_p", "tile_num"]], "fp_tile_s": [[215, 1, 1, "c.fp_tile_s", "core_mask"], [215, 1, 1, "c.fp_tile_s", "input"], [215, 1, 1, "c.fp_tile_s", "input_shape"], [215, 1, 1, "c.fp_tile_s", "output"], [215, 1, 1, "c.fp_tile_s", "stride"], [215, 1, 1, "c.fp_tile_s", "tile_dim"], [215, 1, 1, "c.fp_tile_s", "tile_num"]], "fp_to_i8_quant_p": [[146, 1, 1, "c.fp_to_i8_quant_p", "input"], [146, 1, 1, "c.fp_to_i8_quant_p", "length"], [146, 1, 1, "c.fp_to_i8_quant_p", "output"], [146, 1, 1, "c.fp_to_i8_quant_p", "scale"], [146, 1, 1, "c.fp_to_i8_quant_p", "zp"]], "fp_to_i8_quant_s": [[146, 1, 1, "c.fp_to_i8_quant_s", "core_mask"], [146, 1, 1, "c.fp_to_i8_quant_s", "input"], [146, 1, 1, "c.fp_to_i8_quant_s", "length"], [146, 1, 1, "c.fp_to_i8_quant_s", "output"], [146, 1, 1, "c.fp_to_i8_quant_s", "scale"], [146, 1, 1, "c.fp_to_i8_quant_s", "zp"]], "fp_topk_fusion_p": [[216, 1, 1, "c.fp_topk_fusion_p", "input"], [216, 1, 1, "c.fp_topk_fusion_p", "output"], [216, 1, 1, "c.fp_topk_fusion_p", "output_index"], [216, 1, 1, "c.fp_topk_fusion_p", "parameter"]], "fp_topk_fusion_s": [[216, 1, 1, "c.fp_topk_fusion_s", "core_mask"], [216, 1, 1, "c.fp_topk_fusion_s", "input"], [216, 1, 1, "c.fp_topk_fusion_s", "output"], [216, 1, 1, "c.fp_topk_fusion_s", "output_index"], [216, 1, 1, "c.fp_topk_fusion_s", "parameter"]], "fp_transpose_p": [[217, 1, 1, "c.fp_transpose_p", "in_data"], [217, 1, 1, "c.fp_transpose_p", "num_axes"], [217, 1, 1, "c.fp_transpose_p", "out_data"], [217, 1, 1, "c.fp_transpose_p", "out_strides"], [217, 1, 1, "c.fp_transpose_p", "output_shape"], [217, 1, 1, "c.fp_transpose_p", "perm"], [217, 1, 1, "c.fp_transpose_p", "strides"]], "fp_transpose_s": [[217, 1, 1, "c.fp_transpose_s", "core_mask"], [217, 1, 1, "c.fp_transpose_s", "in_data"], [217, 1, 1, "c.fp_transpose_s", "num_axes"], [217, 1, 1, "c.fp_transpose_s", "out_data"], [217, 1, 1, "c.fp_transpose_s", "out_strides"], [217, 1, 1, "c.fp_transpose_s", "output_shape"], [217, 1, 1, "c.fp_transpose_s", "perm"], [217, 1, 1, "c.fp_transpose_s", "strides"]], "fp_tril_p": [[218, 1, 1, "c.fp_tril_p", "dst"], [218, 1, 1, "c.fp_tril_p", "height"], [218, 1, 1, "c.fp_tril_p", "k"], [218, 1, 1, "c.fp_tril_p", "out_elems"], [218, 1, 1, "c.fp_tril_p", "src"], [218, 1, 1, "c.fp_tril_p", "width"]], "fp_tril_s": [[218, 1, 1, "c.fp_tril_s", "core_mask"], [218, 1, 1, "c.fp_tril_s", "dst"], [218, 1, 1, "c.fp_tril_s", "height"], [218, 1, 1, "c.fp_tril_s", "k"], [218, 1, 1, "c.fp_tril_s", "out_elems"], [218, 1, 1, "c.fp_tril_s", "src"], [218, 1, 1, "c.fp_tril_s", "width"]], "fp_triu_p": [[219, 1, 1, "c.fp_triu_p", "dst"], [219, 1, 1, "c.fp_triu_p", "height"], [219, 1, 1, "c.fp_triu_p", "k"], [219, 1, 1, "c.fp_triu_p", "out_elems"], [219, 1, 1, "c.fp_triu_p", "src"], [219, 1, 1, "c.fp_triu_p", "width"]], "fp_triu_s": [[219, 1, 1, "c.fp_triu_s", "core_mask"], [219, 1, 1, "c.fp_triu_s", "dst"], [219, 1, 1, "c.fp_triu_s", "height"], [219, 1, 1, "c.fp_triu_s", "k"], [219, 1, 1, "c.fp_triu_s", "out_elems"], [219, 1, 1, "c.fp_triu_s", "src"], [219, 1, 1, "c.fp_triu_s", "width"]], "fp_uniform_real_p": [[220, 1, 1, "c.fp_uniform_real_p", "length"], [220, 1, 1, "c.fp_uniform_real_p", "output"], [220, 1, 1, "c.fp_uniform_real_p", "seed"], [220, 1, 1, "c.fp_uniform_real_p", "seed2"]], "fp_uniform_real_s": [[220, 1, 1, "c.fp_uniform_real_s", "core_mask"], [220, 1, 1, "c.fp_uniform_real_s", "length"], [220, 1, 1, "c.fp_uniform_real_s", "output"], [220, 1, 1, "c.fp_uniform_real_s", "seed"], [220, 1, 1, "c.fp_uniform_real_s", "seed2"]], "fp_unsorted_segment_sum_p": [[222, 1, 1, "c.fp_unsorted_segment_sum_p", "dim0"], [222, 1, 1, "c.fp_unsorted_segment_sum_p", "dim1"], [222, 1, 1, "c.fp_unsorted_segment_sum_p", "id_max"], [222, 1, 1, "c.fp_unsorted_segment_sum_p", "index"], [222, 1, 1, "c.fp_unsorted_segment_sum_p", "input"], [222, 1, 1, "c.fp_unsorted_segment_sum_p", "output"]], "fp_unsorted_segment_sum_s": [[222, 1, 1, "c.fp_unsorted_segment_sum_s", "core_mask"], [222, 1, 1, "c.fp_unsorted_segment_sum_s", "dim0"], [222, 1, 1, "c.fp_unsorted_segment_sum_s", "dim1"], [222, 1, 1, "c.fp_unsorted_segment_sum_s", "id_max"], [222, 1, 1, "c.fp_unsorted_segment_sum_s", "index"], [222, 1, 1, "c.fp_unsorted_segment_sum_s", "input"], [222, 1, 1, "c.fp_unsorted_segment_sum_s", "output"]], "fp_where_p": [[225, 1, 1, "c.fp_where_p", "condition"], [225, 1, 1, "c.fp_where_p", "input0"], [225, 1, 1, "c.fp_where_p", "input1"], [225, 1, 1, "c.fp_where_p", "length"], [225, 1, 1, "c.fp_where_p", "output"]], "fp_where_s": [[225, 1, 1, "c.fp_where_s", "condition"], [225, 1, 1, "c.fp_where_s", "core_mask"], [225, 1, 1, "c.fp_where_s", "input0"], [225, 1, 1, "c.fp_where_s", "input1"], [225, 1, 1, "c.fp_where_s", "length"], [225, 1, 1, "c.fp_where_s", "output"]], "fp_zerolike_p": [[226, 1, 1, "c.fp_zerolike_p", "length"], [226, 1, 1, "c.fp_zerolike_p", "output"]], "fp_zerolike_s": [[226, 1, 1, "c.fp_zerolike_s", "core_mask"], [226, 1, 1, "c.fp_zerolike_s", "length"], [226, 1, 1, "c.fp_zerolike_s", "output"]], "hp_Gru_p": [[95, 1, 1, "c.hp_Gru_p", "buffer"], [95, 1, 1, "c.hp_Gru_p", "core_mask"], [95, 1, 1, "c.hp_Gru_p", "gru_param"], [95, 1, 1, "c.hp_Gru_p", "hidden_state"], [95, 1, 1, "c.hp_Gru_p", "input"], [95, 1, 1, "c.hp_Gru_p", "input_bias"], [95, 1, 1, "c.hp_Gru_p", "output"], [95, 1, 1, "c.hp_Gru_p", "state_bias"], [95, 1, 1, "c.hp_Gru_p", "weight_g"], [95, 1, 1, "c.hp_Gru_p", "weight_r"]], "hp_Gru_s": [[95, 1, 1, "c.hp_Gru_s", "buffer"], [95, 1, 1, "c.hp_Gru_s", "core_mask"], [95, 1, 1, "c.hp_Gru_s", "gru_param"], [95, 1, 1, "c.hp_Gru_s", "hidden_state"], [95, 1, 1, "c.hp_Gru_s", "input"], [95, 1, 1, "c.hp_Gru_s", "input_bias"], [95, 1, 1, "c.hp_Gru_s", "output"], [95, 1, 1, "c.hp_Gru_s", "state_bias"], [95, 1, 1, "c.hp_Gru_s", "weight_g"], [95, 1, 1, "c.hp_Gru_s", "weight_r"]], "hp_Lstm_p": [[117, 1, 1, "c.hp_Lstm_p", "buffer"], [117, 1, 1, "c.hp_Lstm_p", "cell_state"], [117, 1, 1, "c.hp_Lstm_p", "hidden_state"], [117, 1, 1, "c.hp_Lstm_p", "input"], [117, 1, 1, "c.hp_Lstm_p", "input_bias"], [117, 1, 1, "c.hp_Lstm_p", "lstm_param"], [117, 1, 1, "c.hp_Lstm_p", "output"], [117, 1, 1, "c.hp_Lstm_p", "state_bias"], [117, 1, 1, "c.hp_Lstm_p", "weight_h"], [117, 1, 1, "c.hp_Lstm_p", "weight_i"]], "hp_Lstm_s": [[117, 1, 1, "c.hp_Lstm_s", "buffer"], [117, 1, 1, "c.hp_Lstm_s", "cell_state"], [117, 1, 1, "c.hp_Lstm_s", "core_mask"], [117, 1, 1, "c.hp_Lstm_s", "hidden_state"], [117, 1, 1, "c.hp_Lstm_s", "input"], [117, 1, 1, "c.hp_Lstm_s", "input_bias"], [117, 1, 1, "c.hp_Lstm_s", "lstm_param"], [117, 1, 1, "c.hp_Lstm_s", "output"], [117, 1, 1, "c.hp_Lstm_s", "state_bias"], [117, 1, 1, "c.hp_Lstm_s", "weight_h"], [117, 1, 1, "c.hp_Lstm_s", "weight_i"]], "hp_QuantData_p": [[66, 1, 1, "c.hp_QuantData_p", "axis_num"], [66, 1, 1, "c.hp_QuantData_p", "element_num"], [66, 1, 1, "c.hp_QuantData_p", "quant_values"], [66, 1, 1, "c.hp_QuantData_p", "real_values"], [66, 1, 1, "c.hp_QuantData_p", "scale"], [66, 1, 1, "c.hp_QuantData_p", "segment_num"], [66, 1, 1, "c.hp_QuantData_p", "zp"]], "hp_QuantData_s": [[66, 1, 1, "c.hp_QuantData_s", "axis_num"], [66, 1, 1, "c.hp_QuantData_s", "core_mask"], [66, 1, 1, "c.hp_QuantData_s", "element_num"], [66, 1, 1, "c.hp_QuantData_s", "quant_values"], [66, 1, 1, "c.hp_QuantData_s", "real_values"], [66, 1, 1, "c.hp_QuantData_s", "scale"], [66, 1, 1, "c.hp_QuantData_s", "segment_num"], [66, 1, 1, "c.hp_QuantData_s", "zp"]], "hp_Unique_p": [[221, 1, 1, "c.hp_Unique_p", "input"], [221, 1, 1, "c.hp_Unique_p", "input_len"], [221, 1, 1, "c.hp_Unique_p", "output0"], [221, 1, 1, "c.hp_Unique_p", "output0_len"]], "hp_Unique_s": [[221, 1, 1, "c.hp_Unique_s", "core_mask"], [221, 1, 1, "c.hp_Unique_s", "input"], [221, 1, 1, "c.hp_Unique_s", "input_len"], [221, 1, 1, "c.hp_Unique_s", "output0"], [221, 1, 1, "c.hp_Unique_s", "output0_len"]], "hp_abs_p": [[10, 1, 1, "c.hp_abs_p", "dst_data"], [10, 1, 1, "c.hp_abs_p", "length"], [10, 1, 1, "c.hp_abs_p", "src_data"]], "hp_abs_s": [[10, 1, 1, "c.hp_abs_s", "core_mask"], [10, 1, 1, "c.hp_abs_s", "dst_data"], [10, 1, 1, "c.hp_abs_s", "length"], [10, 1, 1, "c.hp_abs_s", "src_data"]], "hp_absgrad_p": [[11, 1, 1, "c.hp_absgrad_p", "input0"], [11, 1, 1, "c.hp_absgrad_p", "input1"], [11, 1, 1, "c.hp_absgrad_p", "output"], [11, 1, 1, "c.hp_absgrad_p", "size"]], "hp_absgrad_s": [[11, 1, 1, "c.hp_absgrad_s", "core_mask"], [11, 1, 1, "c.hp_absgrad_s", "input0"], [11, 1, 1, "c.hp_absgrad_s", "input1"], [11, 1, 1, "c.hp_absgrad_s", "output"], [11, 1, 1, "c.hp_absgrad_s", "size"]], "hp_adamweightdecay_p": [[15, 1, 1, "c.hp_adamweightdecay_p", "beta1"], [15, 1, 1, "c.hp_adamweightdecay_p", "beta2"], [15, 1, 1, "c.hp_adamweightdecay_p", "decay"], [15, 1, 1, "c.hp_adamweightdecay_p", "epsilon"], [15, 1, 1, "c.hp_adamweightdecay_p", "gradient"], [15, 1, 1, "c.hp_adamweightdecay_p", "length"], [15, 1, 1, "c.hp_adamweightdecay_p", "lr"], [15, 1, 1, "c.hp_adamweightdecay_p", "m"], [15, 1, 1, "c.hp_adamweightdecay_p", "v"], [15, 1, 1, "c.hp_adamweightdecay_p", "var"]], "hp_adamweightdecay_s": [[15, 1, 1, "c.hp_adamweightdecay_s", "beta1"], [15, 1, 1, "c.hp_adamweightdecay_s", "beta2"], [15, 1, 1, "c.hp_adamweightdecay_s", "core_mask"], [15, 1, 1, "c.hp_adamweightdecay_s", "decay"], [15, 1, 1, "c.hp_adamweightdecay_s", "end"], [15, 1, 1, "c.hp_adamweightdecay_s", "epsilon"], [15, 1, 1, "c.hp_adamweightdecay_s", "gradient"], [15, 1, 1, "c.hp_adamweightdecay_s", "lr"], [15, 1, 1, "c.hp_adamweightdecay_s", "m"], [15, 1, 1, "c.hp_adamweightdecay_s", "start"], [15, 1, 1, "c.hp_adamweightdecay_s", "v"], [15, 1, 1, "c.hp_adamweightdecay_s", "var"]], "hp_adder_p": [[16, 1, 1, "c.hp_adder_p", "bias"], [16, 1, 1, "c.hp_adder_p", "conv_param"], [16, 1, 1, "c.hp_adder_p", "core_mask"], [16, 1, 1, "c.hp_adder_p", "input_w"], [16, 1, 1, "c.hp_adder_p", "input_x"], [16, 1, 1, "c.hp_adder_p", "out_y"]], "hp_adder_s": [[16, 1, 1, "c.hp_adder_s", "bias"], [16, 1, 1, "c.hp_adder_s", "core_mask"], [16, 1, 1, "c.hp_adder_s", "input_w"], [16, 1, 1, "c.hp_adder_s", "input_x"], [16, 1, 1, "c.hp_adder_s", "out_y"], [16, 1, 1, "c.hp_adder_s", "param"]], "hp_addext_p": [[17, 1, 1, "c.hp_addext_p", "alpha"], [17, 1, 1, "c.hp_addext_p", "in0"], [17, 1, 1, "c.hp_addext_p", "in1"], [17, 1, 1, "c.hp_addext_p", "out"], [17, 1, 1, "c.hp_addext_p", "size"]], "hp_addext_s": [[17, 1, 1, "c.hp_addext_s", "alpha"], [17, 1, 1, "c.hp_addext_s", "core_mask"], [17, 1, 1, "c.hp_addext_s", "in0"], [17, 1, 1, "c.hp_addext_s", "in1"], [17, 1, 1, "c.hp_addext_s", "out"], [17, 1, 1, "c.hp_addext_s", "size"]], "hp_addgrad_p": [[18, 1, 1, "c.hp_addgrad_p", "dx1"], [18, 1, 1, "c.hp_addgrad_p", "dx2"], [18, 1, 1, "c.hp_addgrad_p", "dy"], [18, 1, 1, "c.hp_addgrad_p", "dy_dims"], [18, 1, 1, "c.hp_addgrad_p", "num_dims"], [18, 1, 1, "c.hp_addgrad_p", "x1_dims"], [18, 1, 1, "c.hp_addgrad_p", "x2_dims"]], "hp_addgrad_s": [[18, 1, 1, "c.hp_addgrad_s", "core_mask"], [18, 1, 1, "c.hp_addgrad_s", "dx1"], [18, 1, 1, "c.hp_addgrad_s", "dx2"], [18, 1, 1, "c.hp_addgrad_s", "dy"], [18, 1, 1, "c.hp_addgrad_s", "dy_dims"], [18, 1, 1, "c.hp_addgrad_s", "num_dims"], [18, 1, 1, "c.hp_addgrad_s", "x1_dims"], [18, 1, 1, "c.hp_addgrad_s", "x2_dims"]], "hp_addrelu6_p": [[17, 1, 1, "c.hp_addrelu6_p", "in0"], [17, 1, 1, "c.hp_addrelu6_p", "in1"], [17, 1, 1, "c.hp_addrelu6_p", "out"], [17, 1, 1, "c.hp_addrelu6_p", "size"]], "hp_addrelu6_s": [[17, 1, 1, "c.hp_addrelu6_s", "core_mask"], [17, 1, 1, "c.hp_addrelu6_s", "in0"], [17, 1, 1, "c.hp_addrelu6_s", "in1"], [17, 1, 1, "c.hp_addrelu6_s", "out"], [17, 1, 1, "c.hp_addrelu6_s", "size"]], "hp_addrelu_p": [[17, 1, 1, "c.hp_addrelu_p", "in0"], [17, 1, 1, "c.hp_addrelu_p", "in1"], [17, 1, 1, "c.hp_addrelu_p", "out"], [17, 1, 1, "c.hp_addrelu_p", "size"]], "hp_addrelu_s": [[17, 1, 1, "c.hp_addrelu_s", "core_mask"], [17, 1, 1, "c.hp_addrelu_s", "in0"], [17, 1, 1, "c.hp_addrelu_s", "in1"], [17, 1, 1, "c.hp_addrelu_s", "out"], [17, 1, 1, "c.hp_addrelu_s", "size"]], "hp_affine_p": [[20, 1, 1, "c.hp_affine_p", "params"]], "hp_affine_s": [[20, 1, 1, "c.hp_affine_s", "core_mask"], [20, 1, 1, "c.hp_affine_s", "params"]], "hp_and_p": [[112, 1, 1, "c.hp_and_p", "input0"], [112, 1, 1, "c.hp_and_p", "input1"], [112, 1, 1, "c.hp_and_p", "length"], [112, 1, 1, "c.hp_and_p", "output"]], "hp_and_s": [[112, 1, 1, "c.hp_and_s", "core_mask"], [112, 1, 1, "c.hp_and_s", "input0"], [112, 1, 1, "c.hp_and_s", "input1"], [112, 1, 1, "c.hp_and_s", "length"], [112, 1, 1, "c.hp_and_s", "output"]], "hp_applymomentum_p": [[23, 1, 1, "c.hp_applymomentum_p", "accumulate"], [23, 1, 1, "c.hp_applymomentum_p", "gradient"], [23, 1, 1, "c.hp_applymomentum_p", "learning_rate"], [23, 1, 1, "c.hp_applymomentum_p", "length"], [23, 1, 1, "c.hp_applymomentum_p", "moment"], [23, 1, 1, "c.hp_applymomentum_p", "nesterov"], [23, 1, 1, "c.hp_applymomentum_p", "weight"]], "hp_applymomentum_s": [[23, 1, 1, "c.hp_applymomentum_s", "accumulate"], [23, 1, 1, "c.hp_applymomentum_s", "core_mask"], [23, 1, 1, "c.hp_applymomentum_s", "end"], [23, 1, 1, "c.hp_applymomentum_s", "gradient"], [23, 1, 1, "c.hp_applymomentum_s", "learning_rate"], [23, 1, 1, "c.hp_applymomentum_s", "moment"], [23, 1, 1, "c.hp_applymomentum_s", "nesterov"], [23, 1, 1, "c.hp_applymomentum_s", "start"], [23, 1, 1, "c.hp_applymomentum_s", "weight"]], "hp_argmax_p": [[24, 1, 1, "c.hp_argmax_p", "arg_elements"], [24, 1, 1, "c.hp_argmax_p", "axis"], [24, 1, 1, "c.hp_argmax_p", "in_shape"], [24, 1, 1, "c.hp_argmax_p", "in_strides"], [24, 1, 1, "c.hp_argmax_p", "index"], [24, 1, 1, "c.hp_argmax_p", "input"], [24, 1, 1, "c.hp_argmax_p", "input_shape_size"], [24, 1, 1, "c.hp_argmax_p", "out_strides"], [24, 1, 1, "c.hp_argmax_p", "out_value"], [24, 1, 1, "c.hp_argmax_p", "output"], [24, 1, 1, "c.hp_argmax_p", "output_value"], [24, 1, 1, "c.hp_argmax_p", "topk"]], "hp_argmax_s": [[24, 1, 1, "c.hp_argmax_s", "arg_elements"], [24, 1, 1, "c.hp_argmax_s", "axis"], [24, 1, 1, "c.hp_argmax_s", "core_mask"], [24, 1, 1, "c.hp_argmax_s", "in_shape"], [24, 1, 1, "c.hp_argmax_s", "in_strides"], [24, 1, 1, "c.hp_argmax_s", "index"], [24, 1, 1, "c.hp_argmax_s", "input"], [24, 1, 1, "c.hp_argmax_s", "input_shape_size"], [24, 1, 1, "c.hp_argmax_s", "out_strides"], [24, 1, 1, "c.hp_argmax_s", "out_value"], [24, 1, 1, "c.hp_argmax_s", "output"], [24, 1, 1, "c.hp_argmax_s", "output_value"], [24, 1, 1, "c.hp_argmax_s", "topk"]], "hp_argmin_p": [[25, 1, 1, "c.hp_argmin_p", "arg_elements"], [25, 1, 1, "c.hp_argmin_p", "axis"], [25, 1, 1, "c.hp_argmin_p", "in_shape"], [25, 1, 1, "c.hp_argmin_p", "in_strides"], [25, 1, 1, "c.hp_argmin_p", "index"], [25, 1, 1, "c.hp_argmin_p", "input"], [25, 1, 1, "c.hp_argmin_p", "input_shape_size"], [25, 1, 1, "c.hp_argmin_p", "out_strides"], [25, 1, 1, "c.hp_argmin_p", "out_value"], [25, 1, 1, "c.hp_argmin_p", "output"], [25, 1, 1, "c.hp_argmin_p", "output_value"], [25, 1, 1, "c.hp_argmin_p", "topk"]], "hp_argmin_s": [[25, 1, 1, "c.hp_argmin_s", "arg_elements"], [25, 1, 1, "c.hp_argmin_s", "axis"], [25, 1, 1, "c.hp_argmin_s", "core_mask"], [25, 1, 1, "c.hp_argmin_s", "in_shape"], [25, 1, 1, "c.hp_argmin_s", "in_strides"], [25, 1, 1, "c.hp_argmin_s", "index"], [25, 1, 1, "c.hp_argmin_s", "input"], [25, 1, 1, "c.hp_argmin_s", "input_shape_size"], [25, 1, 1, "c.hp_argmin_s", "out_strides"], [25, 1, 1, "c.hp_argmin_s", "out_value"], [25, 1, 1, "c.hp_argmin_s", "output"], [25, 1, 1, "c.hp_argmin_s", "output_value"], [25, 1, 1, "c.hp_argmin_s", "topk"]], "hp_assign_p": [[27, 1, 1, "c.hp_assign_p", "dst"], [27, 1, 1, "c.hp_assign_p", "length"], [27, 1, 1, "c.hp_assign_p", "src"]], "hp_assign_s": [[27, 1, 1, "c.hp_assign_s", "core_mask"], [27, 1, 1, "c.hp_assign_s", "dst"], [27, 1, 1, "c.hp_assign_s", "length"], [27, 1, 1, "c.hp_assign_s", "src"]], "hp_assignadd_p": [[28, 1, 1, "c.hp_assignadd_p", "input"], [28, 1, 1, "c.hp_assignadd_p", "length"], [28, 1, 1, "c.hp_assignadd_p", "output"]], "hp_assignadd_s": [[28, 1, 1, "c.hp_assignadd_s", "core_mask"], [28, 1, 1, "c.hp_assignadd_s", "input"], [28, 1, 1, "c.hp_assignadd_s", "length"], [28, 1, 1, "c.hp_assignadd_s", "output"]], "hp_avgpool_fusion_p": [[31, 1, 1, "c.hp_avgpool_fusion_p", "batch"], [31, 1, 1, "c.hp_avgpool_fusion_p", "channel"], [31, 1, 1, "c.hp_avgpool_fusion_p", "in_h"], [31, 1, 1, "c.hp_avgpool_fusion_p", "in_w"], [31, 1, 1, "c.hp_avgpool_fusion_p", "input"], [31, 1, 1, "c.hp_avgpool_fusion_p", "max_val"], [31, 1, 1, "c.hp_avgpool_fusion_p", "min_val"], [31, 1, 1, "c.hp_avgpool_fusion_p", "output"], [31, 1, 1, "c.hp_avgpool_fusion_p", "pad_bottom"], [31, 1, 1, "c.hp_avgpool_fusion_p", "pad_left"], [31, 1, 1, "c.hp_avgpool_fusion_p", "pad_right"], [31, 1, 1, "c.hp_avgpool_fusion_p", "pad_top"], [31, 1, 1, "c.hp_avgpool_fusion_p", "stride_h"], [31, 1, 1, "c.hp_avgpool_fusion_p", "stride_w"], [31, 1, 1, "c.hp_avgpool_fusion_p", "win_h"], [31, 1, 1, "c.hp_avgpool_fusion_p", "win_w"]], "hp_avgpool_fusion_s": [[31, 1, 1, "c.hp_avgpool_fusion_s", "batch"], [31, 1, 1, "c.hp_avgpool_fusion_s", "channel"], [31, 1, 1, "c.hp_avgpool_fusion_s", "core_mask"], [31, 1, 1, "c.hp_avgpool_fusion_s", "in_h"], [31, 1, 1, "c.hp_avgpool_fusion_s", "in_w"], [31, 1, 1, "c.hp_avgpool_fusion_s", "input"], [31, 1, 1, "c.hp_avgpool_fusion_s", "max_val"], [31, 1, 1, "c.hp_avgpool_fusion_s", "min_val"], [31, 1, 1, "c.hp_avgpool_fusion_s", "output"], [31, 1, 1, "c.hp_avgpool_fusion_s", "pad_bottom"], [31, 1, 1, "c.hp_avgpool_fusion_s", "pad_left"], [31, 1, 1, "c.hp_avgpool_fusion_s", "pad_right"], [31, 1, 1, "c.hp_avgpool_fusion_s", "pad_top"], [31, 1, 1, "c.hp_avgpool_fusion_s", "stride_h"], [31, 1, 1, "c.hp_avgpool_fusion_s", "stride_w"], [31, 1, 1, "c.hp_avgpool_fusion_s", "win_h"], [31, 1, 1, "c.hp_avgpool_fusion_s", "win_w"]], "hp_avgpoolinggrad_p": [[32, 1, 1, "c.hp_avgpoolinggrad_p", "batch"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "channel"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "end_idx"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "input"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "input_h"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "input_w"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "output"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "output_h"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "output_w"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "pad_l"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "pad_u"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "start_idx"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "stride_h"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "stride_w"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "window_h"], [32, 1, 1, "c.hp_avgpoolinggrad_p", "window_w"]], "hp_avgpoolinggrad_s": [[32, 1, 1, "c.hp_avgpoolinggrad_s", "batch"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "channel"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "core_mask"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "input"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "input_h"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "input_w"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "output"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "output_h"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "output_w"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "pad_l"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "pad_u"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "stride_h"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "stride_w"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "window_h"], [32, 1, 1, "c.hp_avgpoolinggrad_s", "window_w"]], "hp_batchnorm_p": [[33, 1, 1, "c.hp_batchnorm_p", "channel"], [33, 1, 1, "c.hp_batchnorm_p", "epsilon"], [33, 1, 1, "c.hp_batchnorm_p", "input"], [33, 1, 1, "c.hp_batchnorm_p", "mean"], [33, 1, 1, "c.hp_batchnorm_p", "output"], [33, 1, 1, "c.hp_batchnorm_p", "unit"], [33, 1, 1, "c.hp_batchnorm_p", "variance"]], "hp_batchnorm_s": [[33, 1, 1, "c.hp_batchnorm_s", "channel"], [33, 1, 1, "c.hp_batchnorm_s", "core_mask"], [33, 1, 1, "c.hp_batchnorm_s", "epsilon"], [33, 1, 1, "c.hp_batchnorm_s", "input"], [33, 1, 1, "c.hp_batchnorm_s", "mean"], [33, 1, 1, "c.hp_batchnorm_s", "output"], [33, 1, 1, "c.hp_batchnorm_s", "unit"], [33, 1, 1, "c.hp_batchnorm_s", "variance"]], "hp_batchnormgrad_p": [[34, 1, 1, "c.hp_batchnormgrad_p", "batch"], [34, 1, 1, "c.hp_batchnormgrad_p", "channel"], [34, 1, 1, "c.hp_batchnormgrad_p", "dbias"], [34, 1, 1, "c.hp_batchnormgrad_p", "dscale"], [34, 1, 1, "c.hp_batchnormgrad_p", "dx"], [34, 1, 1, "c.hp_batchnormgrad_p", "dy"], [34, 1, 1, "c.hp_batchnormgrad_p", "invar"], [34, 1, 1, "c.hp_batchnormgrad_p", "is_train"], [34, 1, 1, "c.hp_batchnormgrad_p", "mean"], [34, 1, 1, "c.hp_batchnormgrad_p", "scale"], [34, 1, 1, "c.hp_batchnormgrad_p", "x"]], "hp_batchnormgrad_s": [[34, 1, 1, "c.hp_batchnormgrad_s", "batch"], [34, 1, 1, "c.hp_batchnormgrad_s", "channel"], [34, 1, 1, "c.hp_batchnormgrad_s", "core_mask"], [34, 1, 1, "c.hp_batchnormgrad_s", "dbias"], [34, 1, 1, "c.hp_batchnormgrad_s", "dscale"], [34, 1, 1, "c.hp_batchnormgrad_s", "dx"], [34, 1, 1, "c.hp_batchnormgrad_s", "dy"], [34, 1, 1, "c.hp_batchnormgrad_s", "invar"], [34, 1, 1, "c.hp_batchnormgrad_s", "is_train"], [34, 1, 1, "c.hp_batchnormgrad_s", "mean"], [34, 1, 1, "c.hp_batchnormgrad_s", "scale"], [34, 1, 1, "c.hp_batchnormgrad_s", "x"]], "hp_batchtospace_p": [[35, 1, 1, "c.hp_batchtospace_p", "block_size"], [35, 1, 1, "c.hp_batchtospace_p", "crops"], [35, 1, 1, "c.hp_batchtospace_p", "data_size"], [35, 1, 1, "c.hp_batchtospace_p", "input"], [35, 1, 1, "c.hp_batchtospace_p", "input_shape"], [35, 1, 1, "c.hp_batchtospace_p", "output"]], "hp_batchtospace_s": [[35, 1, 1, "c.hp_batchtospace_s", "block_size"], [35, 1, 1, "c.hp_batchtospace_s", "core_mask"], [35, 1, 1, "c.hp_batchtospace_s", "crops"], [35, 1, 1, "c.hp_batchtospace_s", "data_size"], [35, 1, 1, "c.hp_batchtospace_s", "input"], [35, 1, 1, "c.hp_batchtospace_s", "input_shape"], [35, 1, 1, "c.hp_batchtospace_s", "output"]], "hp_batchtospacend_p": [[36, 1, 1, "c.hp_batchtospacend_p", "block_size"], [36, 1, 1, "c.hp_batchtospacend_p", "crops"], [36, 1, 1, "c.hp_batchtospacend_p", "data_size"], [36, 1, 1, "c.hp_batchtospacend_p", "input"], [36, 1, 1, "c.hp_batchtospacend_p", "input_shape"], [36, 1, 1, "c.hp_batchtospacend_p", "output"]], "hp_batchtospacend_s": [[36, 1, 1, "c.hp_batchtospacend_s", "block_size"], [36, 1, 1, "c.hp_batchtospacend_s", "core_mask"], [36, 1, 1, "c.hp_batchtospacend_s", "crops"], [36, 1, 1, "c.hp_batchtospacend_s", "data_size"], [36, 1, 1, "c.hp_batchtospacend_s", "input"], [36, 1, 1, "c.hp_batchtospacend_s", "input_shape"], [36, 1, 1, "c.hp_batchtospacend_s", "output"]], "hp_biasadd_p": [[37, 1, 1, "c.hp_biasadd_p", "data_format"], [37, 1, 1, "c.hp_biasadd_p", "dims"], [37, 1, 1, "c.hp_biasadd_p", "input_bias"], [37, 1, 1, "c.hp_biasadd_p", "input_x"], [37, 1, 1, "c.hp_biasadd_p", "length"], [37, 1, 1, "c.hp_biasadd_p", "output"], [37, 1, 1, "c.hp_biasadd_p", "shape_size"]], "hp_biasadd_s": [[37, 1, 1, "c.hp_biasadd_s", "core_mask"], [37, 1, 1, "c.hp_biasadd_s", "data_format"], [37, 1, 1, "c.hp_biasadd_s", "dims"], [37, 1, 1, "c.hp_biasadd_s", "input_bias"], [37, 1, 1, "c.hp_biasadd_s", "input_x"], [37, 1, 1, "c.hp_biasadd_s", "length"], [37, 1, 1, "c.hp_biasadd_s", "output"], [37, 1, 1, "c.hp_biasadd_s", "shape_size"]], "hp_biasaddgrad_p": [[38, 1, 1, "c.hp_biasaddgrad_p", "dbias"], [38, 1, 1, "c.hp_biasaddgrad_p", "dy"], [38, 1, 1, "c.hp_biasaddgrad_p", "dy_dims"], [38, 1, 1, "c.hp_biasaddgrad_p", "shape_size"]], "hp_biasaddgrad_s": [[38, 1, 1, "c.hp_biasaddgrad_s", "core_mask"], [38, 1, 1, "c.hp_biasaddgrad_s", "dbias"], [38, 1, 1, "c.hp_biasaddgrad_s", "dy"], [38, 1, 1, "c.hp_biasaddgrad_s", "dy_dims"], [38, 1, 1, "c.hp_biasaddgrad_s", "shape_size"]], "hp_binarycrossentropy_p": [[39, 1, 1, "c.hp_binarycrossentropy_p", "input_size"], [39, 1, 1, "c.hp_binarycrossentropy_p", "input_x"], [39, 1, 1, "c.hp_binarycrossentropy_p", "input_y"], [39, 1, 1, "c.hp_binarycrossentropy_p", "loss"], [39, 1, 1, "c.hp_binarycrossentropy_p", "reduction"], [39, 1, 1, "c.hp_binarycrossentropy_p", "tmp_loss"], [39, 1, 1, "c.hp_binarycrossentropy_p", "weight"], [39, 1, 1, "c.hp_binarycrossentropy_p", "weight_defined"]], "hp_binarycrossentropy_s": [[39, 1, 1, "c.hp_binarycrossentropy_s", "core_mask"], [39, 1, 1, "c.hp_binarycrossentropy_s", "input_size"], [39, 1, 1, "c.hp_binarycrossentropy_s", "input_x"], [39, 1, 1, "c.hp_binarycrossentropy_s", "input_y"], [39, 1, 1, "c.hp_binarycrossentropy_s", "loss"], [39, 1, 1, "c.hp_binarycrossentropy_s", "reduction"], [39, 1, 1, "c.hp_binarycrossentropy_s", "tmp_loss"], [39, 1, 1, "c.hp_binarycrossentropy_s", "weight"], [39, 1, 1, "c.hp_binarycrossentropy_s", "weight_defined"]], "hp_binarycrossentropygrad_p": [[40, 1, 1, "c.hp_binarycrossentropygrad_p", "dloss"], [40, 1, 1, "c.hp_binarycrossentropygrad_p", "dx"], [40, 1, 1, "c.hp_binarycrossentropygrad_p", "input_size"], [40, 1, 1, "c.hp_binarycrossentropygrad_p", "input_x"], [40, 1, 1, "c.hp_binarycrossentropygrad_p", "input_y"], [40, 1, 1, "c.hp_binarycrossentropygrad_p", "reduction"], [40, 1, 1, "c.hp_binarycrossentropygrad_p", "weight"], [40, 1, 1, "c.hp_binarycrossentropygrad_p", "weight_defined"]], "hp_binarycrossentropygrad_s": [[40, 1, 1, "c.hp_binarycrossentropygrad_s", "core_mask"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "dloss"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "dx"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "input_size"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "input_x"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "input_y"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "reduction"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "weight"], [40, 1, 1, "c.hp_binarycrossentropygrad_s", "weight_defined"]], "hp_broadcastto_p": [[41, 1, 1, "c.hp_broadcastto_p", "data_size"], [41, 1, 1, "c.hp_broadcastto_p", "input"], [41, 1, 1, "c.hp_broadcastto_p", "input_shape"], [41, 1, 1, "c.hp_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.hp_broadcastto_p", "output"], [41, 1, 1, "c.hp_broadcastto_p", "output_shape"], [41, 1, 1, "c.hp_broadcastto_p", "output_shape_size"]], "hp_broadcastto_s": [[41, 1, 1, "c.hp_broadcastto_s", "core_mask"], [41, 1, 1, "c.hp_broadcastto_s", "data_size"], [41, 1, 1, "c.hp_broadcastto_s", "input"], [41, 1, 1, "c.hp_broadcastto_s", "input_shape"], [41, 1, 1, "c.hp_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.hp_broadcastto_s", "output"], [41, 1, 1, "c.hp_broadcastto_s", "output_shape"], [41, 1, 1, "c.hp_broadcastto_s", "output_shape_size"]], "hp_ceil_p": [[43, 1, 1, "c.hp_ceil_p", "input_size"], [43, 1, 1, "c.hp_ceil_p", "input_x"], [43, 1, 1, "c.hp_ceil_p", "output"]], "hp_ceil_s": [[43, 1, 1, "c.hp_ceil_s", "core_mask"], [43, 1, 1, "c.hp_ceil_s", "input_size"], [43, 1, 1, "c.hp_ceil_s", "input_x"], [43, 1, 1, "c.hp_ceil_s", "output"]], "hp_celu_p": [[12, 1, 1, "c.hp_celu_p", "Input0"], [12, 1, 1, "c.hp_celu_p", "alpha"], [12, 1, 1, "c.hp_celu_p", "length"], [12, 1, 1, "c.hp_celu_p", "output"]], "hp_celu_s": [[12, 1, 1, "c.hp_celu_s", "Input0"], [12, 1, 1, "c.hp_celu_s", "alpha"], [12, 1, 1, "c.hp_celu_s", "core_mask"], [12, 1, 1, "c.hp_celu_s", "length"], [12, 1, 1, "c.hp_celu_s", "output"]], "hp_clip_p": [[12, 1, 1, "c.hp_clip_p", "Input0"], [12, 1, 1, "c.hp_clip_p", "length"], [12, 1, 1, "c.hp_clip_p", "max_val"], [12, 1, 1, "c.hp_clip_p", "min_val"], [12, 1, 1, "c.hp_clip_p", "output"]], "hp_clip_s": [[12, 1, 1, "c.hp_clip_s", "Input0"], [12, 1, 1, "c.hp_clip_s", "core_mask"], [12, 1, 1, "c.hp_clip_s", "length"], [12, 1, 1, "c.hp_clip_s", "max_val"], [12, 1, 1, "c.hp_clip_s", "min_val"], [12, 1, 1, "c.hp_clip_s", "output"]], "hp_concat_p": [[45, 1, 1, "c.hp_concat_p", "axis"], [45, 1, 1, "c.hp_concat_p", "input_ndim"], [45, 1, 1, "c.hp_concat_p", "input_shapes"], [45, 1, 1, "c.hp_concat_p", "inputs"], [45, 1, 1, "c.hp_concat_p", "num_inputs"], [45, 1, 1, "c.hp_concat_p", "output"]], "hp_concat_s": [[45, 1, 1, "c.hp_concat_s", "axis"], [45, 1, 1, "c.hp_concat_s", "core_mask"], [45, 1, 1, "c.hp_concat_s", "input_ndim"], [45, 1, 1, "c.hp_concat_s", "input_shapes"], [45, 1, 1, "c.hp_concat_s", "inputs"], [45, 1, 1, "c.hp_concat_s", "num_inputs"], [45, 1, 1, "c.hp_concat_s", "output"]], "hp_conv2d_p": [[47, 1, 1, "c.hp_conv2d_p", "bias"], [47, 1, 1, "c.hp_conv2d_p", "conv_param"], [47, 1, 1, "c.hp_conv2d_p", "core_mask"], [47, 1, 1, "c.hp_conv2d_p", "input_w"], [47, 1, 1, "c.hp_conv2d_p", "input_x"], [47, 1, 1, "c.hp_conv2d_p", "out_y"]], "hp_conv2d_s": [[47, 1, 1, "c.hp_conv2d_s", "bias"], [47, 1, 1, "c.hp_conv2d_s", "conv_param"], [47, 1, 1, "c.hp_conv2d_s", "core_mask"], [47, 1, 1, "c.hp_conv2d_s", "input_w"], [47, 1, 1, "c.hp_conv2d_s", "input_x"], [47, 1, 1, "c.hp_conv2d_s", "out_y"]], "hp_conv2dbackpropfilterfusion_p": [[49, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "conv_param"], [49, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "dw"], [49, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "dy"], [49, 1, 1, "c.hp_conv2dbackpropfilterfusion_p", "x"]], "hp_conv2dbackpropfilterfusion_s": [[49, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "conv_param"], [49, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "core_mask"], [49, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "dw"], [49, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "dy"], [49, 1, 1, "c.hp_conv2dbackpropfilterfusion_s", "x"]], "hp_conv2dbackpropinputfusion_p": [[50, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "conv_param"], [50, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "dx"], [50, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "dy"], [50, 1, 1, "c.hp_conv2dbackpropinputfusion_p", "w"]], "hp_conv2dbackpropinputfusion_s": [[50, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "conv_param"], [50, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "core_mask"], [50, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "dx"], [50, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "dy"], [50, 1, 1, "c.hp_conv2dbackpropinputfusion_s", "w"]], "hp_convtranspose_p": [[48, 1, 1, "c.hp_convtranspose_p", "bias"], [48, 1, 1, "c.hp_convtranspose_p", "conv_param"], [48, 1, 1, "c.hp_convtranspose_p", "core_mask"], [48, 1, 1, "c.hp_convtranspose_p", "input_w"], [48, 1, 1, "c.hp_convtranspose_p", "input_x"], [48, 1, 1, "c.hp_convtranspose_p", "out_y"]], "hp_convtranspose_s": [[48, 1, 1, "c.hp_convtranspose_s", "bias"], [48, 1, 1, "c.hp_convtranspose_s", "conv_param"], [48, 1, 1, "c.hp_convtranspose_s", "core_mask"], [48, 1, 1, "c.hp_convtranspose_s", "input_w"], [48, 1, 1, "c.hp_convtranspose_s", "input_x"], [48, 1, 1, "c.hp_convtranspose_s", "out_y"]], "hp_cos_p": [[51, 1, 1, "c.hp_cos_p", "dst_data"], [51, 1, 1, "c.hp_cos_p", "length"], [51, 1, 1, "c.hp_cos_p", "src_data"]], "hp_cos_s": [[51, 1, 1, "c.hp_cos_s", "core_mask"], [51, 1, 1, "c.hp_cos_s", "dst_data"], [51, 1, 1, "c.hp_cos_s", "length"], [51, 1, 1, "c.hp_cos_s", "src_data"]], "hp_crop_and_resize_anycore": [[53, 1, 1, "c.hp_crop_and_resize_anycore", "box_idx"], [53, 1, 1, "c.hp_crop_and_resize_anycore", "boxes"], [53, 1, 1, "c.hp_crop_and_resize_anycore", "core_mask"], [53, 1, 1, "c.hp_crop_and_resize_anycore", "dst"], [53, 1, 1, "c.hp_crop_and_resize_anycore", "extrapolation_value"], [53, 1, 1, "c.hp_crop_and_resize_anycore", "param"], [53, 1, 1, "c.hp_crop_and_resize_anycore", "src"]], "hp_cumsum_p": [[54, 1, 1, "c.hp_cumsum_p", "axis_dim"], [54, 1, 1, "c.hp_cumsum_p", "exclusive"], [54, 1, 1, "c.hp_cumsum_p", "inner_dim"], [54, 1, 1, "c.hp_cumsum_p", "input"], [54, 1, 1, "c.hp_cumsum_p", "out_dim"], [54, 1, 1, "c.hp_cumsum_p", "output"]], "hp_cumsum_s": [[54, 1, 1, "c.hp_cumsum_s", "axis_dim"], [54, 1, 1, "c.hp_cumsum_s", "core_mask"], [54, 1, 1, "c.hp_cumsum_s", "exclusive"], [54, 1, 1, "c.hp_cumsum_s", "inner_dim"], [54, 1, 1, "c.hp_cumsum_s", "input"], [54, 1, 1, "c.hp_cumsum_s", "out_dim"], [54, 1, 1, "c.hp_cumsum_s", "output"]], "hp_deconvgradfilter_p": [[58, 1, 1, "c.hp_deconvgradfilter_p", "dw_data"], [58, 1, 1, "c.hp_deconvgradfilter_p", "dy_data"], [58, 1, 1, "c.hp_deconvgradfilter_p", "param"], [58, 1, 1, "c.hp_deconvgradfilter_p", "x_data"]], "hp_deconvgradfilter_s": [[58, 1, 1, "c.hp_deconvgradfilter_s", "core_mask"], [58, 1, 1, "c.hp_deconvgradfilter_s", "dw_data"], [58, 1, 1, "c.hp_deconvgradfilter_s", "dy_data"], [58, 1, 1, "c.hp_deconvgradfilter_s", "param"], [58, 1, 1, "c.hp_deconvgradfilter_s", "x_data"]], "hp_depthtospace_p": [[59, 1, 1, "c.hp_depthtospace_p", "block_size"], [59, 1, 1, "c.hp_depthtospace_p", "data_size"], [59, 1, 1, "c.hp_depthtospace_p", "in_shape"], [59, 1, 1, "c.hp_depthtospace_p", "input"], [59, 1, 1, "c.hp_depthtospace_p", "output"]], "hp_depthtospace_s": [[59, 1, 1, "c.hp_depthtospace_s", "block_size"], [59, 1, 1, "c.hp_depthtospace_s", "core_mask"], [59, 1, 1, "c.hp_depthtospace_s", "data_size"], [59, 1, 1, "c.hp_depthtospace_s", "in_shape"], [59, 1, 1, "c.hp_depthtospace_s", "input"], [59, 1, 1, "c.hp_depthtospace_s", "output"]], "hp_detection_post_process_p": [[60, 1, 1, "c.hp_detection_post_process_p", "anchors"], [60, 1, 1, "c.hp_detection_post_process_p", "input_boxes"], [60, 1, 1, "c.hp_detection_post_process_p", "input_scores"], [60, 1, 1, "c.hp_detection_post_process_p", "output_boxes"], [60, 1, 1, "c.hp_detection_post_process_p", "output_classes"], [60, 1, 1, "c.hp_detection_post_process_p", "output_num"], [60, 1, 1, "c.hp_detection_post_process_p", "output_scores"], [60, 1, 1, "c.hp_detection_post_process_p", "param"]], "hp_detection_post_process_s": [[60, 1, 1, "c.hp_detection_post_process_s", "anchors"], [60, 1, 1, "c.hp_detection_post_process_s", "core_mask"], [60, 1, 1, "c.hp_detection_post_process_s", "input_boxes"], [60, 1, 1, "c.hp_detection_post_process_s", "input_scores"], [60, 1, 1, "c.hp_detection_post_process_s", "output_boxes"], [60, 1, 1, "c.hp_detection_post_process_s", "output_classes"], [60, 1, 1, "c.hp_detection_post_process_s", "output_num"], [60, 1, 1, "c.hp_detection_post_process_s", "output_scores"], [60, 1, 1, "c.hp_detection_post_process_s", "param"]], "hp_div_fusion_p": [[61, 1, 1, "c.hp_div_fusion_p", "input0"], [61, 1, 1, "c.hp_div_fusion_p", "input1"], [61, 1, 1, "c.hp_div_fusion_p", "length"], [61, 1, 1, "c.hp_div_fusion_p", "output"]], "hp_div_fusion_s": [[61, 1, 1, "c.hp_div_fusion_s", "core_mask"], [61, 1, 1, "c.hp_div_fusion_s", "input0"], [61, 1, 1, "c.hp_div_fusion_s", "input1"], [61, 1, 1, "c.hp_div_fusion_s", "length"], [61, 1, 1, "c.hp_div_fusion_s", "output"]], "hp_dropout_p": [[63, 1, 1, "c.hp_dropout_p", "input"], [63, 1, 1, "c.hp_dropout_p", "length"], [63, 1, 1, "c.hp_dropout_p", "mask"], [63, 1, 1, "c.hp_dropout_p", "output"], [63, 1, 1, "c.hp_dropout_p", "scale"]], "hp_dropout_s": [[63, 1, 1, "c.hp_dropout_s", "core_mask"], [63, 1, 1, "c.hp_dropout_s", "input"], [63, 1, 1, "c.hp_dropout_s", "length"], [63, 1, 1, "c.hp_dropout_s", "mask"], [63, 1, 1, "c.hp_dropout_s", "output"], [63, 1, 1, "c.hp_dropout_s", "scale"]], "hp_dropoutgrad_p": [[64, 1, 1, "c.hp_dropoutgrad_p", "input"], [64, 1, 1, "c.hp_dropoutgrad_p", "length"], [64, 1, 1, "c.hp_dropoutgrad_p", "mask"], [64, 1, 1, "c.hp_dropoutgrad_p", "output"], [64, 1, 1, "c.hp_dropoutgrad_p", "scale"]], "hp_dropoutgrad_s": [[64, 1, 1, "c.hp_dropoutgrad_s", "core_mask"], [64, 1, 1, "c.hp_dropoutgrad_s", "input"], [64, 1, 1, "c.hp_dropoutgrad_s", "length"], [64, 1, 1, "c.hp_dropoutgrad_s", "mask"], [64, 1, 1, "c.hp_dropoutgrad_s", "output"], [64, 1, 1, "c.hp_dropoutgrad_s", "scale"]], "hp_eltwise_p": [[67, 1, 1, "c.hp_eltwise_p", "Input0"], [67, 1, 1, "c.hp_eltwise_p", "Input1"], [67, 1, 1, "c.hp_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.hp_eltwise_p", "length"], [67, 1, 1, "c.hp_eltwise_p", "output"]], "hp_eltwise_s": [[67, 1, 1, "c.hp_eltwise_s", "Input0"], [67, 1, 1, "c.hp_eltwise_s", "Input1"], [67, 1, 1, "c.hp_eltwise_s", "core_mask"], [67, 1, 1, "c.hp_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.hp_eltwise_s", "length"], [67, 1, 1, "c.hp_eltwise_s", "output"]], "hp_elu_p": [[12, 1, 1, "c.hp_elu_p", "Input0"], [12, 1, 1, "c.hp_elu_p", "alpha"], [12, 1, 1, "c.hp_elu_p", "length"], [12, 1, 1, "c.hp_elu_p", "output"]], "hp_elu_s": [[12, 1, 1, "c.hp_elu_s", "Input0"], [12, 1, 1, "c.hp_elu_s", "alpha"], [12, 1, 1, "c.hp_elu_s", "core_mask"], [12, 1, 1, "c.hp_elu_s", "length"], [12, 1, 1, "c.hp_elu_s", "output"]], "hp_embeddinglookup_p": [[69, 1, 1, "c.hp_embeddinglookup_p", "Input0"], [69, 1, 1, "c.hp_embeddinglookup_p", "Input1"], [69, 1, 1, "c.hp_embeddinglookup_p", "length"], [69, 1, 1, "c.hp_embeddinglookup_p", "output"]], "hp_embeddinglookup_s": [[69, 1, 1, "c.hp_embeddinglookup_s", "core_mask"], [69, 1, 1, "c.hp_embeddinglookup_s", "ids"], [69, 1, 1, "c.hp_embeddinglookup_s", "ids_size_"], [69, 1, 1, "c.hp_embeddinglookup_s", "input_data"], [69, 1, 1, "c.hp_embeddinglookup_s", "is_regulated"], [69, 1, 1, "c.hp_embeddinglookup_s", "layer_num_"], [69, 1, 1, "c.hp_embeddinglookup_s", "layer_size_"], [69, 1, 1, "c.hp_embeddinglookup_s", "max_norm_"], [69, 1, 1, "c.hp_embeddinglookup_s", "output"]], "hp_equal_p": [[70, 1, 1, "c.hp_equal_p", "Input0"], [70, 1, 1, "c.hp_equal_p", "Input1"], [70, 1, 1, "c.hp_equal_p", "length"], [70, 1, 1, "c.hp_equal_p", "output"]], "hp_equal_s": [[70, 1, 1, "c.hp_equal_s", "Input0"], [70, 1, 1, "c.hp_equal_s", "Input1"], [70, 1, 1, "c.hp_equal_s", "core_mask"], [70, 1, 1, "c.hp_equal_s", "length"], [70, 1, 1, "c.hp_equal_s", "output"]], "hp_erf_p": [[71, 1, 1, "c.hp_erf_p", "input"], [71, 1, 1, "c.hp_erf_p", "length"], [71, 1, 1, "c.hp_erf_p", "output"]], "hp_erf_s": [[71, 1, 1, "c.hp_erf_s", "core_mask"], [71, 1, 1, "c.hp_erf_s", "input"], [71, 1, 1, "c.hp_erf_s", "length"], [71, 1, 1, "c.hp_erf_s", "output"]], "hp_expfusion_p": [[73, 1, 1, "c.hp_expfusion_p", "dst_data"], [73, 1, 1, "c.hp_expfusion_p", "in_scale"], [73, 1, 1, "c.hp_expfusion_p", "length"], [73, 1, 1, "c.hp_expfusion_p", "out_scale"], [73, 1, 1, "c.hp_expfusion_p", "scale"], [73, 1, 1, "c.hp_expfusion_p", "src_data"]], "hp_expfusion_s": [[73, 1, 1, "c.hp_expfusion_s", "core_mask"], [73, 1, 1, "c.hp_expfusion_s", "dst_data"], [73, 1, 1, "c.hp_expfusion_s", "in_scale"], [73, 1, 1, "c.hp_expfusion_s", "length"], [73, 1, 1, "c.hp_expfusion_s", "out_scale"], [73, 1, 1, "c.hp_expfusion_s", "scale"], [73, 1, 1, "c.hp_expfusion_s", "src_data"]], "hp_fake_quant_with_min_max_vars_p": [[74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "length"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "max_val"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "min_val"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "output"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "quant_max"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "quant_min"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "src"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_p", "symmetric"]], "hp_fake_quant_with_min_max_vars_per_channel_p": [[75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "channel_num"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "length"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "max_val"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "min_val"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "output"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "quant_max"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "quant_min"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "src"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_p", "symmetric"]], "hp_fake_quant_with_min_max_vars_per_channel_s": [[75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "channel_num"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "core_mask"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "length"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "max_val"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "min_val"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "output"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "quant_max"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "quant_min"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "src"], [75, 1, 1, "c.hp_fake_quant_with_min_max_vars_per_channel_s", "symmetric"]], "hp_fake_quant_with_min_max_vars_s": [[74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "core_mask"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "length"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "max_val"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "min_val"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "output"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "quant_max"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "quant_min"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "src"], [74, 1, 1, "c.hp_fake_quant_with_min_max_vars_s", "symmetric"]], "hp_fill_p": [[78, 1, 1, "c.hp_fill_p", "output"], [78, 1, 1, "c.hp_fill_p", "param"], [78, 1, 1, "c.hp_fill_p", "value"]], "hp_fill_s": [[78, 1, 1, "c.hp_fill_s", "core_mask"], [78, 1, 1, "c.hp_fill_s", "output"], [78, 1, 1, "c.hp_fill_s", "param"], [78, 1, 1, "c.hp_fill_s", "value"]], "hp_flattengrad_p": [[81, 1, 1, "c.hp_flattengrad_p", "input"], [81, 1, 1, "c.hp_flattengrad_p", "origin_ndim"], [81, 1, 1, "c.hp_flattengrad_p", "origin_shape"], [81, 1, 1, "c.hp_flattengrad_p", "output"]], "hp_flattengrad_s": [[81, 1, 1, "c.hp_flattengrad_s", "core_mask"], [81, 1, 1, "c.hp_flattengrad_s", "input"], [81, 1, 1, "c.hp_flattengrad_s", "origin_ndim"], [81, 1, 1, "c.hp_flattengrad_s", "origin_shape"], [81, 1, 1, "c.hp_flattengrad_s", "output"]], "hp_floor_p": [[82, 1, 1, "c.hp_floor_p", "dst_data"], [82, 1, 1, "c.hp_floor_p", "length"], [82, 1, 1, "c.hp_floor_p", "src_data"]], "hp_floor_s": [[82, 1, 1, "c.hp_floor_s", "core_mask"], [82, 1, 1, "c.hp_floor_s", "dst_data"], [82, 1, 1, "c.hp_floor_s", "length"], [82, 1, 1, "c.hp_floor_s", "src_data"]], "hp_floordiv_p": [[83, 1, 1, "c.hp_floordiv_p", "dst_data"], [83, 1, 1, "c.hp_floordiv_p", "length"], [83, 1, 1, "c.hp_floordiv_p", "src_data0"], [83, 1, 1, "c.hp_floordiv_p", "src_data1"]], "hp_floordiv_s": [[83, 1, 1, "c.hp_floordiv_s", "core_mask"], [83, 1, 1, "c.hp_floordiv_s", "dst_data"], [83, 1, 1, "c.hp_floordiv_s", "length"], [83, 1, 1, "c.hp_floordiv_s", "src_data0"], [83, 1, 1, "c.hp_floordiv_s", "src_data1"]], "hp_floormod_p": [[84, 1, 1, "c.hp_floormod_p", "input0"], [84, 1, 1, "c.hp_floormod_p", "input1"], [84, 1, 1, "c.hp_floormod_p", "output"], [84, 1, 1, "c.hp_floormod_p", "size"]], "hp_floormod_s": [[84, 1, 1, "c.hp_floormod_s", "core_mask"], [84, 1, 1, "c.hp_floormod_s", "input0"], [84, 1, 1, "c.hp_floormod_s", "input1"], [84, 1, 1, "c.hp_floormod_s", "output"], [84, 1, 1, "c.hp_floormod_s", "size"]], "hp_formattranspose_p": [[85, 1, 1, "c.hp_formattranspose_p", "batch"], [85, 1, 1, "c.hp_formattranspose_p", "channel"], [85, 1, 1, "c.hp_formattranspose_p", "dst_data"], [85, 1, 1, "c.hp_formattranspose_p", "dst_format"], [85, 1, 1, "c.hp_formattranspose_p", "plane"], [85, 1, 1, "c.hp_formattranspose_p", "src_data"], [85, 1, 1, "c.hp_formattranspose_p", "src_format"]], "hp_formattranspose_s": [[85, 1, 1, "c.hp_formattranspose_s", "batch"], [85, 1, 1, "c.hp_formattranspose_s", "channel"], [85, 1, 1, "c.hp_formattranspose_s", "core_mask"], [85, 1, 1, "c.hp_formattranspose_s", "dst_data"], [85, 1, 1, "c.hp_formattranspose_s", "dst_format"], [85, 1, 1, "c.hp_formattranspose_s", "plane"], [85, 1, 1, "c.hp_formattranspose_s", "src_data"], [85, 1, 1, "c.hp_formattranspose_s", "src_format"]], "hp_fullconnection_p": [[86, 1, 1, "c.hp_fullconnection_p", "A"], [86, 1, 1, "c.hp_fullconnection_p", "B"], [86, 1, 1, "c.hp_fullconnection_p", "C"], [86, 1, 1, "c.hp_fullconnection_p", "K"], [86, 1, 1, "c.hp_fullconnection_p", "M"], [86, 1, 1, "c.hp_fullconnection_p", "N"], [86, 1, 1, "c.hp_fullconnection_p", "activation_type"], [86, 1, 1, "c.hp_fullconnection_p", "bias"]], "hp_fullconnection_s": [[86, 1, 1, "c.hp_fullconnection_s", "A"], [86, 1, 1, "c.hp_fullconnection_s", "B"], [86, 1, 1, "c.hp_fullconnection_s", "C"], [86, 1, 1, "c.hp_fullconnection_s", "K"], [86, 1, 1, "c.hp_fullconnection_s", "M"], [86, 1, 1, "c.hp_fullconnection_s", "N"], [86, 1, 1, "c.hp_fullconnection_s", "activation_type"], [86, 1, 1, "c.hp_fullconnection_s", "bias"], [86, 1, 1, "c.hp_fullconnection_s", "core_mask"]], "hp_fusedbatchnorm_p": [[87, 1, 1, "c.hp_fusedbatchnorm_p", "channel"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "epsilon"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "input"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "mean"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "offset"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "output"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "scale"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "unit"], [87, 1, 1, "c.hp_fusedbatchnorm_p", "variance"]], "hp_fusedbatchnorm_s": [[87, 1, 1, "c.hp_fusedbatchnorm_s", "channel"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "core_mask"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "epsilon"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "input"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "mean"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "offset"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "output"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "scale"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "unit"], [87, 1, 1, "c.hp_fusedbatchnorm_s", "variance"]], "hp_gather_nd_p": [[89, 1, 1, "c.hp_gather_nd_p", "indices"], [89, 1, 1, "c.hp_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.hp_gather_nd_p", "indices_shape"], [89, 1, 1, "c.hp_gather_nd_p", "input"], [89, 1, 1, "c.hp_gather_nd_p", "input_ndim"], [89, 1, 1, "c.hp_gather_nd_p", "input_shape"], [89, 1, 1, "c.hp_gather_nd_p", "output"]], "hp_gather_nd_s": [[89, 1, 1, "c.hp_gather_nd_s", "core_mask"], [89, 1, 1, "c.hp_gather_nd_s", "indices"], [89, 1, 1, "c.hp_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.hp_gather_nd_s", "indices_shape"], [89, 1, 1, "c.hp_gather_nd_s", "input"], [89, 1, 1, "c.hp_gather_nd_s", "input_ndim"], [89, 1, 1, "c.hp_gather_nd_s", "input_shape"], [89, 1, 1, "c.hp_gather_nd_s", "output"]], "hp_gather_p": [[88, 1, 1, "c.hp_gather_p", "axis"], [88, 1, 1, "c.hp_gather_p", "batch_dims"], [88, 1, 1, "c.hp_gather_p", "indices"], [88, 1, 1, "c.hp_gather_p", "indices_ndim"], [88, 1, 1, "c.hp_gather_p", "indices_shape"], [88, 1, 1, "c.hp_gather_p", "input"], [88, 1, 1, "c.hp_gather_p", "input_ndim"], [88, 1, 1, "c.hp_gather_p", "input_shape"], [88, 1, 1, "c.hp_gather_p", "output"]], "hp_gather_s": [[88, 1, 1, "c.hp_gather_s", "axis"], [88, 1, 1, "c.hp_gather_s", "batch_dims"], [88, 1, 1, "c.hp_gather_s", "core_mask"], [88, 1, 1, "c.hp_gather_s", "indices"], [88, 1, 1, "c.hp_gather_s", "indices_ndim"], [88, 1, 1, "c.hp_gather_s", "indices_shape"], [88, 1, 1, "c.hp_gather_s", "input"], [88, 1, 1, "c.hp_gather_s", "input_ndim"], [88, 1, 1, "c.hp_gather_s", "input_shape"], [88, 1, 1, "c.hp_gather_s", "output"]], "hp_gelu_p": [[12, 1, 1, "c.hp_gelu_p", "Input0"], [12, 1, 1, "c.hp_gelu_p", "approximate"], [12, 1, 1, "c.hp_gelu_p", "length"], [12, 1, 1, "c.hp_gelu_p", "output"]], "hp_gelu_s": [[12, 1, 1, "c.hp_gelu_s", "Input0"], [12, 1, 1, "c.hp_gelu_s", "approximate"], [12, 1, 1, "c.hp_gelu_s", "core_mask"], [12, 1, 1, "c.hp_gelu_s", "length"], [12, 1, 1, "c.hp_gelu_s", "output"]], "hp_glu_p": [[91, 1, 1, "c.hp_glu_p", "in_data"], [91, 1, 1, "c.hp_glu_p", "input_shape"], [91, 1, 1, "c.hp_glu_p", "len"], [91, 1, 1, "c.hp_glu_p", "ndim"], [91, 1, 1, "c.hp_glu_p", "num_split"], [91, 1, 1, "c.hp_glu_p", "out_data"], [91, 1, 1, "c.hp_glu_p", "split_data"], [91, 1, 1, "c.hp_glu_p", "split_dim"], [91, 1, 1, "c.hp_glu_p", "split_sizes"], [91, 1, 1, "c.hp_glu_p", "strides"]], "hp_glu_s": [[91, 1, 1, "c.hp_glu_s", "core_mask"], [91, 1, 1, "c.hp_glu_s", "in_data"], [91, 1, 1, "c.hp_glu_s", "input_shape"], [91, 1, 1, "c.hp_glu_s", "len"], [91, 1, 1, "c.hp_glu_s", "ndim"], [91, 1, 1, "c.hp_glu_s", "num_split"], [91, 1, 1, "c.hp_glu_s", "out_data"], [91, 1, 1, "c.hp_glu_s", "split_data"], [91, 1, 1, "c.hp_glu_s", "split_dim"], [91, 1, 1, "c.hp_glu_s", "split_sizes"], [91, 1, 1, "c.hp_glu_s", "strides"]], "hp_graddiv1l_p": [[62, 1, 1, "c.hp_graddiv1l_p", "dx1"], [62, 1, 1, "c.hp_graddiv1l_p", "dx2"], [62, 1, 1, "c.hp_graddiv1l_p", "dy"], [62, 1, 1, "c.hp_graddiv1l_p", "indices"], [62, 1, 1, "c.hp_graddiv1l_p", "large_multiples"], [62, 1, 1, "c.hp_graddiv1l_p", "large_shape"], [62, 1, 1, "c.hp_graddiv1l_p", "large_strides"], [62, 1, 1, "c.hp_graddiv1l_p", "ndims"], [62, 1, 1, "c.hp_graddiv1l_p", "out_shape"], [62, 1, 1, "c.hp_graddiv1l_p", "out_strides"], [62, 1, 1, "c.hp_graddiv1l_p", "small_multiples"], [62, 1, 1, "c.hp_graddiv1l_p", "small_shape"], [62, 1, 1, "c.hp_graddiv1l_p", "small_strides"], [62, 1, 1, "c.hp_graddiv1l_p", "tile_data0"], [62, 1, 1, "c.hp_graddiv1l_p", "tile_data1"], [62, 1, 1, "c.hp_graddiv1l_p", "tile_data2"], [62, 1, 1, "c.hp_graddiv1l_p", "x1"], [62, 1, 1, "c.hp_graddiv1l_p", "x2"]], "hp_graddiv1l_s": [[62, 1, 1, "c.hp_graddiv1l_s", "core_mask"], [62, 1, 1, "c.hp_graddiv1l_s", "dx1"], [62, 1, 1, "c.hp_graddiv1l_s", "dx2"], [62, 1, 1, "c.hp_graddiv1l_s", "dy"], [62, 1, 1, "c.hp_graddiv1l_s", "indices"], [62, 1, 1, "c.hp_graddiv1l_s", "large_multiples"], [62, 1, 1, "c.hp_graddiv1l_s", "large_shape"], [62, 1, 1, "c.hp_graddiv1l_s", "large_strides"], [62, 1, 1, "c.hp_graddiv1l_s", "ndims"], [62, 1, 1, "c.hp_graddiv1l_s", "out_shape"], [62, 1, 1, "c.hp_graddiv1l_s", "out_strides"], [62, 1, 1, "c.hp_graddiv1l_s", "small_multiples"], [62, 1, 1, "c.hp_graddiv1l_s", "small_shape"], [62, 1, 1, "c.hp_graddiv1l_s", "small_strides"], [62, 1, 1, "c.hp_graddiv1l_s", "tile_data0"], [62, 1, 1, "c.hp_graddiv1l_s", "tile_data1"], [62, 1, 1, "c.hp_graddiv1l_s", "tile_data2"], [62, 1, 1, "c.hp_graddiv1l_s", "x1"], [62, 1, 1, "c.hp_graddiv1l_s", "x2"]], "hp_graddiv2l_p": [[62, 1, 1, "c.hp_graddiv2l_p", "dx1"], [62, 1, 1, "c.hp_graddiv2l_p", "dx2"], [62, 1, 1, "c.hp_graddiv2l_p", "dy"], [62, 1, 1, "c.hp_graddiv2l_p", "indices"], [62, 1, 1, "c.hp_graddiv2l_p", "large_multiples"], [62, 1, 1, "c.hp_graddiv2l_p", "large_shape"], [62, 1, 1, "c.hp_graddiv2l_p", "large_strides"], [62, 1, 1, "c.hp_graddiv2l_p", "ndims"], [62, 1, 1, "c.hp_graddiv2l_p", "out_shape"], [62, 1, 1, "c.hp_graddiv2l_p", "out_strides"], [62, 1, 1, "c.hp_graddiv2l_p", "small_multiples"], [62, 1, 1, "c.hp_graddiv2l_p", "small_shape"], [62, 1, 1, "c.hp_graddiv2l_p", "small_strides"], [62, 1, 1, "c.hp_graddiv2l_p", "tile_data0"], [62, 1, 1, "c.hp_graddiv2l_p", "tile_data1"], [62, 1, 1, "c.hp_graddiv2l_p", "tile_data2"], [62, 1, 1, "c.hp_graddiv2l_p", "x1"], [62, 1, 1, "c.hp_graddiv2l_p", "x2"]], "hp_graddiv2l_s": [[62, 1, 1, "c.hp_graddiv2l_s", "core_mask"], [62, 1, 1, "c.hp_graddiv2l_s", "dx1"], [62, 1, 1, "c.hp_graddiv2l_s", "dx2"], [62, 1, 1, "c.hp_graddiv2l_s", "dy"], [62, 1, 1, "c.hp_graddiv2l_s", "indices"], [62, 1, 1, "c.hp_graddiv2l_s", "large_multiples"], [62, 1, 1, "c.hp_graddiv2l_s", "large_shape"], [62, 1, 1, "c.hp_graddiv2l_s", "large_strides"], [62, 1, 1, "c.hp_graddiv2l_s", "ndims"], [62, 1, 1, "c.hp_graddiv2l_s", "out_shape"], [62, 1, 1, "c.hp_graddiv2l_s", "out_strides"], [62, 1, 1, "c.hp_graddiv2l_s", "small_multiples"], [62, 1, 1, "c.hp_graddiv2l_s", "small_shape"], [62, 1, 1, "c.hp_graddiv2l_s", "small_strides"], [62, 1, 1, "c.hp_graddiv2l_s", "tile_data0"], [62, 1, 1, "c.hp_graddiv2l_s", "tile_data1"], [62, 1, 1, "c.hp_graddiv2l_s", "tile_data2"], [62, 1, 1, "c.hp_graddiv2l_s", "x1"], [62, 1, 1, "c.hp_graddiv2l_s", "x2"]], "hp_graddiv_p": [[62, 1, 1, "c.hp_graddiv_p", "dx1"], [62, 1, 1, "c.hp_graddiv_p", "dx2"], [62, 1, 1, "c.hp_graddiv_p", "dy"], [62, 1, 1, "c.hp_graddiv_p", "large_multiples"], [62, 1, 1, "c.hp_graddiv_p", "large_shape"], [62, 1, 1, "c.hp_graddiv_p", "large_strides"], [62, 1, 1, "c.hp_graddiv_p", "ndims"], [62, 1, 1, "c.hp_graddiv_p", "out_shape"], [62, 1, 1, "c.hp_graddiv_p", "out_strides"], [62, 1, 1, "c.hp_graddiv_p", "small_multiples"], [62, 1, 1, "c.hp_graddiv_p", "small_shape"], [62, 1, 1, "c.hp_graddiv_p", "small_strides"], [62, 1, 1, "c.hp_graddiv_p", "tile_data0"], [62, 1, 1, "c.hp_graddiv_p", "tile_data1"], [62, 1, 1, "c.hp_graddiv_p", "tile_data2"], [62, 1, 1, "c.hp_graddiv_p", "x1"], [62, 1, 1, "c.hp_graddiv_p", "x2"]], "hp_graddiv_s": [[62, 1, 1, "c.hp_graddiv_s", "core_mask"], [62, 1, 1, "c.hp_graddiv_s", "dx1"], [62, 1, 1, "c.hp_graddiv_s", "dx2"], [62, 1, 1, "c.hp_graddiv_s", "dy"], [62, 1, 1, "c.hp_graddiv_s", "indices"], [62, 1, 1, "c.hp_graddiv_s", "large_multiples"], [62, 1, 1, "c.hp_graddiv_s", "large_shape"], [62, 1, 1, "c.hp_graddiv_s", "large_strides"], [62, 1, 1, "c.hp_graddiv_s", "ndims"], [62, 1, 1, "c.hp_graddiv_s", "out_shape"], [62, 1, 1, "c.hp_graddiv_s", "out_strides"], [62, 1, 1, "c.hp_graddiv_s", "small_multiples"], [62, 1, 1, "c.hp_graddiv_s", "small_shape"], [62, 1, 1, "c.hp_graddiv_s", "small_strides"], [62, 1, 1, "c.hp_graddiv_s", "tile_data0"], [62, 1, 1, "c.hp_graddiv_s", "tile_data1"], [62, 1, 1, "c.hp_graddiv_s", "tile_data2"], [62, 1, 1, "c.hp_graddiv_s", "x1"], [62, 1, 1, "c.hp_graddiv_s", "x2"]], "hp_gradmul1l_p": [[131, 1, 1, "c.hp_gradmul1l_p", "dx1"], [131, 1, 1, "c.hp_gradmul1l_p", "dx2"], [131, 1, 1, "c.hp_gradmul1l_p", "dy"], [131, 1, 1, "c.hp_gradmul1l_p", "indices"], [131, 1, 1, "c.hp_gradmul1l_p", "large_multiples"], [131, 1, 1, "c.hp_gradmul1l_p", "large_shape"], [131, 1, 1, "c.hp_gradmul1l_p", "large_strides"], [131, 1, 1, "c.hp_gradmul1l_p", "ndims"], [131, 1, 1, "c.hp_gradmul1l_p", "out_shape"], [131, 1, 1, "c.hp_gradmul1l_p", "out_strides"], [131, 1, 1, "c.hp_gradmul1l_p", "small_multiples"], [131, 1, 1, "c.hp_gradmul1l_p", "small_shape"], [131, 1, 1, "c.hp_gradmul1l_p", "small_strides"], [131, 1, 1, "c.hp_gradmul1l_p", "tile_data0"], [131, 1, 1, "c.hp_gradmul1l_p", "tile_data1"], [131, 1, 1, "c.hp_gradmul1l_p", "x1"], [131, 1, 1, "c.hp_gradmul1l_p", "x2"]], "hp_gradmul1l_s": [[131, 1, 1, "c.hp_gradmul1l_s", "core_mask"], [131, 1, 1, "c.hp_gradmul1l_s", "dx1"], [131, 1, 1, "c.hp_gradmul1l_s", "dx2"], [131, 1, 1, "c.hp_gradmul1l_s", "dy"], [131, 1, 1, "c.hp_gradmul1l_s", "indices"], [131, 1, 1, "c.hp_gradmul1l_s", "large_multiples"], [131, 1, 1, "c.hp_gradmul1l_s", "large_shape"], [131, 1, 1, "c.hp_gradmul1l_s", "large_strides"], [131, 1, 1, "c.hp_gradmul1l_s", "ndims"], [131, 1, 1, "c.hp_gradmul1l_s", "out_shape"], [131, 1, 1, "c.hp_gradmul1l_s", "out_strides"], [131, 1, 1, "c.hp_gradmul1l_s", "small_multiples"], [131, 1, 1, "c.hp_gradmul1l_s", "small_shape"], [131, 1, 1, "c.hp_gradmul1l_s", "small_strides"], [131, 1, 1, "c.hp_gradmul1l_s", "tile_data0"], [131, 1, 1, "c.hp_gradmul1l_s", "tile_data1"], [131, 1, 1, "c.hp_gradmul1l_s", "x1"], [131, 1, 1, "c.hp_gradmul1l_s", "x2"]], "hp_gradmul2l_p": [[131, 1, 1, "c.hp_gradmul2l_p", "dx1"], [131, 1, 1, "c.hp_gradmul2l_p", "dx2"], [131, 1, 1, "c.hp_gradmul2l_p", "dy"], [131, 1, 1, "c.hp_gradmul2l_p", "indices"], [131, 1, 1, "c.hp_gradmul2l_p", "large_multiples"], [131, 1, 1, "c.hp_gradmul2l_p", "large_shape"], [131, 1, 1, "c.hp_gradmul2l_p", "large_strides"], [131, 1, 1, "c.hp_gradmul2l_p", "ndims"], [131, 1, 1, "c.hp_gradmul2l_p", "out_shape"], [131, 1, 1, "c.hp_gradmul2l_p", "out_strides"], [131, 1, 1, "c.hp_gradmul2l_p", "small_multiples"], [131, 1, 1, "c.hp_gradmul2l_p", "small_shape"], [131, 1, 1, "c.hp_gradmul2l_p", "small_strides"], [131, 1, 1, "c.hp_gradmul2l_p", "tile_data0"], [131, 1, 1, "c.hp_gradmul2l_p", "tile_data1"], [131, 1, 1, "c.hp_gradmul2l_p", "x1"], [131, 1, 1, "c.hp_gradmul2l_p", "x2"]], "hp_gradmul2l_s": [[131, 1, 1, "c.hp_gradmul2l_s", "core_mask"], [131, 1, 1, "c.hp_gradmul2l_s", "dx1"], [131, 1, 1, "c.hp_gradmul2l_s", "dx2"], [131, 1, 1, "c.hp_gradmul2l_s", "dy"], [131, 1, 1, "c.hp_gradmul2l_s", "indices"], [131, 1, 1, "c.hp_gradmul2l_s", "large_multiples"], [131, 1, 1, "c.hp_gradmul2l_s", "large_shape"], [131, 1, 1, "c.hp_gradmul2l_s", "large_strides"], [131, 1, 1, "c.hp_gradmul2l_s", "ndims"], [131, 1, 1, "c.hp_gradmul2l_s", "out_shape"], [131, 1, 1, "c.hp_gradmul2l_s", "out_strides"], [131, 1, 1, "c.hp_gradmul2l_s", "small_multiples"], [131, 1, 1, "c.hp_gradmul2l_s", "small_shape"], [131, 1, 1, "c.hp_gradmul2l_s", "small_strides"], [131, 1, 1, "c.hp_gradmul2l_s", "tile_data0"], [131, 1, 1, "c.hp_gradmul2l_s", "tile_data1"], [131, 1, 1, "c.hp_gradmul2l_s", "x1"], [131, 1, 1, "c.hp_gradmul2l_s", "x2"]], "hp_gradmul_p": [[131, 1, 1, "c.hp_gradmul_p", "dx1"], [131, 1, 1, "c.hp_gradmul_p", "dx2"], [131, 1, 1, "c.hp_gradmul_p", "dy"], [131, 1, 1, "c.hp_gradmul_p", "large_multiples"], [131, 1, 1, "c.hp_gradmul_p", "large_shape"], [131, 1, 1, "c.hp_gradmul_p", "large_strides"], [131, 1, 1, "c.hp_gradmul_p", "ndims"], [131, 1, 1, "c.hp_gradmul_p", "out_shape"], [131, 1, 1, "c.hp_gradmul_p", "out_strides"], [131, 1, 1, "c.hp_gradmul_p", "small_multiples"], [131, 1, 1, "c.hp_gradmul_p", "small_shape"], [131, 1, 1, "c.hp_gradmul_p", "small_strides"], [131, 1, 1, "c.hp_gradmul_p", "tile_data0"], [131, 1, 1, "c.hp_gradmul_p", "tile_data1"], [131, 1, 1, "c.hp_gradmul_p", "x1"], [131, 1, 1, "c.hp_gradmul_p", "x2"]], "hp_gradmul_s": [[131, 1, 1, "c.hp_gradmul_s", "core_mask"], [131, 1, 1, "c.hp_gradmul_s", "dx1"], [131, 1, 1, "c.hp_gradmul_s", "dx2"], [131, 1, 1, "c.hp_gradmul_s", "dy"], [131, 1, 1, "c.hp_gradmul_s", "indices"], [131, 1, 1, "c.hp_gradmul_s", "large_multiples"], [131, 1, 1, "c.hp_gradmul_s", "large_shape"], [131, 1, 1, "c.hp_gradmul_s", "large_strides"], [131, 1, 1, "c.hp_gradmul_s", "ndims"], [131, 1, 1, "c.hp_gradmul_s", "out_shape"], [131, 1, 1, "c.hp_gradmul_s", "out_strides"], [131, 1, 1, "c.hp_gradmul_s", "small_multiples"], [131, 1, 1, "c.hp_gradmul_s", "small_shape"], [131, 1, 1, "c.hp_gradmul_s", "small_strides"], [131, 1, 1, "c.hp_gradmul_s", "tile_data0"], [131, 1, 1, "c.hp_gradmul_s", "tile_data1"], [131, 1, 1, "c.hp_gradmul_s", "x1"], [131, 1, 1, "c.hp_gradmul_s", "x2"]], "hp_greater_p": [[92, 1, 1, "c.hp_greater_p", "element_num"], [92, 1, 1, "c.hp_greater_p", "in_elements_num0"], [92, 1, 1, "c.hp_greater_p", "input1"], [92, 1, 1, "c.hp_greater_p", "input2"], [92, 1, 1, "c.hp_greater_p", "optimize"], [92, 1, 1, "c.hp_greater_p", "output"]], "hp_greater_s": [[92, 1, 1, "c.hp_greater_s", "core_mask"], [92, 1, 1, "c.hp_greater_s", "element_num"], [92, 1, 1, "c.hp_greater_s", "in_elements_num0"], [92, 1, 1, "c.hp_greater_s", "input1"], [92, 1, 1, "c.hp_greater_s", "input2"], [92, 1, 1, "c.hp_greater_s", "optimize"], [92, 1, 1, "c.hp_greater_s", "output"]], "hp_greaterequal_p": [[93, 1, 1, "c.hp_greaterequal_p", "element_num"], [93, 1, 1, "c.hp_greaterequal_p", "in_elements_num0"], [93, 1, 1, "c.hp_greaterequal_p", "input1"], [93, 1, 1, "c.hp_greaterequal_p", "input2"], [93, 1, 1, "c.hp_greaterequal_p", "optimize"], [93, 1, 1, "c.hp_greaterequal_p", "output"]], "hp_greaterequal_s": [[93, 1, 1, "c.hp_greaterequal_s", "core_mask"], [93, 1, 1, "c.hp_greaterequal_s", "element_num"], [93, 1, 1, "c.hp_greaterequal_s", "in_elements_num0"], [93, 1, 1, "c.hp_greaterequal_s", "input1"], [93, 1, 1, "c.hp_greaterequal_s", "input2"], [93, 1, 1, "c.hp_greaterequal_s", "optimize"], [93, 1, 1, "c.hp_greaterequal_s", "output"]], "hp_groupnormfusion_p": [[94, 1, 1, "c.hp_groupnormfusion_p", "batch"], [94, 1, 1, "c.hp_groupnormfusion_p", "channel"], [94, 1, 1, "c.hp_groupnormfusion_p", "epsilon"], [94, 1, 1, "c.hp_groupnormfusion_p", "input"], [94, 1, 1, "c.hp_groupnormfusion_p", "mean"], [94, 1, 1, "c.hp_groupnormfusion_p", "num_groups"], [94, 1, 1, "c.hp_groupnormfusion_p", "offset"], [94, 1, 1, "c.hp_groupnormfusion_p", "output"], [94, 1, 1, "c.hp_groupnormfusion_p", "scale"], [94, 1, 1, "c.hp_groupnormfusion_p", "unit"], [94, 1, 1, "c.hp_groupnormfusion_p", "variance"]], "hp_groupnormfusion_s": [[94, 1, 1, "c.hp_groupnormfusion_s", "batch"], [94, 1, 1, "c.hp_groupnormfusion_s", "channel"], [94, 1, 1, "c.hp_groupnormfusion_s", "core_mask"], [94, 1, 1, "c.hp_groupnormfusion_s", "epsilon"], [94, 1, 1, "c.hp_groupnormfusion_s", "input"], [94, 1, 1, "c.hp_groupnormfusion_s", "mean"], [94, 1, 1, "c.hp_groupnormfusion_s", "num_groups"], [94, 1, 1, "c.hp_groupnormfusion_s", "offset"], [94, 1, 1, "c.hp_groupnormfusion_s", "output"], [94, 1, 1, "c.hp_groupnormfusion_s", "scale"], [94, 1, 1, "c.hp_groupnormfusion_s", "unit"], [94, 1, 1, "c.hp_groupnormfusion_s", "variance"]], "hp_hardshrink_p": [[12, 1, 1, "c.hp_hardshrink_p", "Input0"], [12, 1, 1, "c.hp_hardshrink_p", "lambd"], [12, 1, 1, "c.hp_hardshrink_p", "length"], [12, 1, 1, "c.hp_hardshrink_p", "output"]], "hp_hardshrink_s": [[12, 1, 1, "c.hp_hardshrink_s", "Input0"], [12, 1, 1, "c.hp_hardshrink_s", "core_mask"], [12, 1, 1, "c.hp_hardshrink_s", "lambd"], [12, 1, 1, "c.hp_hardshrink_s", "length"], [12, 1, 1, "c.hp_hardshrink_s", "output"]], "hp_hardtanh_p": [[12, 1, 1, "c.hp_hardtanh_p", "Input0"], [12, 1, 1, "c.hp_hardtanh_p", "length"], [12, 1, 1, "c.hp_hardtanh_p", "max_val"], [12, 1, 1, "c.hp_hardtanh_p", "min_val"], [12, 1, 1, "c.hp_hardtanh_p", "output"]], "hp_hardtanh_s": [[12, 1, 1, "c.hp_hardtanh_s", "Input0"], [12, 1, 1, "c.hp_hardtanh_s", "core_mask"], [12, 1, 1, "c.hp_hardtanh_s", "length"], [12, 1, 1, "c.hp_hardtanh_s", "max_val"], [12, 1, 1, "c.hp_hardtanh_s", "min_val"], [12, 1, 1, "c.hp_hardtanh_s", "output"]], "hp_hsigmoid_p": [[12, 1, 1, "c.hp_hsigmoid_p", "Input0"], [12, 1, 1, "c.hp_hsigmoid_p", "length"], [12, 1, 1, "c.hp_hsigmoid_p", "output"]], "hp_hsigmoid_s": [[12, 1, 1, "c.hp_hsigmoid_s", "Input0"], [12, 1, 1, "c.hp_hsigmoid_s", "core_mask"], [12, 1, 1, "c.hp_hsigmoid_s", "length"], [12, 1, 1, "c.hp_hsigmoid_s", "output"]], "hp_hswish_p": [[12, 1, 1, "c.hp_hswish_p", "Input0"], [12, 1, 1, "c.hp_hswish_p", "length"], [12, 1, 1, "c.hp_hswish_p", "output"]], "hp_hswish_s": [[12, 1, 1, "c.hp_hswish_s", "Input0"], [12, 1, 1, "c.hp_hswish_s", "core_mask"], [12, 1, 1, "c.hp_hswish_s", "length"], [12, 1, 1, "c.hp_hswish_s", "output"]], "hp_instancenorm_p": [[97, 1, 1, "c.hp_instancenorm_p", "batch"], [97, 1, 1, "c.hp_instancenorm_p", "beta"], [97, 1, 1, "c.hp_instancenorm_p", "channel"], [97, 1, 1, "c.hp_instancenorm_p", "epsilon"], [97, 1, 1, "c.hp_instancenorm_p", "gamma"], [97, 1, 1, "c.hp_instancenorm_p", "inner_size"], [97, 1, 1, "c.hp_instancenorm_p", "input"], [97, 1, 1, "c.hp_instancenorm_p", "output"]], "hp_instancenorm_s": [[97, 1, 1, "c.hp_instancenorm_s", "batch"], [97, 1, 1, "c.hp_instancenorm_s", "beta"], [97, 1, 1, "c.hp_instancenorm_s", "channel"], [97, 1, 1, "c.hp_instancenorm_s", "core_mask"], [97, 1, 1, "c.hp_instancenorm_s", "epsilon"], [97, 1, 1, "c.hp_instancenorm_s", "gamma"], [97, 1, 1, "c.hp_instancenorm_s", "inner_size"], [97, 1, 1, "c.hp_instancenorm_s", "input"], [97, 1, 1, "c.hp_instancenorm_s", "output"]], "hp_isfinite_p": [[99, 1, 1, "c.hp_isfinite_p", "Input"], [99, 1, 1, "c.hp_isfinite_p", "length"], [99, 1, 1, "c.hp_isfinite_p", "output"]], "hp_isfinite_s": [[99, 1, 1, "c.hp_isfinite_s", "Input"], [99, 1, 1, "c.hp_isfinite_s", "core_mask"], [99, 1, 1, "c.hp_isfinite_s", "length"], [99, 1, 1, "c.hp_isfinite_s", "output"]], "hp_l2norm_p": [[100, 1, 1, "c.hp_l2norm_p", "Input"], [100, 1, 1, "c.hp_l2norm_p", "is_relu"], [100, 1, 1, "c.hp_l2norm_p", "is_relu6"], [100, 1, 1, "c.hp_l2norm_p", "length"], [100, 1, 1, "c.hp_l2norm_p", "output"], [100, 1, 1, "c.hp_l2norm_p", "sqrt_sum"]], "hp_l2norm_s": [[100, 1, 1, "c.hp_l2norm_s", "Input"], [100, 1, 1, "c.hp_l2norm_s", "core_mask"], [100, 1, 1, "c.hp_l2norm_s", "is_relu"], [100, 1, 1, "c.hp_l2norm_s", "is_relu6"], [100, 1, 1, "c.hp_l2norm_s", "length"], [100, 1, 1, "c.hp_l2norm_s", "output"], [100, 1, 1, "c.hp_l2norm_s", "sqrt_sum"]], "hp_layernormfusion_p": [[101, 1, 1, "c.hp_layernormfusion_p", "beta_data"], [101, 1, 1, "c.hp_layernormfusion_p", "dst_data"], [101, 1, 1, "c.hp_layernormfusion_p", "epsilon"], [101, 1, 1, "c.hp_layernormfusion_p", "gamma_data"], [101, 1, 1, "c.hp_layernormfusion_p", "norm_inner_size"], [101, 1, 1, "c.hp_layernormfusion_p", "norm_outer_size"], [101, 1, 1, "c.hp_layernormfusion_p", "out_mean"], [101, 1, 1, "c.hp_layernormfusion_p", "out_variance"], [101, 1, 1, "c.hp_layernormfusion_p", "param_inner_size"], [101, 1, 1, "c.hp_layernormfusion_p", "param_outer_size"], [101, 1, 1, "c.hp_layernormfusion_p", "src_data"]], "hp_layernormfusion_s": [[101, 1, 1, "c.hp_layernormfusion_s", "beta_data"], [101, 1, 1, "c.hp_layernormfusion_s", "core_mask"], [101, 1, 1, "c.hp_layernormfusion_s", "dst_data"], [101, 1, 1, "c.hp_layernormfusion_s", "epsilon"], [101, 1, 1, "c.hp_layernormfusion_s", "gamma_data"], [101, 1, 1, "c.hp_layernormfusion_s", "length"], [101, 1, 1, "c.hp_layernormfusion_s", "norm_inner_size"], [101, 1, 1, "c.hp_layernormfusion_s", "norm_outer_size"], [101, 1, 1, "c.hp_layernormfusion_s", "out_mean"], [101, 1, 1, "c.hp_layernormfusion_s", "out_variance"], [101, 1, 1, "c.hp_layernormfusion_s", "param_inner_size"], [101, 1, 1, "c.hp_layernormfusion_s", "param_outer_size"], [101, 1, 1, "c.hp_layernormfusion_s", "src_data"]], "hp_layernormgrad_p": [[102, 1, 1, "c.hp_layernormgrad_p", "block_num"], [102, 1, 1, "c.hp_layernormgrad_p", "block_size"], [102, 1, 1, "c.hp_layernormgrad_p", "db"], [102, 1, 1, "c.hp_layernormgrad_p", "dg"], [102, 1, 1, "c.hp_layernormgrad_p", "dx"], [102, 1, 1, "c.hp_layernormgrad_p", "dy"], [102, 1, 1, "c.hp_layernormgrad_p", "gamma"], [102, 1, 1, "c.hp_layernormgrad_p", "mean"], [102, 1, 1, "c.hp_layernormgrad_p", "param_num"], [102, 1, 1, "c.hp_layernormgrad_p", "param_size"], [102, 1, 1, "c.hp_layernormgrad_p", "var"], [102, 1, 1, "c.hp_layernormgrad_p", "x"]], "hp_layernormgrad_s": [[102, 1, 1, "c.hp_layernormgrad_s", "block_num"], [102, 1, 1, "c.hp_layernormgrad_s", "block_size"], [102, 1, 1, "c.hp_layernormgrad_s", "core_mask"], [102, 1, 1, "c.hp_layernormgrad_s", "db"], [102, 1, 1, "c.hp_layernormgrad_s", "dg"], [102, 1, 1, "c.hp_layernormgrad_s", "dx"], [102, 1, 1, "c.hp_layernormgrad_s", "dy"], [102, 1, 1, "c.hp_layernormgrad_s", "gamma"], [102, 1, 1, "c.hp_layernormgrad_s", "mean"], [102, 1, 1, "c.hp_layernormgrad_s", "param_num"], [102, 1, 1, "c.hp_layernormgrad_s", "param_size"], [102, 1, 1, "c.hp_layernormgrad_s", "var"], [102, 1, 1, "c.hp_layernormgrad_s", "x"]], "hp_leaky_relu_p": [[103, 1, 1, "c.hp_leaky_relu_p", "alpha"], [103, 1, 1, "c.hp_leaky_relu_p", "core_mask"], [103, 1, 1, "c.hp_leaky_relu_p", "elem_cnt"], [103, 1, 1, "c.hp_leaky_relu_p", "input"], [103, 1, 1, "c.hp_leaky_relu_p", "output"]], "hp_leaky_relu_s": [[103, 1, 1, "c.hp_leaky_relu_s", "alpha"], [103, 1, 1, "c.hp_leaky_relu_s", "core_mask"], [103, 1, 1, "c.hp_leaky_relu_s", "elem_cnt"], [103, 1, 1, "c.hp_leaky_relu_s", "input"], [103, 1, 1, "c.hp_leaky_relu_s", "output"]], "hp_less_p": [[104, 1, 1, "c.hp_less_p", "Input0"], [104, 1, 1, "c.hp_less_p", "Input1"], [104, 1, 1, "c.hp_less_p", "in_elements_num0"], [104, 1, 1, "c.hp_less_p", "length"], [104, 1, 1, "c.hp_less_p", "optimize"], [104, 1, 1, "c.hp_less_p", "output"]], "hp_less_s": [[104, 1, 1, "c.hp_less_s", "Input0"], [104, 1, 1, "c.hp_less_s", "Input1"], [104, 1, 1, "c.hp_less_s", "core_mask"], [104, 1, 1, "c.hp_less_s", "in_elements_num0"], [104, 1, 1, "c.hp_less_s", "length"], [104, 1, 1, "c.hp_less_s", "optimize"], [104, 1, 1, "c.hp_less_s", "output"]], "hp_lessequal_p": [[105, 1, 1, "c.hp_lessequal_p", "Input0"], [105, 1, 1, "c.hp_lessequal_p", "Input1"], [105, 1, 1, "c.hp_lessequal_p", "in_elements_num0"], [105, 1, 1, "c.hp_lessequal_p", "length"], [105, 1, 1, "c.hp_lessequal_p", "optimize"], [105, 1, 1, "c.hp_lessequal_p", "output"]], "hp_lessequal_s": [[105, 1, 1, "c.hp_lessequal_s", "Input0"], [105, 1, 1, "c.hp_lessequal_s", "Input1"], [105, 1, 1, "c.hp_lessequal_s", "core_mask"], [105, 1, 1, "c.hp_lessequal_s", "in_elements_num0"], [105, 1, 1, "c.hp_lessequal_s", "length"], [105, 1, 1, "c.hp_lessequal_s", "optimize"], [105, 1, 1, "c.hp_lessequal_s", "output"]], "hp_log1p_p": [[108, 1, 1, "c.hp_log1p_p", "Input"], [108, 1, 1, "c.hp_log1p_p", "length"], [108, 1, 1, "c.hp_log1p_p", "output"]], "hp_log1p_s": [[108, 1, 1, "c.hp_log1p_s", "Input"], [108, 1, 1, "c.hp_log1p_s", "core_mask"], [108, 1, 1, "c.hp_log1p_s", "length"], [108, 1, 1, "c.hp_log1p_s", "output"]], "hp_log_grad_p": [[109, 1, 1, "c.hp_log_grad_p", "input0"], [109, 1, 1, "c.hp_log_grad_p", "input1"], [109, 1, 1, "c.hp_log_grad_p", "length"], [109, 1, 1, "c.hp_log_grad_p", "output"]], "hp_log_grad_s": [[109, 1, 1, "c.hp_log_grad_s", "core_mask"], [109, 1, 1, "c.hp_log_grad_s", "input0"], [109, 1, 1, "c.hp_log_grad_s", "input1"], [109, 1, 1, "c.hp_log_grad_s", "length"], [109, 1, 1, "c.hp_log_grad_s", "output"]], "hp_log_p": [[107, 1, 1, "c.hp_log_p", "input"], [107, 1, 1, "c.hp_log_p", "length"], [107, 1, 1, "c.hp_log_p", "output"]], "hp_log_s": [[107, 1, 1, "c.hp_log_s", "core_mask"], [107, 1, 1, "c.hp_log_s", "input"], [107, 1, 1, "c.hp_log_s", "length"], [107, 1, 1, "c.hp_log_s", "output"]], "hp_logical_not_p": [[110, 1, 1, "c.hp_logical_not_p", "input"], [110, 1, 1, "c.hp_logical_not_p", "length"], [110, 1, 1, "c.hp_logical_not_p", "output"]], "hp_logical_not_s": [[110, 1, 1, "c.hp_logical_not_s", "core_mask"], [110, 1, 1, "c.hp_logical_not_s", "input"], [110, 1, 1, "c.hp_logical_not_s", "length"], [110, 1, 1, "c.hp_logical_not_s", "output"]], "hp_logical_or_p": [[111, 1, 1, "c.hp_logical_or_p", "input0"], [111, 1, 1, "c.hp_logical_or_p", "input1"], [111, 1, 1, "c.hp_logical_or_p", "length"], [111, 1, 1, "c.hp_logical_or_p", "output"]], "hp_logical_or_s": [[111, 1, 1, "c.hp_logical_or_s", "core_mask"], [111, 1, 1, "c.hp_logical_or_s", "input0"], [111, 1, 1, "c.hp_logical_or_s", "input1"], [111, 1, 1, "c.hp_logical_or_s", "length"], [111, 1, 1, "c.hp_logical_or_s", "output"]], "hp_logsoftmax_p": [[113, 1, 1, "c.hp_logsoftmax_p", "axis"], [113, 1, 1, "c.hp_logsoftmax_p", "axis_size"], [113, 1, 1, "c.hp_logsoftmax_p", "inner_size"], [113, 1, 1, "c.hp_logsoftmax_p", "input_ptr"], [113, 1, 1, "c.hp_logsoftmax_p", "n_dim"], [113, 1, 1, "c.hp_logsoftmax_p", "output_ptr"], [113, 1, 1, "c.hp_logsoftmax_p", "outter_size"], [113, 1, 1, "c.hp_logsoftmax_p", "sum_data"]], "hp_logsoftmax_s": [[113, 1, 1, "c.hp_logsoftmax_s", "axis"], [113, 1, 1, "c.hp_logsoftmax_s", "axis_size"], [113, 1, 1, "c.hp_logsoftmax_s", "core_mask"], [113, 1, 1, "c.hp_logsoftmax_s", "inner_size"], [113, 1, 1, "c.hp_logsoftmax_s", "input_ptr"], [113, 1, 1, "c.hp_logsoftmax_s", "n_dim"], [113, 1, 1, "c.hp_logsoftmax_s", "output_ptr"], [113, 1, 1, "c.hp_logsoftmax_s", "outter_size"], [113, 1, 1, "c.hp_logsoftmax_s", "sum_data"]], "hp_lpnorm_p": [[114, 1, 1, "c.hp_lpnorm_p", "batch"], [114, 1, 1, "c.hp_lpnorm_p", "beta"], [114, 1, 1, "c.hp_lpnorm_p", "channel"], [114, 1, 1, "c.hp_lpnorm_p", "epsilon"], [114, 1, 1, "c.hp_lpnorm_p", "gamma"], [114, 1, 1, "c.hp_lpnorm_p", "inner_size"], [114, 1, 1, "c.hp_lpnorm_p", "input"], [114, 1, 1, "c.hp_lpnorm_p", "output"], [114, 1, 1, "c.hp_lpnorm_p", "p"]], "hp_lpnorm_s": [[114, 1, 1, "c.hp_lpnorm_s", "batch"], [114, 1, 1, "c.hp_lpnorm_s", "beta"], [114, 1, 1, "c.hp_lpnorm_s", "channel"], [114, 1, 1, "c.hp_lpnorm_s", "core_mask"], [114, 1, 1, "c.hp_lpnorm_s", "epsilon"], [114, 1, 1, "c.hp_lpnorm_s", "gamma"], [114, 1, 1, "c.hp_lpnorm_s", "inner_size"], [114, 1, 1, "c.hp_lpnorm_s", "input"], [114, 1, 1, "c.hp_lpnorm_s", "output"], [114, 1, 1, "c.hp_lpnorm_s", "p"]], "hp_lrelu_p": [[12, 1, 1, "c.hp_lrelu_p", "Input0"], [12, 1, 1, "c.hp_lrelu_p", "alpha"], [12, 1, 1, "c.hp_lrelu_p", "length"], [12, 1, 1, "c.hp_lrelu_p", "output"]], "hp_lrelu_s": [[12, 1, 1, "c.hp_lrelu_s", "Input0"], [12, 1, 1, "c.hp_lrelu_s", "alpha"], [12, 1, 1, "c.hp_lrelu_s", "core_mask"], [12, 1, 1, "c.hp_lrelu_s", "length"], [12, 1, 1, "c.hp_lrelu_s", "output"]], "hp_lrn_p": [[115, 1, 1, "c.hp_lrn_p", "alpha"], [115, 1, 1, "c.hp_lrn_p", "beta"], [115, 1, 1, "c.hp_lrn_p", "bias"], [115, 1, 1, "c.hp_lrn_p", "channel"], [115, 1, 1, "c.hp_lrn_p", "depth_radius"], [115, 1, 1, "c.hp_lrn_p", "input"], [115, 1, 1, "c.hp_lrn_p", "out_size"], [115, 1, 1, "c.hp_lrn_p", "output"]], "hp_lrn_s": [[115, 1, 1, "c.hp_lrn_s", "alpha"], [115, 1, 1, "c.hp_lrn_s", "beta"], [115, 1, 1, "c.hp_lrn_s", "bias"], [115, 1, 1, "c.hp_lrn_s", "channel"], [115, 1, 1, "c.hp_lrn_s", "core_mask"], [115, 1, 1, "c.hp_lrn_s", "depth_radius"], [115, 1, 1, "c.hp_lrn_s", "input"], [115, 1, 1, "c.hp_lrn_s", "out_size"], [115, 1, 1, "c.hp_lrn_s", "output"]], "hp_lsh_projection_p": [[116, 1, 1, "c.hp_lsh_projection_p", "bits_per_hash"], [116, 1, 1, "c.hp_lsh_projection_p", "feature"], [116, 1, 1, "c.hp_lsh_projection_p", "feature_num"], [116, 1, 1, "c.hp_lsh_projection_p", "hash_group_num"], [116, 1, 1, "c.hp_lsh_projection_p", "hash_seed"], [116, 1, 1, "c.hp_lsh_projection_p", "output"], [116, 1, 1, "c.hp_lsh_projection_p", "weight"]], "hp_lsh_projection_s": [[116, 1, 1, "c.hp_lsh_projection_s", "bits_per_hash"], [116, 1, 1, "c.hp_lsh_projection_s", "core_mask"], [116, 1, 1, "c.hp_lsh_projection_s", "feature"], [116, 1, 1, "c.hp_lsh_projection_s", "feature_num"], [116, 1, 1, "c.hp_lsh_projection_s", "hash_group_num"], [116, 1, 1, "c.hp_lsh_projection_s", "hash_seed"], [116, 1, 1, "c.hp_lsh_projection_s", "output"], [116, 1, 1, "c.hp_lsh_projection_s", "weight"]], "hp_lstmgrad_p": [[118, 1, 1, "c.hp_lstmgrad_p", "dynamic_params"], [118, 1, 1, "c.hp_lstmgrad_p", "params"]], "hp_lstmgrad_s": [[118, 1, 1, "c.hp_lstmgrad_s", "core_mask"], [118, 1, 1, "c.hp_lstmgrad_s", "dynamic_params"], [118, 1, 1, "c.hp_lstmgrad_s", "params"]], "hp_lstmgraddata_p": [[119, 1, 1, "c.hp_lstmgraddata_p", "dynamic_params"], [119, 1, 1, "c.hp_lstmgraddata_p", "params"]], "hp_lstmgraddata_s": [[119, 1, 1, "c.hp_lstmgraddata_s", "core_mask"], [119, 1, 1, "c.hp_lstmgraddata_s", "dynamic_params"], [119, 1, 1, "c.hp_lstmgraddata_s", "params"]], "hp_lstmgradweight_p": [[120, 1, 1, "c.hp_lstmgradweight_p", "dynamic_params"], [120, 1, 1, "c.hp_lstmgradweight_p", "params"]], "hp_lstmgradweight_s": [[120, 1, 1, "c.hp_lstmgradweight_s", "core_mask"], [120, 1, 1, "c.hp_lstmgradweight_s", "dynamic_params"], [120, 1, 1, "c.hp_lstmgradweight_s", "params"]], "hp_maximum_p": [[122, 1, 1, "c.hp_maximum_p", "input0"], [122, 1, 1, "c.hp_maximum_p", "input1"], [122, 1, 1, "c.hp_maximum_p", "length"], [122, 1, 1, "c.hp_maximum_p", "output"]], "hp_maximum_s": [[122, 1, 1, "c.hp_maximum_s", "core_mask"], [122, 1, 1, "c.hp_maximum_s", "input0"], [122, 1, 1, "c.hp_maximum_s", "input1"], [122, 1, 1, "c.hp_maximum_s", "length"], [122, 1, 1, "c.hp_maximum_s", "output"]], "hp_maximumgrad_p": [[123, 1, 1, "c.hp_maximumgrad_p", "Input0"], [123, 1, 1, "c.hp_maximumgrad_p", "Input0_dims"], [123, 1, 1, "c.hp_maximumgrad_p", "Input1"], [123, 1, 1, "c.hp_maximumgrad_p", "Input1_dims"], [123, 1, 1, "c.hp_maximumgrad_p", "dx0"], [123, 1, 1, "c.hp_maximumgrad_p", "dx1"], [123, 1, 1, "c.hp_maximumgrad_p", "dy"], [123, 1, 1, "c.hp_maximumgrad_p", "num_dims"]], "hp_maximumgrad_s": [[123, 1, 1, "c.hp_maximumgrad_s", "Input0"], [123, 1, 1, "c.hp_maximumgrad_s", "Input0_dims"], [123, 1, 1, "c.hp_maximumgrad_s", "Input1"], [123, 1, 1, "c.hp_maximumgrad_s", "Input1_dims"], [123, 1, 1, "c.hp_maximumgrad_s", "core_mask"], [123, 1, 1, "c.hp_maximumgrad_s", "dx0"], [123, 1, 1, "c.hp_maximumgrad_s", "dx1"], [123, 1, 1, "c.hp_maximumgrad_s", "dy"], [123, 1, 1, "c.hp_maximumgrad_s", "num_dims"]], "hp_maxpool_fusion_p": [[124, 1, 1, "c.hp_maxpool_fusion_p", "params"]], "hp_maxpool_fusion_s": [[124, 1, 1, "c.hp_maxpool_fusion_s", "core_mask"], [124, 1, 1, "c.hp_maxpool_fusion_s", "params"]], "hp_maxpool_grad_p": [[125, 1, 1, "c.hp_maxpool_grad_p", "params"]], "hp_maxpool_grad_s": [[125, 1, 1, "c.hp_maxpool_grad_s", "core_mask"], [125, 1, 1, "c.hp_maxpool_grad_s", "params"]], "hp_mfcc_p": [[126, 1, 1, "c.hp_mfcc_p", "mfcc_params"], [126, 1, 1, "c.hp_mfcc_p", "mfcc_workspace"], [126, 1, 1, "c.hp_mfcc_p", "spec_params"], [126, 1, 1, "c.hp_mfcc_p", "spec_workspace"]], "hp_mfcc_s": [[126, 1, 1, "c.hp_mfcc_s", "core_mask"], [126, 1, 1, "c.hp_mfcc_s", "mfcc_params"], [126, 1, 1, "c.hp_mfcc_s", "mfcc_workspace"], [126, 1, 1, "c.hp_mfcc_s", "spec_params"], [126, 1, 1, "c.hp_mfcc_s", "spec_workspace"]], "hp_minimum_p": [[127, 1, 1, "c.hp_minimum_p", "input0"], [127, 1, 1, "c.hp_minimum_p", "input1"], [127, 1, 1, "c.hp_minimum_p", "length"], [127, 1, 1, "c.hp_minimum_p", "output"]], "hp_minimum_s": [[127, 1, 1, "c.hp_minimum_s", "core_mask"], [127, 1, 1, "c.hp_minimum_s", "input0"], [127, 1, 1, "c.hp_minimum_s", "input1"], [127, 1, 1, "c.hp_minimum_s", "length"], [127, 1, 1, "c.hp_minimum_s", "output"]], "hp_minimumgrad_p": [[128, 1, 1, "c.hp_minimumgrad_p", "Input0"], [128, 1, 1, "c.hp_minimumgrad_p", "Input0_dims"], [128, 1, 1, "c.hp_minimumgrad_p", "Input1"], [128, 1, 1, "c.hp_minimumgrad_p", "Input1_dims"], [128, 1, 1, "c.hp_minimumgrad_p", "dx0"], [128, 1, 1, "c.hp_minimumgrad_p", "dx1"], [128, 1, 1, "c.hp_minimumgrad_p", "dy"], [128, 1, 1, "c.hp_minimumgrad_p", "num_dims"]], "hp_minimumgrad_s": [[128, 1, 1, "c.hp_minimumgrad_s", "Input0"], [128, 1, 1, "c.hp_minimumgrad_s", "Input0_dims"], [128, 1, 1, "c.hp_minimumgrad_s", "Input1"], [128, 1, 1, "c.hp_minimumgrad_s", "Input1_dims"], [128, 1, 1, "c.hp_minimumgrad_s", "core_mask"], [128, 1, 1, "c.hp_minimumgrad_s", "dx0"], [128, 1, 1, "c.hp_minimumgrad_s", "dx1"], [128, 1, 1, "c.hp_minimumgrad_s", "dy"], [128, 1, 1, "c.hp_minimumgrad_s", "num_dims"]], "hp_mod_p": [[129, 1, 1, "c.hp_mod_p", "input0"], [129, 1, 1, "c.hp_mod_p", "input1"], [129, 1, 1, "c.hp_mod_p", "length"], [129, 1, 1, "c.hp_mod_p", "output"]], "hp_mod_s": [[129, 1, 1, "c.hp_mod_s", "core_mask"], [129, 1, 1, "c.hp_mod_s", "input0"], [129, 1, 1, "c.hp_mod_s", "input1"], [129, 1, 1, "c.hp_mod_s", "length"], [129, 1, 1, "c.hp_mod_s", "output"]], "hp_mul_p": [[130, 1, 1, "c.hp_mul_p", "input0"], [130, 1, 1, "c.hp_mul_p", "input1"], [130, 1, 1, "c.hp_mul_p", "length"], [130, 1, 1, "c.hp_mul_p", "output"]], "hp_mul_s": [[130, 1, 1, "c.hp_mul_s", "core_mask"], [130, 1, 1, "c.hp_mul_s", "input0"], [130, 1, 1, "c.hp_mul_s", "input1"], [130, 1, 1, "c.hp_mul_s", "length"], [130, 1, 1, "c.hp_mul_s", "output"]], "hp_neg_grad_p": [[133, 1, 1, "c.hp_neg_grad_p", "Input"], [133, 1, 1, "c.hp_neg_grad_p", "length"], [133, 1, 1, "c.hp_neg_grad_p", "output"]], "hp_neg_grad_s": [[133, 1, 1, "c.hp_neg_grad_s", "Input"], [133, 1, 1, "c.hp_neg_grad_s", "core_mask"], [133, 1, 1, "c.hp_neg_grad_s", "length"], [133, 1, 1, "c.hp_neg_grad_s", "output"]], "hp_neg_p": [[132, 1, 1, "c.hp_neg_p", "Input"], [132, 1, 1, "c.hp_neg_p", "length"], [132, 1, 1, "c.hp_neg_p", "output"]], "hp_neg_s": [[132, 1, 1, "c.hp_neg_s", "Input"], [132, 1, 1, "c.hp_neg_s", "core_mask"], [132, 1, 1, "c.hp_neg_s", "length"], [132, 1, 1, "c.hp_neg_s", "output"]], "hp_nllloss_p": [[134, 1, 1, "c.hp_nllloss_p", "batch_size"], [134, 1, 1, "c.hp_nllloss_p", "class_num"], [134, 1, 1, "c.hp_nllloss_p", "labels"], [134, 1, 1, "c.hp_nllloss_p", "log_probs"], [134, 1, 1, "c.hp_nllloss_p", "loss"], [134, 1, 1, "c.hp_nllloss_p", "reduction_type"], [134, 1, 1, "c.hp_nllloss_p", "total_weight"], [134, 1, 1, "c.hp_nllloss_p", "weight"]], "hp_nllloss_s": [[134, 1, 1, "c.hp_nllloss_s", "batch_size"], [134, 1, 1, "c.hp_nllloss_s", "class_num"], [134, 1, 1, "c.hp_nllloss_s", "core_mask"], [134, 1, 1, "c.hp_nllloss_s", "labels"], [134, 1, 1, "c.hp_nllloss_s", "log_probs"], [134, 1, 1, "c.hp_nllloss_s", "loss"], [134, 1, 1, "c.hp_nllloss_s", "reduction_type"], [134, 1, 1, "c.hp_nllloss_s", "total_weight"], [134, 1, 1, "c.hp_nllloss_s", "weight"]], "hp_nlllossgrad_p": [[135, 1, 1, "c.hp_nlllossgrad_p", "batch"], [135, 1, 1, "c.hp_nlllossgrad_p", "class_num"], [135, 1, 1, "c.hp_nlllossgrad_p", "labels"], [135, 1, 1, "c.hp_nlllossgrad_p", "logits"], [135, 1, 1, "c.hp_nlllossgrad_p", "logits_grad"], [135, 1, 1, "c.hp_nlllossgrad_p", "loss_grad"], [135, 1, 1, "c.hp_nlllossgrad_p", "reduction_type"], [135, 1, 1, "c.hp_nlllossgrad_p", "total_weight"], [135, 1, 1, "c.hp_nlllossgrad_p", "weight"]], "hp_nlllossgrad_s": [[135, 1, 1, "c.hp_nlllossgrad_s", "batch"], [135, 1, 1, "c.hp_nlllossgrad_s", "class_num"], [135, 1, 1, "c.hp_nlllossgrad_s", "core_mask"], [135, 1, 1, "c.hp_nlllossgrad_s", "labels"], [135, 1, 1, "c.hp_nlllossgrad_s", "logits"], [135, 1, 1, "c.hp_nlllossgrad_s", "logits_grad"], [135, 1, 1, "c.hp_nlllossgrad_s", "loss_grad"], [135, 1, 1, "c.hp_nlllossgrad_s", "reduction_type"], [135, 1, 1, "c.hp_nlllossgrad_s", "total_weight"], [135, 1, 1, "c.hp_nlllossgrad_s", "weight"]], "hp_non_max_suppression_p": [[136, 1, 1, "c.hp_non_max_suppression_p", "param"]], "hp_non_max_suppression_s": [[136, 1, 1, "c.hp_non_max_suppression_s", "core_mask"], [136, 1, 1, "c.hp_non_max_suppression_s", "param"]], "hp_not_equal_p": [[138, 1, 1, "c.hp_not_equal_p", "Input0"], [138, 1, 1, "c.hp_not_equal_p", "Input1"], [138, 1, 1, "c.hp_not_equal_p", "length"], [138, 1, 1, "c.hp_not_equal_p", "output"]], "hp_not_equal_s": [[138, 1, 1, "c.hp_not_equal_s", "Input0"], [138, 1, 1, "c.hp_not_equal_s", "Input1"], [138, 1, 1, "c.hp_not_equal_s", "core_mask"], [138, 1, 1, "c.hp_not_equal_s", "length"], [138, 1, 1, "c.hp_not_equal_s", "output"]], "hp_onehot_p": [[139, 1, 1, "c.hp_onehot_p", "axis"], [139, 1, 1, "c.hp_onehot_p", "depth"], [139, 1, 1, "c.hp_onehot_p", "indices"], [139, 1, 1, "c.hp_onehot_p", "indices_shape"], [139, 1, 1, "c.hp_onehot_p", "indices_shape_size"], [139, 1, 1, "c.hp_onehot_p", "on_off"], [139, 1, 1, "c.hp_onehot_p", "output"], [139, 1, 1, "c.hp_onehot_p", "support_neg_index"]], "hp_onehot_s": [[139, 1, 1, "c.hp_onehot_s", "axis"], [139, 1, 1, "c.hp_onehot_s", "core_mask"], [139, 1, 1, "c.hp_onehot_s", "depth"], [139, 1, 1, "c.hp_onehot_s", "indices"], [139, 1, 1, "c.hp_onehot_s", "indices_shape"], [139, 1, 1, "c.hp_onehot_s", "indices_shape_size"], [139, 1, 1, "c.hp_onehot_s", "on_off"], [139, 1, 1, "c.hp_onehot_s", "output"], [139, 1, 1, "c.hp_onehot_s", "support_neg_index"]], "hp_ones_like_p": [[140, 1, 1, "c.hp_ones_like_p", "length"], [140, 1, 1, "c.hp_ones_like_p", "output"]], "hp_ones_like_s": [[140, 1, 1, "c.hp_ones_like_s", "core_mask"], [140, 1, 1, "c.hp_ones_like_s", "length"], [140, 1, 1, "c.hp_ones_like_s", "output"]], "hp_padfusion_p": [[141, 1, 1, "c.hp_padfusion_p", "params"]], "hp_padfusion_s": [[141, 1, 1, "c.hp_padfusion_s", "core_mask"], [141, 1, 1, "c.hp_padfusion_s", "params"]], "hp_pow_fusion_p": [[142, 1, 1, "c.hp_pow_fusion_p", "Input"], [142, 1, 1, "c.hp_pow_fusion_p", "broadcast"], [142, 1, 1, "c.hp_pow_fusion_p", "exponent"], [142, 1, 1, "c.hp_pow_fusion_p", "length_in"], [142, 1, 1, "c.hp_pow_fusion_p", "output"], [142, 1, 1, "c.hp_pow_fusion_p", "scale"], [142, 1, 1, "c.hp_pow_fusion_p", "shift"]], "hp_pow_fusion_s": [[142, 1, 1, "c.hp_pow_fusion_s", "Input"], [142, 1, 1, "c.hp_pow_fusion_s", "broadcast"], [142, 1, 1, "c.hp_pow_fusion_s", "core_mask"], [142, 1, 1, "c.hp_pow_fusion_s", "exponent"], [142, 1, 1, "c.hp_pow_fusion_s", "length_in"], [142, 1, 1, "c.hp_pow_fusion_s", "output"], [142, 1, 1, "c.hp_pow_fusion_s", "scale"], [142, 1, 1, "c.hp_pow_fusion_s", "shift"]], "hp_power_grad_p": [[143, 1, 1, "c.hp_power_grad_p", "Input1"], [143, 1, 1, "c.hp_power_grad_p", "Input2"], [143, 1, 1, "c.hp_power_grad_p", "length"], [143, 1, 1, "c.hp_power_grad_p", "output"], [143, 1, 1, "c.hp_power_grad_p", "power"], [143, 1, 1, "c.hp_power_grad_p", "scale"], [143, 1, 1, "c.hp_power_grad_p", "shift"]], "hp_power_grad_s": [[143, 1, 1, "c.hp_power_grad_s", "Input1"], [143, 1, 1, "c.hp_power_grad_s", "Input2"], [143, 1, 1, "c.hp_power_grad_s", "core_mask"], [143, 1, 1, "c.hp_power_grad_s", "length"], [143, 1, 1, "c.hp_power_grad_s", "output"], [143, 1, 1, "c.hp_power_grad_s", "power"], [143, 1, 1, "c.hp_power_grad_s", "scale"], [143, 1, 1, "c.hp_power_grad_s", "shift"]], "hp_prelufusion_p": [[144, 1, 1, "c.hp_prelufusion_p", "dst_data"], [144, 1, 1, "c.hp_prelufusion_p", "end"], [144, 1, 1, "c.hp_prelufusion_p", "slope"], [144, 1, 1, "c.hp_prelufusion_p", "src_data"], [144, 1, 1, "c.hp_prelufusion_p", "start"]], "hp_prelufusion_s": [[144, 1, 1, "c.hp_prelufusion_s", "core_mask"], [144, 1, 1, "c.hp_prelufusion_s", "dst_data"], [144, 1, 1, "c.hp_prelufusion_s", "end"], [144, 1, 1, "c.hp_prelufusion_s", "slope"], [144, 1, 1, "c.hp_prelufusion_s", "src_data"], [144, 1, 1, "c.hp_prelufusion_s", "start"]], "hp_priorbox_p": [[145, 1, 1, "c.hp_priorbox_p", "different_aspect_ratios"], [145, 1, 1, "c.hp_priorbox_p", "different_aspect_ratios_size"], [145, 1, 1, "c.hp_priorbox_p", "fmap_h"], [145, 1, 1, "c.hp_priorbox_p", "fmap_w"], [145, 1, 1, "c.hp_priorbox_p", "max_sizes"], [145, 1, 1, "c.hp_priorbox_p", "max_sizes_size"], [145, 1, 1, "c.hp_priorbox_p", "min_sizes"], [145, 1, 1, "c.hp_priorbox_p", "min_sizes_size"], [145, 1, 1, "c.hp_priorbox_p", "offset"], [145, 1, 1, "c.hp_priorbox_p", "output"], [145, 1, 1, "c.hp_priorbox_p", "output_size"], [145, 1, 1, "c.hp_priorbox_p", "step_h"], [145, 1, 1, "c.hp_priorbox_p", "step_w"]], "hp_priorbox_s": [[145, 1, 1, "c.hp_priorbox_s", "core_mask"], [145, 1, 1, "c.hp_priorbox_s", "different_aspect_ratios"], [145, 1, 1, "c.hp_priorbox_s", "different_aspect_ratios_size"], [145, 1, 1, "c.hp_priorbox_s", "fmap_h"], [145, 1, 1, "c.hp_priorbox_s", "fmap_w"], [145, 1, 1, "c.hp_priorbox_s", "max_sizes"], [145, 1, 1, "c.hp_priorbox_s", "max_sizes_size"], [145, 1, 1, "c.hp_priorbox_s", "min_sizes"], [145, 1, 1, "c.hp_priorbox_s", "min_sizes_size"], [145, 1, 1, "c.hp_priorbox_s", "offset"], [145, 1, 1, "c.hp_priorbox_s", "output"], [145, 1, 1, "c.hp_priorbox_s", "output_size"], [145, 1, 1, "c.hp_priorbox_s", "step_h"], [145, 1, 1, "c.hp_priorbox_s", "step_w"]], "hp_random_normal_p": [[148, 1, 1, "c.hp_random_normal_p", "length"], [148, 1, 1, "c.hp_random_normal_p", "mean"], [148, 1, 1, "c.hp_random_normal_p", "output"], [148, 1, 1, "c.hp_random_normal_p", "scale"], [148, 1, 1, "c.hp_random_normal_p", "seed"]], "hp_random_normal_s": [[148, 1, 1, "c.hp_random_normal_s", "core_mask"], [148, 1, 1, "c.hp_random_normal_s", "length"], [148, 1, 1, "c.hp_random_normal_s", "mean"], [148, 1, 1, "c.hp_random_normal_s", "output"], [148, 1, 1, "c.hp_random_normal_s", "scale"], [148, 1, 1, "c.hp_random_normal_s", "seed"]], "hp_random_standard_normal_p": [[149, 1, 1, "c.hp_random_standard_normal_p", "length"], [149, 1, 1, "c.hp_random_standard_normal_p", "output"], [149, 1, 1, "c.hp_random_standard_normal_p", "seed"]], "hp_random_standard_normal_s": [[149, 1, 1, "c.hp_random_standard_normal_s", "core_mask"], [149, 1, 1, "c.hp_random_standard_normal_s", "length"], [149, 1, 1, "c.hp_random_standard_normal_s", "output"], [149, 1, 1, "c.hp_random_standard_normal_s", "seed"]], "hp_real_div_p": [[152, 1, 1, "c.hp_real_div_p", "input0"], [152, 1, 1, "c.hp_real_div_p", "input1"], [152, 1, 1, "c.hp_real_div_p", "length"], [152, 1, 1, "c.hp_real_div_p", "output"]], "hp_real_div_s": [[152, 1, 1, "c.hp_real_div_s", "core_mask"], [152, 1, 1, "c.hp_real_div_s", "input0"], [152, 1, 1, "c.hp_real_div_s", "input1"], [152, 1, 1, "c.hp_real_div_s", "length"], [152, 1, 1, "c.hp_real_div_s", "output"]], "hp_reciprocal_p": [[153, 1, 1, "c.hp_reciprocal_p", "Input"], [153, 1, 1, "c.hp_reciprocal_p", "length"], [153, 1, 1, "c.hp_reciprocal_p", "output"]], "hp_reciprocal_s": [[153, 1, 1, "c.hp_reciprocal_s", "Input"], [153, 1, 1, "c.hp_reciprocal_s", "core_mask"], [153, 1, 1, "c.hp_reciprocal_s", "length"], [153, 1, 1, "c.hp_reciprocal_s", "output"]], "hp_reduce_p": [[154, 1, 1, "c.hp_reduce_p", "core_mask"], [154, 1, 1, "c.hp_reduce_p", "dst_data"], [154, 1, 1, "c.hp_reduce_p", "param"], [154, 1, 1, "c.hp_reduce_p", "src_data"], [154, 1, 1, "c.hp_reduce_p", "tmp_dst_data"], [154, 1, 1, "c.hp_reduce_p", "tmp_src_data"]], "hp_reduce_s": [[154, 1, 1, "c.hp_reduce_s", "core_mask"], [154, 1, 1, "c.hp_reduce_s", "dst_data"], [154, 1, 1, "c.hp_reduce_s", "param"], [154, 1, 1, "c.hp_reduce_s", "src_data"]], "hp_reduceall_p": [[21, 1, 1, "c.hp_reduceall_p", "axis_size"], [21, 1, 1, "c.hp_reduceall_p", "dst_data"], [21, 1, 1, "c.hp_reduceall_p", "inner_size"], [21, 1, 1, "c.hp_reduceall_p", "outer_size"], [21, 1, 1, "c.hp_reduceall_p", "src_data"]], "hp_reduceall_s": [[21, 1, 1, "c.hp_reduceall_s", "axis_size"], [21, 1, 1, "c.hp_reduceall_s", "core_mask"], [21, 1, 1, "c.hp_reduceall_s", "dst_data"], [21, 1, 1, "c.hp_reduceall_s", "inner_size"], [21, 1, 1, "c.hp_reduceall_s", "outer_size"], [21, 1, 1, "c.hp_reduceall_s", "src_data"]], "hp_reducescatter_p": [[155, 1, 1, "c.hp_reducescatter_p", "data_size"], [155, 1, 1, "c.hp_reducescatter_p", "input_data"], [155, 1, 1, "c.hp_reducescatter_p", "output_data"], [155, 1, 1, "c.hp_reducescatter_p", "reduce_type"]], "hp_reducescatter_s": [[155, 1, 1, "c.hp_reducescatter_s", "core_mask"], [155, 1, 1, "c.hp_reducescatter_s", "data_size"], [155, 1, 1, "c.hp_reducescatter_s", "input_data"], [155, 1, 1, "c.hp_reducescatter_s", "output_data"], [155, 1, 1, "c.hp_reducescatter_s", "reduce_type"]], "hp_relu6_p": [[12, 1, 1, "c.hp_relu6_p", "Input0"], [12, 1, 1, "c.hp_relu6_p", "length"], [12, 1, 1, "c.hp_relu6_p", "output"]], "hp_relu6_s": [[12, 1, 1, "c.hp_relu6_s", "Input0"], [12, 1, 1, "c.hp_relu6_s", "core_mask"], [12, 1, 1, "c.hp_relu6_s", "length"], [12, 1, 1, "c.hp_relu6_s", "output"]], "hp_relu_grad_p": [[13, 1, 1, "c.hp_relu_grad_p", "dst"], [13, 1, 1, "c.hp_relu_grad_p", "length"], [13, 1, 1, "c.hp_relu_grad_p", "src0"], [13, 1, 1, "c.hp_relu_grad_p", "src1"]], "hp_relu_grad_s": [[13, 1, 1, "c.hp_relu_grad_s", "core_mask"], [13, 1, 1, "c.hp_relu_grad_s", "dst"], [13, 1, 1, "c.hp_relu_grad_s", "length"], [13, 1, 1, "c.hp_relu_grad_s", "src0"], [13, 1, 1, "c.hp_relu_grad_s", "src1"]], "hp_relu_p": [[12, 1, 1, "c.hp_relu_p", "Input0"], [12, 1, 1, "c.hp_relu_p", "length"], [12, 1, 1, "c.hp_relu_p", "output"]], "hp_relu_s": [[12, 1, 1, "c.hp_relu_s", "Input0"], [12, 1, 1, "c.hp_relu_s", "core_mask"], [12, 1, 1, "c.hp_relu_s", "length"], [12, 1, 1, "c.hp_relu_s", "output"]], "hp_reshape_p": [[156, 1, 1, "c.hp_reshape_p", "input"], [156, 1, 1, "c.hp_reshape_p", "length"], [156, 1, 1, "c.hp_reshape_p", "output"]], "hp_reshape_s": [[156, 1, 1, "c.hp_reshape_s", "core_mask"], [156, 1, 1, "c.hp_reshape_s", "input"], [156, 1, 1, "c.hp_reshape_s", "length"], [156, 1, 1, "c.hp_reshape_s", "output"]], "hp_resize_anycore": [[157, 1, 1, "c.hp_resize_anycore", "core_mask"], [157, 1, 1, "c.hp_resize_anycore", "input"], [157, 1, 1, "c.hp_resize_anycore", "output"], [157, 1, 1, "c.hp_resize_anycore", "param"]], "hp_resizebilineargrad_p": [[158, 1, 1, "c.hp_resizebilineargrad_p", "align_corners"], [158, 1, 1, "c.hp_resizebilineargrad_p", "batch_size"], [158, 1, 1, "c.hp_resizebilineargrad_p", "channel"], [158, 1, 1, "c.hp_resizebilineargrad_p", "format"], [158, 1, 1, "c.hp_resizebilineargrad_p", "height_scale"], [158, 1, 1, "c.hp_resizebilineargrad_p", "in_addr"], [158, 1, 1, "c.hp_resizebilineargrad_p", "in_height"], [158, 1, 1, "c.hp_resizebilineargrad_p", "in_width"], [158, 1, 1, "c.hp_resizebilineargrad_p", "out_addr"], [158, 1, 1, "c.hp_resizebilineargrad_p", "out_height"], [158, 1, 1, "c.hp_resizebilineargrad_p", "out_width"], [158, 1, 1, "c.hp_resizebilineargrad_p", "width_scale"]], "hp_resizebilineargrad_s": [[158, 1, 1, "c.hp_resizebilineargrad_s", "align_corners"], [158, 1, 1, "c.hp_resizebilineargrad_s", "batch_size"], [158, 1, 1, "c.hp_resizebilineargrad_s", "channel"], [158, 1, 1, "c.hp_resizebilineargrad_s", "core_mask"], [158, 1, 1, "c.hp_resizebilineargrad_s", "format"], [158, 1, 1, "c.hp_resizebilineargrad_s", "height_scale"], [158, 1, 1, "c.hp_resizebilineargrad_s", "in_addr"], [158, 1, 1, "c.hp_resizebilineargrad_s", "in_height"], [158, 1, 1, "c.hp_resizebilineargrad_s", "in_width"], [158, 1, 1, "c.hp_resizebilineargrad_s", "out_addr"], [158, 1, 1, "c.hp_resizebilineargrad_s", "out_height"], [158, 1, 1, "c.hp_resizebilineargrad_s", "out_width"], [158, 1, 1, "c.hp_resizebilineargrad_s", "width_scale"]], "hp_resizenearestneighborgrad_p": [[158, 1, 1, "c.hp_resizenearestneighborgrad_p", "align_corners"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "batch_size"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "channel"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "format"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "height_scale"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "in_addr"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "in_height"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "in_width"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "out_addr"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "out_height"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "out_width"], [158, 1, 1, "c.hp_resizenearestneighborgrad_p", "width_scale"]], "hp_resizenearestneighborgrad_s": [[158, 1, 1, "c.hp_resizenearestneighborgrad_s", "align_corners"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "batch_size"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "channel"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "core_mask"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "format"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "height_scale"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "in_addr"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "in_height"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "in_width"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "out_addr"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "out_height"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "out_width"], [158, 1, 1, "c.hp_resizenearestneighborgrad_s", "width_scale"]], "hp_roipooling_p": [[162, 1, 1, "c.hp_roipooling_p", "in_ptr"], [162, 1, 1, "c.hp_roipooling_p", "input_c"], [162, 1, 1, "c.hp_roipooling_p", "input_h"], [162, 1, 1, "c.hp_roipooling_p", "input_n"], [162, 1, 1, "c.hp_roipooling_p", "input_w"], [162, 1, 1, "c.hp_roipooling_p", "max_c"], [162, 1, 1, "c.hp_roipooling_p", "num_rois"], [162, 1, 1, "c.hp_roipooling_p", "out_ptr"], [162, 1, 1, "c.hp_roipooling_p", "pooled_height"], [162, 1, 1, "c.hp_roipooling_p", "pooled_width"], [162, 1, 1, "c.hp_roipooling_p", "roi"], [162, 1, 1, "c.hp_roipooling_p", "scale"]], "hp_roipooling_s": [[162, 1, 1, "c.hp_roipooling_s", "core_mask"], [162, 1, 1, "c.hp_roipooling_s", "in_ptr"], [162, 1, 1, "c.hp_roipooling_s", "input_c"], [162, 1, 1, "c.hp_roipooling_s", "input_h"], [162, 1, 1, "c.hp_roipooling_s", "input_n"], [162, 1, 1, "c.hp_roipooling_s", "input_w"], [162, 1, 1, "c.hp_roipooling_s", "max_c"], [162, 1, 1, "c.hp_roipooling_s", "num_rois"], [162, 1, 1, "c.hp_roipooling_s", "out_ptr"], [162, 1, 1, "c.hp_roipooling_s", "pooled_height"], [162, 1, 1, "c.hp_roipooling_s", "pooled_width"], [162, 1, 1, "c.hp_roipooling_s", "roi"], [162, 1, 1, "c.hp_roipooling_s", "scale"]], "hp_round_p": [[163, 1, 1, "c.hp_round_p", "input"], [163, 1, 1, "c.hp_round_p", "length"], [163, 1, 1, "c.hp_round_p", "output"]], "hp_round_s": [[163, 1, 1, "c.hp_round_s", "core_mask"], [163, 1, 1, "c.hp_round_s", "input"], [163, 1, 1, "c.hp_round_s", "length"], [163, 1, 1, "c.hp_round_s", "output"]], "hp_rsqrt_p": [[164, 1, 1, "c.hp_rsqrt_p", "dst"], [164, 1, 1, "c.hp_rsqrt_p", "length"], [164, 1, 1, "c.hp_rsqrt_p", "src"]], "hp_rsqrt_s": [[164, 1, 1, "c.hp_rsqrt_s", "core_mask"], [164, 1, 1, "c.hp_rsqrt_s", "dst"], [164, 1, 1, "c.hp_rsqrt_s", "length"], [164, 1, 1, "c.hp_rsqrt_s", "src"]], "hp_rsqrtgrad_p": [[165, 1, 1, "c.hp_rsqrtgrad_p", "input1"], [165, 1, 1, "c.hp_rsqrtgrad_p", "input2"], [165, 1, 1, "c.hp_rsqrtgrad_p", "output"], [165, 1, 1, "c.hp_rsqrtgrad_p", "size"]], "hp_rsqrtgrad_s": [[165, 1, 1, "c.hp_rsqrtgrad_s", "core_mask"], [165, 1, 1, "c.hp_rsqrtgrad_s", "input1"], [165, 1, 1, "c.hp_rsqrtgrad_s", "input2"], [165, 1, 1, "c.hp_rsqrtgrad_s", "output"], [165, 1, 1, "c.hp_rsqrtgrad_s", "size"]], "hp_scalefusion_p": [[166, 1, 1, "c.hp_scalefusion_p", "bias"], [166, 1, 1, "c.hp_scalefusion_p", "dst_data"], [166, 1, 1, "c.hp_scalefusion_p", "length"], [166, 1, 1, "c.hp_scalefusion_p", "scale"], [166, 1, 1, "c.hp_scalefusion_p", "src_data"]], "hp_scalefusion_s": [[166, 1, 1, "c.hp_scalefusion_s", "bias"], [166, 1, 1, "c.hp_scalefusion_s", "core_mask"], [166, 1, 1, "c.hp_scalefusion_s", "dst_data"], [166, 1, 1, "c.hp_scalefusion_s", "length"], [166, 1, 1, "c.hp_scalefusion_s", "scale"], [166, 1, 1, "c.hp_scalefusion_s", "src_data"]], "hp_scatter_elements_p": [[167, 1, 1, "c.hp_scatter_elements_p", "core_mask"], [167, 1, 1, "c.hp_scatter_elements_p", "indices"], [167, 1, 1, "c.hp_scatter_elements_p", "input"], [167, 1, 1, "c.hp_scatter_elements_p", "output"], [167, 1, 1, "c.hp_scatter_elements_p", "param"], [167, 1, 1, "c.hp_scatter_elements_p", "updates"]], "hp_scatter_elements_s": [[167, 1, 1, "c.hp_scatter_elements_s", "core_mask"], [167, 1, 1, "c.hp_scatter_elements_s", "indices"], [167, 1, 1, "c.hp_scatter_elements_s", "input"], [167, 1, 1, "c.hp_scatter_elements_s", "output"], [167, 1, 1, "c.hp_scatter_elements_s", "param"], [167, 1, 1, "c.hp_scatter_elements_s", "updates"]], "hp_scatter_nd_p": [[168, 1, 1, "c.hp_scatter_nd_p", "indices"], [168, 1, 1, "c.hp_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.hp_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.hp_scatter_nd_p", "output"], [168, 1, 1, "c.hp_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.hp_scatter_nd_p", "output_shape"], [168, 1, 1, "c.hp_scatter_nd_p", "updates"]], "hp_scatter_nd_s": [[168, 1, 1, "c.hp_scatter_nd_s", "core_mask"], [168, 1, 1, "c.hp_scatter_nd_s", "indices"], [168, 1, 1, "c.hp_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.hp_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.hp_scatter_nd_s", "output"], [168, 1, 1, "c.hp_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.hp_scatter_nd_s", "output_shape"], [168, 1, 1, "c.hp_scatter_nd_s", "updates"]], "hp_scatter_nd_update_p": [[169, 1, 1, "c.hp_scatter_nd_update_p", "indices"], [169, 1, 1, "c.hp_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.hp_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.hp_scatter_nd_update_p", "output"], [169, 1, 1, "c.hp_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.hp_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.hp_scatter_nd_update_p", "updates"]], "hp_scatter_nd_update_s": [[169, 1, 1, "c.hp_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.hp_scatter_nd_update_s", "indices"], [169, 1, 1, "c.hp_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.hp_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.hp_scatter_nd_update_s", "output"], [169, 1, 1, "c.hp_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.hp_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.hp_scatter_nd_update_s", "updates"]], "hp_select_p": [[170, 1, 1, "c.hp_select_p", "condition"], [170, 1, 1, "c.hp_select_p", "index_list1"], [170, 1, 1, "c.hp_select_p", "index_list2"], [170, 1, 1, "c.hp_select_p", "index_list3"], [170, 1, 1, "c.hp_select_p", "input0"], [170, 1, 1, "c.hp_select_p", "input1"], [170, 1, 1, "c.hp_select_p", "is_broadcast"], [170, 1, 1, "c.hp_select_p", "output"], [170, 1, 1, "c.hp_select_p", "output_dims"], [170, 1, 1, "c.hp_select_p", "output_dims_num"]], "hp_select_s": [[170, 1, 1, "c.hp_select_s", "condition"], [170, 1, 1, "c.hp_select_s", "core_mask"], [170, 1, 1, "c.hp_select_s", "index_list1"], [170, 1, 1, "c.hp_select_s", "index_list2"], [170, 1, 1, "c.hp_select_s", "index_list3"], [170, 1, 1, "c.hp_select_s", "input0"], [170, 1, 1, "c.hp_select_s", "input1"], [170, 1, 1, "c.hp_select_s", "is_broadcast"], [170, 1, 1, "c.hp_select_s", "output"], [170, 1, 1, "c.hp_select_s", "output_dims"], [170, 1, 1, "c.hp_select_s", "output_dims_num"]], "hp_sgd_p": [[171, 1, 1, "c.hp_sgd_p", "accumulate"], [171, 1, 1, "c.hp_sgd_p", "dampening"], [171, 1, 1, "c.hp_sgd_p", "gradient"], [171, 1, 1, "c.hp_sgd_p", "learning_rate"], [171, 1, 1, "c.hp_sgd_p", "length"], [171, 1, 1, "c.hp_sgd_p", "moment"], [171, 1, 1, "c.hp_sgd_p", "nesterov"], [171, 1, 1, "c.hp_sgd_p", "weight"], [171, 1, 1, "c.hp_sgd_p", "weight_decay"]], "hp_sgd_s": [[171, 1, 1, "c.hp_sgd_s", "accumulate"], [171, 1, 1, "c.hp_sgd_s", "core_mask"], [171, 1, 1, "c.hp_sgd_s", "dampening"], [171, 1, 1, "c.hp_sgd_s", "end"], [171, 1, 1, "c.hp_sgd_s", "gradient"], [171, 1, 1, "c.hp_sgd_s", "learning_rate"], [171, 1, 1, "c.hp_sgd_s", "moment"], [171, 1, 1, "c.hp_sgd_s", "nesterov"], [171, 1, 1, "c.hp_sgd_s", "start"], [171, 1, 1, "c.hp_sgd_s", "weight"], [171, 1, 1, "c.hp_sgd_s", "weight_decay"]], "hp_sigmoid_p": [[12, 1, 1, "c.hp_sigmoid_p", "Input0"], [12, 1, 1, "c.hp_sigmoid_p", "length"], [12, 1, 1, "c.hp_sigmoid_p", "output"]], "hp_sigmoid_s": [[12, 1, 1, "c.hp_sigmoid_s", "Input0"], [12, 1, 1, "c.hp_sigmoid_s", "core_mask"], [12, 1, 1, "c.hp_sigmoid_s", "length"], [12, 1, 1, "c.hp_sigmoid_s", "output"]], "hp_sigmoidcrossentropywithlogits_p": [[174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_p", "input0"], [174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_p", "input1"], [174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_p", "length"], [174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_p", "output"]], "hp_sigmoidcrossentropywithlogits_s": [[174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_s", "core_mask"], [174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_s", "input0"], [174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_s", "input1"], [174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_s", "length"], [174, 1, 1, "c.hp_sigmoidcrossentropywithlogits_s", "output"]], "hp_sigmoidcrossentropywithlogitsgrad_p": [[173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_p", "Input0"], [173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_p", "Input1"], [173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_p", "length"], [173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_p", "output"]], "hp_sigmoidcrossentropywithlogitsgrad_s": [[173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_s", "Input0"], [173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_s", "Input1"], [173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_s", "core_mask"], [173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_s", "length"], [173, 1, 1, "c.hp_sigmoidcrossentropywithlogitsgrad_s", "output"]], "hp_sin_p": [[175, 1, 1, "c.hp_sin_p", "dst_data"], [175, 1, 1, "c.hp_sin_p", "length"], [175, 1, 1, "c.hp_sin_p", "src_data"]], "hp_sin_s": [[175, 1, 1, "c.hp_sin_s", "core_mask"], [175, 1, 1, "c.hp_sin_s", "dst_data"], [175, 1, 1, "c.hp_sin_s", "length"], [175, 1, 1, "c.hp_sin_s", "src_data"]], "hp_slice_p": [[178, 1, 1, "c.hp_slice_p", "begin"], [178, 1, 1, "c.hp_slice_p", "input"], [178, 1, 1, "c.hp_slice_p", "input_shape"], [178, 1, 1, "c.hp_slice_p", "ndim"], [178, 1, 1, "c.hp_slice_p", "output"], [178, 1, 1, "c.hp_slice_p", "size"]], "hp_slice_s": [[178, 1, 1, "c.hp_slice_s", "begin"], [178, 1, 1, "c.hp_slice_s", "core_mask"], [178, 1, 1, "c.hp_slice_s", "input"], [178, 1, 1, "c.hp_slice_s", "input_shape"], [178, 1, 1, "c.hp_slice_s", "ndim"], [178, 1, 1, "c.hp_slice_s", "output"], [178, 1, 1, "c.hp_slice_s", "size"]], "hp_smoothl1loss_p": [[179, 1, 1, "c.hp_smoothl1loss_p", "beta"], [179, 1, 1, "c.hp_smoothl1loss_p", "length"], [179, 1, 1, "c.hp_smoothl1loss_p", "out"], [179, 1, 1, "c.hp_smoothl1loss_p", "predict"], [179, 1, 1, "c.hp_smoothl1loss_p", "target"]], "hp_smoothl1loss_s": [[179, 1, 1, "c.hp_smoothl1loss_s", "beta"], [179, 1, 1, "c.hp_smoothl1loss_s", "core_mask"], [179, 1, 1, "c.hp_smoothl1loss_s", "length"], [179, 1, 1, "c.hp_smoothl1loss_s", "out"], [179, 1, 1, "c.hp_smoothl1loss_s", "predict"], [179, 1, 1, "c.hp_smoothl1loss_s", "target"]], "hp_smoothl1lossgrad_p": [[180, 1, 1, "c.hp_smoothl1lossgrad_p", "beta"], [180, 1, 1, "c.hp_smoothl1lossgrad_p", "dx1"], [180, 1, 1, "c.hp_smoothl1lossgrad_p", "dy"], [180, 1, 1, "c.hp_smoothl1lossgrad_p", "length"], [180, 1, 1, "c.hp_smoothl1lossgrad_p", "x1"], [180, 1, 1, "c.hp_smoothl1lossgrad_p", "x2"]], "hp_smoothl1lossgrad_s": [[180, 1, 1, "c.hp_smoothl1lossgrad_s", "beta"], [180, 1, 1, "c.hp_smoothl1lossgrad_s", "core_mask"], [180, 1, 1, "c.hp_smoothl1lossgrad_s", "dx1"], [180, 1, 1, "c.hp_smoothl1lossgrad_s", "dy"], [180, 1, 1, "c.hp_smoothl1lossgrad_s", "length"], [180, 1, 1, "c.hp_smoothl1lossgrad_s", "x1"], [180, 1, 1, "c.hp_smoothl1lossgrad_s", "x2"]], "hp_softmax_cross_entropy_with_logits_p": [[182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "batch_size"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "grads"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "labels"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "logits"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "need_grads"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "num_of_classes"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "output"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "probs"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_p", "sum_data"]], "hp_softmax_cross_entropy_with_logits_s": [[182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "batch_size"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "core_mask"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "grads"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "labels"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "logits"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "need_grads"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "num_of_classes"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "output"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "probs"], [182, 1, 1, "c.hp_softmax_cross_entropy_with_logits_s", "sum_data"]], "hp_softmax_p": [[181, 1, 1, "c.hp_softmax_p", "axis"], [181, 1, 1, "c.hp_softmax_p", "axis_size"], [181, 1, 1, "c.hp_softmax_p", "inner_size"], [181, 1, 1, "c.hp_softmax_p", "input_ptr"], [181, 1, 1, "c.hp_softmax_p", "n_dim"], [181, 1, 1, "c.hp_softmax_p", "output_ptr"], [181, 1, 1, "c.hp_softmax_p", "outter_size"], [181, 1, 1, "c.hp_softmax_p", "sum_data"]], "hp_softmax_s": [[181, 1, 1, "c.hp_softmax_s", "axis"], [181, 1, 1, "c.hp_softmax_s", "axis_size"], [181, 1, 1, "c.hp_softmax_s", "core_mask"], [181, 1, 1, "c.hp_softmax_s", "inner_size"], [181, 1, 1, "c.hp_softmax_s", "input_ptr"], [181, 1, 1, "c.hp_softmax_s", "n_dim"], [181, 1, 1, "c.hp_softmax_s", "output_ptr"], [181, 1, 1, "c.hp_softmax_s", "outter_size"], [181, 1, 1, "c.hp_softmax_s", "sum_data"]], "hp_softplus_p": [[12, 1, 1, "c.hp_softplus_p", "Input0"], [12, 1, 1, "c.hp_softplus_p", "length"], [12, 1, 1, "c.hp_softplus_p", "output"]], "hp_softplus_s": [[12, 1, 1, "c.hp_softplus_s", "Input0"], [12, 1, 1, "c.hp_softplus_s", "core_mask"], [12, 1, 1, "c.hp_softplus_s", "length"], [12, 1, 1, "c.hp_softplus_s", "output"]], "hp_softshrink_p": [[12, 1, 1, "c.hp_softshrink_p", "Input0"], [12, 1, 1, "c.hp_softshrink_p", "lambd"], [12, 1, 1, "c.hp_softshrink_p", "length"], [12, 1, 1, "c.hp_softshrink_p", "output"]], "hp_softshrink_s": [[12, 1, 1, "c.hp_softshrink_s", "Input0"], [12, 1, 1, "c.hp_softshrink_s", "core_mask"], [12, 1, 1, "c.hp_softshrink_s", "lambd"], [12, 1, 1, "c.hp_softshrink_s", "length"], [12, 1, 1, "c.hp_softshrink_s", "output"]], "hp_softsignopt_p": [[12, 1, 1, "c.hp_softsignopt_p", "Input0"], [12, 1, 1, "c.hp_softsignopt_p", "length"], [12, 1, 1, "c.hp_softsignopt_p", "output"]], "hp_softsignopt_s": [[12, 1, 1, "c.hp_softsignopt_s", "Input0"], [12, 1, 1, "c.hp_softsignopt_s", "core_mask"], [12, 1, 1, "c.hp_softsignopt_s", "length"], [12, 1, 1, "c.hp_softsignopt_s", "output"]], "hp_spacetobatch_p": [[183, 1, 1, "c.hp_spacetobatch_p", "block_size"], [183, 1, 1, "c.hp_spacetobatch_p", "data_size"], [183, 1, 1, "c.hp_spacetobatch_p", "input"], [183, 1, 1, "c.hp_spacetobatch_p", "input_shape"], [183, 1, 1, "c.hp_spacetobatch_p", "output"], [183, 1, 1, "c.hp_spacetobatch_p", "paddings"]], "hp_spacetobatch_s": [[183, 1, 1, "c.hp_spacetobatch_s", "block_size"], [183, 1, 1, "c.hp_spacetobatch_s", "core_mask"], [183, 1, 1, "c.hp_spacetobatch_s", "data_size"], [183, 1, 1, "c.hp_spacetobatch_s", "input"], [183, 1, 1, "c.hp_spacetobatch_s", "input_shape"], [183, 1, 1, "c.hp_spacetobatch_s", "output"], [183, 1, 1, "c.hp_spacetobatch_s", "paddings"]], "hp_spacetobatchnd_p": [[184, 1, 1, "c.hp_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.hp_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.hp_spacetobatchnd_p", "input"], [184, 1, 1, "c.hp_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.hp_spacetobatchnd_p", "output"], [184, 1, 1, "c.hp_spacetobatchnd_p", "paddings"]], "hp_spacetobatchnd_s": [[184, 1, 1, "c.hp_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.hp_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.hp_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.hp_spacetobatchnd_s", "input"], [184, 1, 1, "c.hp_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.hp_spacetobatchnd_s", "output"], [184, 1, 1, "c.hp_spacetobatchnd_s", "paddings"]], "hp_spacetodepth_p": [[185, 1, 1, "c.hp_spacetodepth_p", "block"], [185, 1, 1, "c.hp_spacetodepth_p", "data_size"], [185, 1, 1, "c.hp_spacetodepth_p", "in_shape"], [185, 1, 1, "c.hp_spacetodepth_p", "input"], [185, 1, 1, "c.hp_spacetodepth_p", "output"]], "hp_spacetodepth_s": [[185, 1, 1, "c.hp_spacetodepth_s", "block"], [185, 1, 1, "c.hp_spacetodepth_s", "core_mask"], [185, 1, 1, "c.hp_spacetodepth_s", "data_size"], [185, 1, 1, "c.hp_spacetodepth_s", "in_shape"], [185, 1, 1, "c.hp_spacetodepth_s", "input"], [185, 1, 1, "c.hp_spacetodepth_s", "output"]], "hp_sparse_softmax_cross_entropy_with_logits_p": [[186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "axis_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "batch_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "inner_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "input"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "is_grad"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "labels"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "losses"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "number_of_classes"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "output"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "outter_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "partial_losses"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_p", "sum_data"]], "hp_sparse_softmax_cross_entropy_with_logits_s": [[186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "axis_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "batch_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "core_mask"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "inner_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "input"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "is_grad"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "labels"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "losses"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "number_of_classes"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "output"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "outter_size"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "partial_losses"], [186, 1, 1, "c.hp_sparse_softmax_cross_entropy_with_logits_s", "sum_data"]], "hp_sparsefillemptyrows_p": [[187, 1, 1, "c.hp_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_p", "values_ptr"]], "hp_sparsefillemptyrows_s": [[187, 1, 1, "c.hp_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.hp_sparsefillemptyrows_s", "values_ptr"]], "hp_sparsesegmentsum_p": [[189, 1, 1, "c.hp_sparsesegmentsum_p", "in_data"], [189, 1, 1, "c.hp_sparsesegmentsum_p", "in_data_shape"], [189, 1, 1, "c.hp_sparsesegmentsum_p", "in_data_shape_size"], [189, 1, 1, "c.hp_sparsesegmentsum_p", "in_indices"], [189, 1, 1, "c.hp_sparsesegmentsum_p", "in_indices_size"], [189, 1, 1, "c.hp_sparsesegmentsum_p", "in_segment_ids"], [189, 1, 1, "c.hp_sparsesegmentsum_p", "out_data"], [189, 1, 1, "c.hp_sparsesegmentsum_p", "out_data_shape"]], "hp_sparsesegmentsum_s": [[189, 1, 1, "c.hp_sparsesegmentsum_s", "core_mask"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "in_data"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "in_data_shape"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "in_data_shape_size"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "in_indices"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "in_indices_size"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "in_segment_ids"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "out_data"], [189, 1, 1, "c.hp_sparsesegmentsum_s", "out_data_shape"]], "hp_sparsetodense_p": [[190, 1, 1, "c.hp_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.hp_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.hp_sparsetodense_p", "output"], [190, 1, 1, "c.hp_sparsetodense_p", "output_strides"], [190, 1, 1, "c.hp_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.hp_sparsetodense_p", "sparse_values"]], "hp_sparsetodense_s": [[190, 1, 1, "c.hp_sparsetodense_s", "core_mask"], [190, 1, 1, "c.hp_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.hp_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.hp_sparsetodense_s", "output"], [190, 1, 1, "c.hp_sparsetodense_s", "output_strides"], [190, 1, 1, "c.hp_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.hp_sparsetodense_s", "sparse_values"]], "hp_splice_p": [[191, 1, 1, "c.hp_splice_p", "context_dim"], [191, 1, 1, "c.hp_splice_p", "dst_col"], [191, 1, 1, "c.hp_splice_p", "dst_data"], [191, 1, 1, "c.hp_splice_p", "dst_row"], [191, 1, 1, "c.hp_splice_p", "forward_indexes"], [191, 1, 1, "c.hp_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.hp_splice_p", "src_col"], [191, 1, 1, "c.hp_splice_p", "src_data"], [191, 1, 1, "c.hp_splice_p", "src_row"]], "hp_splice_s": [[191, 1, 1, "c.hp_splice_s", "context_dim"], [191, 1, 1, "c.hp_splice_s", "core_mask"], [191, 1, 1, "c.hp_splice_s", "dst_col"], [191, 1, 1, "c.hp_splice_s", "dst_data"], [191, 1, 1, "c.hp_splice_s", "dst_row"], [191, 1, 1, "c.hp_splice_s", "forward_indexes"], [191, 1, 1, "c.hp_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.hp_splice_s", "src_col"], [191, 1, 1, "c.hp_splice_s", "src_data"], [191, 1, 1, "c.hp_splice_s", "src_row"]], "hp_split_p": [[192, 1, 1, "c.hp_split_p", "axis"], [192, 1, 1, "c.hp_split_p", "input"], [192, 1, 1, "c.hp_split_p", "input_ndim"], [192, 1, 1, "c.hp_split_p", "input_shape"], [192, 1, 1, "c.hp_split_p", "num_split"], [192, 1, 1, "c.hp_split_p", "outputs"], [192, 1, 1, "c.hp_split_p", "split_sizes"]], "hp_split_s": [[192, 1, 1, "c.hp_split_s", "axis"], [192, 1, 1, "c.hp_split_s", "core_mask"], [192, 1, 1, "c.hp_split_s", "input"], [192, 1, 1, "c.hp_split_s", "input_ndim"], [192, 1, 1, "c.hp_split_s", "input_shape"], [192, 1, 1, "c.hp_split_s", "num_split"], [192, 1, 1, "c.hp_split_s", "outputs"], [192, 1, 1, "c.hp_split_s", "split_sizes"]], "hp_split_with_overlap_p": [[193, 1, 1, "c.hp_split_with_overlap_p", "axis"], [193, 1, 1, "c.hp_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.hp_split_with_overlap_p", "input"], [193, 1, 1, "c.hp_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.hp_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.hp_split_with_overlap_p", "num_split"], [193, 1, 1, "c.hp_split_with_overlap_p", "outputs"], [193, 1, 1, "c.hp_split_with_overlap_p", "start_indices"]], "hp_split_with_overlap_s": [[193, 1, 1, "c.hp_split_with_overlap_s", "axis"], [193, 1, 1, "c.hp_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.hp_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.hp_split_with_overlap_s", "input"], [193, 1, 1, "c.hp_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.hp_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.hp_split_with_overlap_s", "num_split"], [193, 1, 1, "c.hp_split_with_overlap_s", "outputs"], [193, 1, 1, "c.hp_split_with_overlap_s", "start_indices"]], "hp_sqrt_p": [[194, 1, 1, "c.hp_sqrt_p", "dst_data"], [194, 1, 1, "c.hp_sqrt_p", "length"], [194, 1, 1, "c.hp_sqrt_p", "src_data"]], "hp_sqrt_s": [[194, 1, 1, "c.hp_sqrt_s", "core_mask"], [194, 1, 1, "c.hp_sqrt_s", "dst_data"], [194, 1, 1, "c.hp_sqrt_s", "length"], [194, 1, 1, "c.hp_sqrt_s", "src_data"]], "hp_sqrtgrad_p": [[195, 1, 1, "c.hp_sqrtgrad_p", "input1"], [195, 1, 1, "c.hp_sqrtgrad_p", "input2"], [195, 1, 1, "c.hp_sqrtgrad_p", "output"], [195, 1, 1, "c.hp_sqrtgrad_p", "size"]], "hp_sqrtgrad_s": [[195, 1, 1, "c.hp_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.hp_sqrtgrad_s", "input1"], [195, 1, 1, "c.hp_sqrtgrad_s", "input2"], [195, 1, 1, "c.hp_sqrtgrad_s", "output"], [195, 1, 1, "c.hp_sqrtgrad_s", "size"]], "hp_square_p": [[196, 1, 1, "c.hp_square_p", "dst"], [196, 1, 1, "c.hp_square_p", "length"], [196, 1, 1, "c.hp_square_p", "src"]], "hp_square_s": [[196, 1, 1, "c.hp_square_s", "core_mask"], [196, 1, 1, "c.hp_square_s", "dst"], [196, 1, 1, "c.hp_square_s", "length"], [196, 1, 1, "c.hp_square_s", "src"]], "hp_squaredifference_p": [[197, 1, 1, "c.hp_squaredifference_p", "input0"], [197, 1, 1, "c.hp_squaredifference_p", "input1"], [197, 1, 1, "c.hp_squaredifference_p", "length"], [197, 1, 1, "c.hp_squaredifference_p", "output"]], "hp_squaredifference_s": [[197, 1, 1, "c.hp_squaredifference_s", "core_mask"], [197, 1, 1, "c.hp_squaredifference_s", "input0"], [197, 1, 1, "c.hp_squaredifference_s", "input1"], [197, 1, 1, "c.hp_squaredifference_s", "length"], [197, 1, 1, "c.hp_squaredifference_s", "output"]], "hp_stack_p": [[199, 1, 1, "c.hp_stack_p", "axis"], [199, 1, 1, "c.hp_stack_p", "input_ndim"], [199, 1, 1, "c.hp_stack_p", "input_shape"], [199, 1, 1, "c.hp_stack_p", "inputs"], [199, 1, 1, "c.hp_stack_p", "num_inputs"], [199, 1, 1, "c.hp_stack_p", "output"]], "hp_stack_s": [[199, 1, 1, "c.hp_stack_s", "axis"], [199, 1, 1, "c.hp_stack_s", "core_mask"], [199, 1, 1, "c.hp_stack_s", "input_ndim"], [199, 1, 1, "c.hp_stack_s", "input_shape"], [199, 1, 1, "c.hp_stack_s", "inputs"], [199, 1, 1, "c.hp_stack_s", "num_inputs"], [199, 1, 1, "c.hp_stack_s", "output"]], "hp_stridedslicegrad_p": [[201, 1, 1, "c.hp_stridedslicegrad_p", "begins"], [201, 1, 1, "c.hp_stridedslicegrad_p", "dx_shape"], [201, 1, 1, "c.hp_stridedslicegrad_p", "in_shape"], [201, 1, 1, "c.hp_stridedslicegrad_p", "inputs"], [201, 1, 1, "c.hp_stridedslicegrad_p", "output"], [201, 1, 1, "c.hp_stridedslicegrad_p", "strides"]], "hp_stridedslicegrad_s": [[201, 1, 1, "c.hp_stridedslicegrad_s", "begins"], [201, 1, 1, "c.hp_stridedslicegrad_s", "core_mask"], [201, 1, 1, "c.hp_stridedslicegrad_s", "dx_shape"], [201, 1, 1, "c.hp_stridedslicegrad_s", "in_shape"], [201, 1, 1, "c.hp_stridedslicegrad_s", "inputs"], [201, 1, 1, "c.hp_stridedslicegrad_s", "output"], [201, 1, 1, "c.hp_stridedslicegrad_s", "strides"]], "hp_subext_p": [[202, 1, 1, "c.hp_subext_p", "alpha"], [202, 1, 1, "c.hp_subext_p", "input0"], [202, 1, 1, "c.hp_subext_p", "input1"], [202, 1, 1, "c.hp_subext_p", "output"], [202, 1, 1, "c.hp_subext_p", "size"]], "hp_subext_s": [[202, 1, 1, "c.hp_subext_s", "alpha"], [202, 1, 1, "c.hp_subext_s", "core_mask"], [202, 1, 1, "c.hp_subext_s", "input0"], [202, 1, 1, "c.hp_subext_s", "input1"], [202, 1, 1, "c.hp_subext_s", "output"], [202, 1, 1, "c.hp_subext_s", "size"]], "hp_subgrad_p": [[203, 1, 1, "c.hp_subgrad_p", "dx1"], [203, 1, 1, "c.hp_subgrad_p", "dx2"], [203, 1, 1, "c.hp_subgrad_p", "dy"], [203, 1, 1, "c.hp_subgrad_p", "dy_dims"], [203, 1, 1, "c.hp_subgrad_p", "num_dims"], [203, 1, 1, "c.hp_subgrad_p", "x1_dims"], [203, 1, 1, "c.hp_subgrad_p", "x2_dims"]], "hp_subgrad_s": [[203, 1, 1, "c.hp_subgrad_s", "core_mask"], [203, 1, 1, "c.hp_subgrad_s", "dx1"], [203, 1, 1, "c.hp_subgrad_s", "dx2"], [203, 1, 1, "c.hp_subgrad_s", "dy"], [203, 1, 1, "c.hp_subgrad_s", "dy_dims"], [203, 1, 1, "c.hp_subgrad_s", "num_dims"], [203, 1, 1, "c.hp_subgrad_s", "x1_dims"], [203, 1, 1, "c.hp_subgrad_s", "x2_dims"]], "hp_subrelu6_p": [[202, 1, 1, "c.hp_subrelu6_p", "input0"], [202, 1, 1, "c.hp_subrelu6_p", "input1"], [202, 1, 1, "c.hp_subrelu6_p", "output"], [202, 1, 1, "c.hp_subrelu6_p", "size"]], "hp_subrelu6_s": [[202, 1, 1, "c.hp_subrelu6_s", "core_mask"], [202, 1, 1, "c.hp_subrelu6_s", "input0"], [202, 1, 1, "c.hp_subrelu6_s", "input1"], [202, 1, 1, "c.hp_subrelu6_s", "output"], [202, 1, 1, "c.hp_subrelu6_s", "size"]], "hp_subrelu_p": [[202, 1, 1, "c.hp_subrelu_p", "input0"], [202, 1, 1, "c.hp_subrelu_p", "input1"], [202, 1, 1, "c.hp_subrelu_p", "output"], [202, 1, 1, "c.hp_subrelu_p", "size"]], "hp_subrelu_s": [[202, 1, 1, "c.hp_subrelu_s", "core_mask"], [202, 1, 1, "c.hp_subrelu_s", "input0"], [202, 1, 1, "c.hp_subrelu_s", "input1"], [202, 1, 1, "c.hp_subrelu_s", "output"], [202, 1, 1, "c.hp_subrelu_s", "size"]], "hp_swish_p": [[12, 1, 1, "c.hp_swish_p", "Input0"], [12, 1, 1, "c.hp_swish_p", "length"], [12, 1, 1, "c.hp_swish_p", "output"]], "hp_swish_s": [[12, 1, 1, "c.hp_swish_s", "Input0"], [12, 1, 1, "c.hp_swish_s", "core_mask"], [12, 1, 1, "c.hp_swish_s", "length"], [12, 1, 1, "c.hp_swish_s", "output"]], "hp_tanh_p": [[12, 1, 1, "c.hp_tanh_p", "Input0"], [12, 1, 1, "c.hp_tanh_p", "length"], [12, 1, 1, "c.hp_tanh_p", "output"]], "hp_tanh_s": [[12, 1, 1, "c.hp_tanh_s", "Input0"], [12, 1, 1, "c.hp_tanh_s", "core_mask"], [12, 1, 1, "c.hp_tanh_s", "length"], [12, 1, 1, "c.hp_tanh_s", "output"]], "hp_tensor_scatter_add_p": [[206, 1, 1, "c.hp_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "input"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "output"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.hp_tensor_scatter_add_p", "updates"]], "hp_tensor_scatter_add_s": [[206, 1, 1, "c.hp_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "input"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "output"], [206, 1, 1, "c.hp_tensor_scatter_add_s", "updates"]], "hp_tensorarrayread_p": [[208, 1, 1, "c.hp_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.hp_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.hp_tensorarrayread_p", "index"], [208, 1, 1, "c.hp_tensorarrayread_p", "output_data"], [208, 1, 1, "c.hp_tensorarrayread_p", "output_size"]], "hp_tensorarrayread_s": [[208, 1, 1, "c.hp_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.hp_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.hp_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.hp_tensorarrayread_s", "index"], [208, 1, 1, "c.hp_tensorarrayread_s", "output_data"], [208, 1, 1, "c.hp_tensorarrayread_s", "output_size"]], "hp_tensorlistfromtensor_p": [[210, 1, 1, "c.hp_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.hp_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.hp_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.hp_tensorlistfromtensor_p", "output_tensors"]], "hp_tensorlistfromtensor_s": [[210, 1, 1, "c.hp_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.hp_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.hp_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.hp_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.hp_tensorlistfromtensor_s", "output_tensors"]], "hp_tile_p": [[215, 1, 1, "c.hp_tile_p", "input"], [215, 1, 1, "c.hp_tile_p", "input_shape"], [215, 1, 1, "c.hp_tile_p", "output"], [215, 1, 1, "c.hp_tile_p", "stride"], [215, 1, 1, "c.hp_tile_p", "tile_dim"], [215, 1, 1, "c.hp_tile_p", "tile_num"]], "hp_tile_s": [[215, 1, 1, "c.hp_tile_s", "core_mask"], [215, 1, 1, "c.hp_tile_s", "input"], [215, 1, 1, "c.hp_tile_s", "input_shape"], [215, 1, 1, "c.hp_tile_s", "output"], [215, 1, 1, "c.hp_tile_s", "stride"], [215, 1, 1, "c.hp_tile_s", "tile_dim"], [215, 1, 1, "c.hp_tile_s", "tile_num"]], "hp_to_i8_quant_p": [[146, 1, 1, "c.hp_to_i8_quant_p", "input"], [146, 1, 1, "c.hp_to_i8_quant_p", "length"], [146, 1, 1, "c.hp_to_i8_quant_p", "output"], [146, 1, 1, "c.hp_to_i8_quant_p", "scale"], [146, 1, 1, "c.hp_to_i8_quant_p", "zp"]], "hp_to_i8_quant_s": [[146, 1, 1, "c.hp_to_i8_quant_s", "core_mask"], [146, 1, 1, "c.hp_to_i8_quant_s", "input"], [146, 1, 1, "c.hp_to_i8_quant_s", "length"], [146, 1, 1, "c.hp_to_i8_quant_s", "output"], [146, 1, 1, "c.hp_to_i8_quant_s", "scale"], [146, 1, 1, "c.hp_to_i8_quant_s", "zp"]], "hp_topk_fusion_p": [[216, 1, 1, "c.hp_topk_fusion_p", "input"], [216, 1, 1, "c.hp_topk_fusion_p", "output"], [216, 1, 1, "c.hp_topk_fusion_p", "output_index"], [216, 1, 1, "c.hp_topk_fusion_p", "parameter"]], "hp_topk_fusion_s": [[216, 1, 1, "c.hp_topk_fusion_s", "core_mask"], [216, 1, 1, "c.hp_topk_fusion_s", "input"], [216, 1, 1, "c.hp_topk_fusion_s", "output"], [216, 1, 1, "c.hp_topk_fusion_s", "output_index"], [216, 1, 1, "c.hp_topk_fusion_s", "parameter"]], "hp_transpose_p": [[217, 1, 1, "c.hp_transpose_p", "in_data"], [217, 1, 1, "c.hp_transpose_p", "num_axes"], [217, 1, 1, "c.hp_transpose_p", "out_data"], [217, 1, 1, "c.hp_transpose_p", "out_strides"], [217, 1, 1, "c.hp_transpose_p", "output_shape"], [217, 1, 1, "c.hp_transpose_p", "perm"], [217, 1, 1, "c.hp_transpose_p", "strides"]], "hp_transpose_s": [[217, 1, 1, "c.hp_transpose_s", "core_mask"], [217, 1, 1, "c.hp_transpose_s", "in_data"], [217, 1, 1, "c.hp_transpose_s", "num_axes"], [217, 1, 1, "c.hp_transpose_s", "out_data"], [217, 1, 1, "c.hp_transpose_s", "out_strides"], [217, 1, 1, "c.hp_transpose_s", "output_shape"], [217, 1, 1, "c.hp_transpose_s", "perm"], [217, 1, 1, "c.hp_transpose_s", "strides"]], "hp_tril_p": [[218, 1, 1, "c.hp_tril_p", "dst"], [218, 1, 1, "c.hp_tril_p", "height"], [218, 1, 1, "c.hp_tril_p", "k"], [218, 1, 1, "c.hp_tril_p", "out_elems"], [218, 1, 1, "c.hp_tril_p", "src"], [218, 1, 1, "c.hp_tril_p", "width"]], "hp_tril_s": [[218, 1, 1, "c.hp_tril_s", "core_mask"], [218, 1, 1, "c.hp_tril_s", "dst"], [218, 1, 1, "c.hp_tril_s", "height"], [218, 1, 1, "c.hp_tril_s", "k"], [218, 1, 1, "c.hp_tril_s", "out_elems"], [218, 1, 1, "c.hp_tril_s", "src"], [218, 1, 1, "c.hp_tril_s", "width"]], "hp_triu_p": [[219, 1, 1, "c.hp_triu_p", "dst"], [219, 1, 1, "c.hp_triu_p", "height"], [219, 1, 1, "c.hp_triu_p", "k"], [219, 1, 1, "c.hp_triu_p", "out_elems"], [219, 1, 1, "c.hp_triu_p", "src"], [219, 1, 1, "c.hp_triu_p", "width"]], "hp_triu_s": [[219, 1, 1, "c.hp_triu_s", "core_mask"], [219, 1, 1, "c.hp_triu_s", "dst"], [219, 1, 1, "c.hp_triu_s", "height"], [219, 1, 1, "c.hp_triu_s", "k"], [219, 1, 1, "c.hp_triu_s", "out_elems"], [219, 1, 1, "c.hp_triu_s", "src"], [219, 1, 1, "c.hp_triu_s", "width"]], "hp_uniform_real_p": [[220, 1, 1, "c.hp_uniform_real_p", "length"], [220, 1, 1, "c.hp_uniform_real_p", "output"], [220, 1, 1, "c.hp_uniform_real_p", "seed"], [220, 1, 1, "c.hp_uniform_real_p", "seed2"]], "hp_uniform_real_s": [[220, 1, 1, "c.hp_uniform_real_s", "core_mask"], [220, 1, 1, "c.hp_uniform_real_s", "length"], [220, 1, 1, "c.hp_uniform_real_s", "output"], [220, 1, 1, "c.hp_uniform_real_s", "seed"], [220, 1, 1, "c.hp_uniform_real_s", "seed2"]], "hp_where_p": [[225, 1, 1, "c.hp_where_p", "condition"], [225, 1, 1, "c.hp_where_p", "input0"], [225, 1, 1, "c.hp_where_p", "input1"], [225, 1, 1, "c.hp_where_p", "length"], [225, 1, 1, "c.hp_where_p", "output"]], "hp_where_s": [[225, 1, 1, "c.hp_where_s", "condition"], [225, 1, 1, "c.hp_where_s", "core_mask"], [225, 1, 1, "c.hp_where_s", "input0"], [225, 1, 1, "c.hp_where_s", "input1"], [225, 1, 1, "c.hp_where_s", "length"], [225, 1, 1, "c.hp_where_s", "output"]], "hp_zerolike_p": [[226, 1, 1, "c.hp_zerolike_p", "length"], [226, 1, 1, "c.hp_zerolike_p", "output"]], "hp_zerolike_s": [[226, 1, 1, "c.hp_zerolike_s", "core_mask"], [226, 1, 1, "c.hp_zerolike_s", "length"], [226, 1, 1, "c.hp_zerolike_s", "output"]], "i16_Unique_p": [[221, 1, 1, "c.i16_Unique_p", "input"], [221, 1, 1, "c.i16_Unique_p", "input_len"], [221, 1, 1, "c.i16_Unique_p", "output0"], [221, 1, 1, "c.i16_Unique_p", "output0_len"]], "i16_Unique_s": [[221, 1, 1, "c.i16_Unique_s", "core_mask"], [221, 1, 1, "c.i16_Unique_s", "input"], [221, 1, 1, "c.i16_Unique_s", "input_len"], [221, 1, 1, "c.i16_Unique_s", "output0"], [221, 1, 1, "c.i16_Unique_s", "output0_len"]], "i16_abs_p": [[10, 1, 1, "c.i16_abs_p", "dst_data"], [10, 1, 1, "c.i16_abs_p", "length"], [10, 1, 1, "c.i16_abs_p", "src_data"]], "i16_abs_s": [[10, 1, 1, "c.i16_abs_s", "core_mask"], [10, 1, 1, "c.i16_abs_s", "dst_data"], [10, 1, 1, "c.i16_abs_s", "length"], [10, 1, 1, "c.i16_abs_s", "src_data"]], "i16_addext_p": [[17, 1, 1, "c.i16_addext_p", "alpha"], [17, 1, 1, "c.i16_addext_p", "in0"], [17, 1, 1, "c.i16_addext_p", "in1"], [17, 1, 1, "c.i16_addext_p", "out"], [17, 1, 1, "c.i16_addext_p", "size"]], "i16_addext_s": [[17, 1, 1, "c.i16_addext_s", "alpha"], [17, 1, 1, "c.i16_addext_s", "core_mask"], [17, 1, 1, "c.i16_addext_s", "in0"], [17, 1, 1, "c.i16_addext_s", "in1"], [17, 1, 1, "c.i16_addext_s", "out"], [17, 1, 1, "c.i16_addext_s", "size"]], "i16_addn_p": [[19, 1, 1, "c.i16_addn_p", "input0"], [19, 1, 1, "c.i16_addn_p", "input1"], [19, 1, 1, "c.i16_addn_p", "length"], [19, 1, 1, "c.i16_addn_p", "output"]], "i16_addn_s": [[19, 1, 1, "c.i16_addn_s", "core_mask"], [19, 1, 1, "c.i16_addn_s", "input0"], [19, 1, 1, "c.i16_addn_s", "input1"], [19, 1, 1, "c.i16_addn_s", "length"], [19, 1, 1, "c.i16_addn_s", "output"]], "i16_addrelu6_p": [[17, 1, 1, "c.i16_addrelu6_p", "in0"], [17, 1, 1, "c.i16_addrelu6_p", "in1"], [17, 1, 1, "c.i16_addrelu6_p", "out"], [17, 1, 1, "c.i16_addrelu6_p", "size"]], "i16_addrelu6_s": [[17, 1, 1, "c.i16_addrelu6_s", "core_mask"], [17, 1, 1, "c.i16_addrelu6_s", "in0"], [17, 1, 1, "c.i16_addrelu6_s", "in1"], [17, 1, 1, "c.i16_addrelu6_s", "out"], [17, 1, 1, "c.i16_addrelu6_s", "size"]], "i16_addrelu_p": [[17, 1, 1, "c.i16_addrelu_p", "in0"], [17, 1, 1, "c.i16_addrelu_p", "in1"], [17, 1, 1, "c.i16_addrelu_p", "out"], [17, 1, 1, "c.i16_addrelu_p", "size"]], "i16_addrelu_s": [[17, 1, 1, "c.i16_addrelu_s", "core_mask"], [17, 1, 1, "c.i16_addrelu_s", "in0"], [17, 1, 1, "c.i16_addrelu_s", "in1"], [17, 1, 1, "c.i16_addrelu_s", "out"], [17, 1, 1, "c.i16_addrelu_s", "size"]], "i16_allgather_p": [[22, 1, 1, "c.i16_allgather_p", "data_size"], [22, 1, 1, "c.i16_allgather_p", "input"], [22, 1, 1, "c.i16_allgather_p", "input_rank"], [22, 1, 1, "c.i16_allgather_p", "output"], [22, 1, 1, "c.i16_allgather_p", "output_rank"]], "i16_allgather_s": [[22, 1, 1, "c.i16_allgather_s", "core_mask"], [22, 1, 1, "c.i16_allgather_s", "data_size"], [22, 1, 1, "c.i16_allgather_s", "input"], [22, 1, 1, "c.i16_allgather_s", "input_rank"], [22, 1, 1, "c.i16_allgather_s", "output"], [22, 1, 1, "c.i16_allgather_s", "output_rank"]], "i16_and_p": [[112, 1, 1, "c.i16_and_p", "input0"], [112, 1, 1, "c.i16_and_p", "input1"], [112, 1, 1, "c.i16_and_p", "length"], [112, 1, 1, "c.i16_and_p", "output"]], "i16_and_s": [[112, 1, 1, "c.i16_and_s", "core_mask"], [112, 1, 1, "c.i16_and_s", "input0"], [112, 1, 1, "c.i16_and_s", "input1"], [112, 1, 1, "c.i16_and_s", "length"], [112, 1, 1, "c.i16_and_s", "output"]], "i16_assign_p": [[27, 1, 1, "c.i16_assign_p", "dst"], [27, 1, 1, "c.i16_assign_p", "length"], [27, 1, 1, "c.i16_assign_p", "src"]], "i16_assign_s": [[27, 1, 1, "c.i16_assign_s", "core_mask"], [27, 1, 1, "c.i16_assign_s", "dst"], [27, 1, 1, "c.i16_assign_s", "length"], [27, 1, 1, "c.i16_assign_s", "src"]], "i16_assignadd_p": [[28, 1, 1, "c.i16_assignadd_p", "input"], [28, 1, 1, "c.i16_assignadd_p", "length"], [28, 1, 1, "c.i16_assignadd_p", "output"]], "i16_assignadd_s": [[28, 1, 1, "c.i16_assignadd_s", "core_mask"], [28, 1, 1, "c.i16_assignadd_s", "input"], [28, 1, 1, "c.i16_assignadd_s", "length"], [28, 1, 1, "c.i16_assignadd_s", "output"]], "i16_batchtospace_p": [[35, 1, 1, "c.i16_batchtospace_p", "block_size"], [35, 1, 1, "c.i16_batchtospace_p", "crops"], [35, 1, 1, "c.i16_batchtospace_p", "data_size"], [35, 1, 1, "c.i16_batchtospace_p", "input"], [35, 1, 1, "c.i16_batchtospace_p", "input_shape"], [35, 1, 1, "c.i16_batchtospace_p", "output"]], "i16_batchtospace_s": [[35, 1, 1, "c.i16_batchtospace_s", "block_size"], [35, 1, 1, "c.i16_batchtospace_s", "core_mask"], [35, 1, 1, "c.i16_batchtospace_s", "crops"], [35, 1, 1, "c.i16_batchtospace_s", "data_size"], [35, 1, 1, "c.i16_batchtospace_s", "input"], [35, 1, 1, "c.i16_batchtospace_s", "input_shape"], [35, 1, 1, "c.i16_batchtospace_s", "output"]], "i16_batchtospacend_p": [[36, 1, 1, "c.i16_batchtospacend_p", "block_size"], [36, 1, 1, "c.i16_batchtospacend_p", "crops"], [36, 1, 1, "c.i16_batchtospacend_p", "data_size"], [36, 1, 1, "c.i16_batchtospacend_p", "input"], [36, 1, 1, "c.i16_batchtospacend_p", "input_shape"], [36, 1, 1, "c.i16_batchtospacend_p", "output"]], "i16_batchtospacend_s": [[36, 1, 1, "c.i16_batchtospacend_s", "block_size"], [36, 1, 1, "c.i16_batchtospacend_s", "core_mask"], [36, 1, 1, "c.i16_batchtospacend_s", "crops"], [36, 1, 1, "c.i16_batchtospacend_s", "data_size"], [36, 1, 1, "c.i16_batchtospacend_s", "input"], [36, 1, 1, "c.i16_batchtospacend_s", "input_shape"], [36, 1, 1, "c.i16_batchtospacend_s", "output"]], "i16_biasadd_p": [[37, 1, 1, "c.i16_biasadd_p", "data_format"], [37, 1, 1, "c.i16_biasadd_p", "dims"], [37, 1, 1, "c.i16_biasadd_p", "input_bias"], [37, 1, 1, "c.i16_biasadd_p", "input_x"], [37, 1, 1, "c.i16_biasadd_p", "length"], [37, 1, 1, "c.i16_biasadd_p", "output"], [37, 1, 1, "c.i16_biasadd_p", "shape_size"]], "i16_biasadd_s": [[37, 1, 1, "c.i16_biasadd_s", "core_mask"], [37, 1, 1, "c.i16_biasadd_s", "data_format"], [37, 1, 1, "c.i16_biasadd_s", "dims"], [37, 1, 1, "c.i16_biasadd_s", "input_bias"], [37, 1, 1, "c.i16_biasadd_s", "input_x"], [37, 1, 1, "c.i16_biasadd_s", "length"], [37, 1, 1, "c.i16_biasadd_s", "output"], [37, 1, 1, "c.i16_biasadd_s", "shape_size"]], "i16_broadcastto_p": [[41, 1, 1, "c.i16_broadcastto_p", "data_size"], [41, 1, 1, "c.i16_broadcastto_p", "input"], [41, 1, 1, "c.i16_broadcastto_p", "input_shape"], [41, 1, 1, "c.i16_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.i16_broadcastto_p", "output"], [41, 1, 1, "c.i16_broadcastto_p", "output_shape"], [41, 1, 1, "c.i16_broadcastto_p", "output_shape_size"]], "i16_broadcastto_s": [[41, 1, 1, "c.i16_broadcastto_s", "core_mask"], [41, 1, 1, "c.i16_broadcastto_s", "data_size"], [41, 1, 1, "c.i16_broadcastto_s", "input"], [41, 1, 1, "c.i16_broadcastto_s", "input_shape"], [41, 1, 1, "c.i16_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.i16_broadcastto_s", "output"], [41, 1, 1, "c.i16_broadcastto_s", "output_shape"], [41, 1, 1, "c.i16_broadcastto_s", "output_shape_size"]], "i16_clip_p": [[44, 1, 1, "c.i16_clip_p", "dst"], [44, 1, 1, "c.i16_clip_p", "length"], [44, 1, 1, "c.i16_clip_p", "max"], [44, 1, 1, "c.i16_clip_p", "min"], [44, 1, 1, "c.i16_clip_p", "src"]], "i16_clip_s": [[44, 1, 1, "c.i16_clip_s", "core_mask"], [44, 1, 1, "c.i16_clip_s", "dst"], [44, 1, 1, "c.i16_clip_s", "length"], [44, 1, 1, "c.i16_clip_s", "max"], [44, 1, 1, "c.i16_clip_s", "min"], [44, 1, 1, "c.i16_clip_s", "src"]], "i16_concat_p": [[45, 1, 1, "c.i16_concat_p", "axis"], [45, 1, 1, "c.i16_concat_p", "input_ndim"], [45, 1, 1, "c.i16_concat_p", "input_shapes"], [45, 1, 1, "c.i16_concat_p", "inputs"], [45, 1, 1, "c.i16_concat_p", "num_inputs"], [45, 1, 1, "c.i16_concat_p", "output"]], "i16_concat_s": [[45, 1, 1, "c.i16_concat_s", "axis"], [45, 1, 1, "c.i16_concat_s", "core_mask"], [45, 1, 1, "c.i16_concat_s", "input_ndim"], [45, 1, 1, "c.i16_concat_s", "input_shapes"], [45, 1, 1, "c.i16_concat_s", "inputs"], [45, 1, 1, "c.i16_concat_s", "num_inputs"], [45, 1, 1, "c.i16_concat_s", "output"]], "i16_constant_of_shape_p": [[46, 1, 1, "c.i16_constant_of_shape_p", "end"], [46, 1, 1, "c.i16_constant_of_shape_p", "output"], [46, 1, 1, "c.i16_constant_of_shape_p", "start"], [46, 1, 1, "c.i16_constant_of_shape_p", "value"]], "i16_constant_of_shape_s": [[46, 1, 1, "c.i16_constant_of_shape_s", "core_mask"], [46, 1, 1, "c.i16_constant_of_shape_s", "end"], [46, 1, 1, "c.i16_constant_of_shape_s", "output"], [46, 1, 1, "c.i16_constant_of_shape_s", "start"], [46, 1, 1, "c.i16_constant_of_shape_s", "value"]], "i16_cos_p": [[51, 1, 1, "c.i16_cos_p", "dst_data"], [51, 1, 1, "c.i16_cos_p", "length"], [51, 1, 1, "c.i16_cos_p", "src_data"]], "i16_cos_s": [[51, 1, 1, "c.i16_cos_s", "core_mask"], [51, 1, 1, "c.i16_cos_s", "dst_data"], [51, 1, 1, "c.i16_cos_s", "length"], [51, 1, 1, "c.i16_cos_s", "src_data"]], "i16_cumsum_p": [[54, 1, 1, "c.i16_cumsum_p", "axis_dim"], [54, 1, 1, "c.i16_cumsum_p", "exclusive"], [54, 1, 1, "c.i16_cumsum_p", "inner_dim"], [54, 1, 1, "c.i16_cumsum_p", "input"], [54, 1, 1, "c.i16_cumsum_p", "out_dim"], [54, 1, 1, "c.i16_cumsum_p", "output"]], "i16_cumsum_s": [[54, 1, 1, "c.i16_cumsum_s", "axis_dim"], [54, 1, 1, "c.i16_cumsum_s", "core_mask"], [54, 1, 1, "c.i16_cumsum_s", "exclusive"], [54, 1, 1, "c.i16_cumsum_s", "inner_dim"], [54, 1, 1, "c.i16_cumsum_s", "input"], [54, 1, 1, "c.i16_cumsum_s", "out_dim"], [54, 1, 1, "c.i16_cumsum_s", "output"]], "i16_depthtospace_p": [[59, 1, 1, "c.i16_depthtospace_p", "block_size"], [59, 1, 1, "c.i16_depthtospace_p", "data_size"], [59, 1, 1, "c.i16_depthtospace_p", "in_shape"], [59, 1, 1, "c.i16_depthtospace_p", "input"], [59, 1, 1, "c.i16_depthtospace_p", "output"]], "i16_depthtospace_s": [[59, 1, 1, "c.i16_depthtospace_s", "block_size"], [59, 1, 1, "c.i16_depthtospace_s", "core_mask"], [59, 1, 1, "c.i16_depthtospace_s", "data_size"], [59, 1, 1, "c.i16_depthtospace_s", "in_shape"], [59, 1, 1, "c.i16_depthtospace_s", "input"], [59, 1, 1, "c.i16_depthtospace_s", "output"]], "i16_div_fusion_p": [[61, 1, 1, "c.i16_div_fusion_p", "input0"], [61, 1, 1, "c.i16_div_fusion_p", "input1"], [61, 1, 1, "c.i16_div_fusion_p", "length"], [61, 1, 1, "c.i16_div_fusion_p", "output"]], "i16_div_fusion_s": [[61, 1, 1, "c.i16_div_fusion_s", "core_mask"], [61, 1, 1, "c.i16_div_fusion_s", "input0"], [61, 1, 1, "c.i16_div_fusion_s", "input1"], [61, 1, 1, "c.i16_div_fusion_s", "length"], [61, 1, 1, "c.i16_div_fusion_s", "output"]], "i16_eltwise_p": [[67, 1, 1, "c.i16_eltwise_p", "Input0"], [67, 1, 1, "c.i16_eltwise_p", "Input1"], [67, 1, 1, "c.i16_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.i16_eltwise_p", "length"], [67, 1, 1, "c.i16_eltwise_p", "output"]], "i16_eltwise_s": [[67, 1, 1, "c.i16_eltwise_s", "Input0"], [67, 1, 1, "c.i16_eltwise_s", "Input1"], [67, 1, 1, "c.i16_eltwise_s", "core_mask"], [67, 1, 1, "c.i16_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.i16_eltwise_s", "length"], [67, 1, 1, "c.i16_eltwise_s", "output"]], "i16_equal_p": [[70, 1, 1, "c.i16_equal_p", "Input0"], [70, 1, 1, "c.i16_equal_p", "Input1"], [70, 1, 1, "c.i16_equal_p", "length"], [70, 1, 1, "c.i16_equal_p", "output"]], "i16_equal_s": [[70, 1, 1, "c.i16_equal_s", "Input0"], [70, 1, 1, "c.i16_equal_s", "Input1"], [70, 1, 1, "c.i16_equal_s", "core_mask"], [70, 1, 1, "c.i16_equal_s", "length"], [70, 1, 1, "c.i16_equal_s", "output"]], "i16_expfusion_p": [[73, 1, 1, "c.i16_expfusion_p", "dst_data"], [73, 1, 1, "c.i16_expfusion_p", "in_scale"], [73, 1, 1, "c.i16_expfusion_p", "length"], [73, 1, 1, "c.i16_expfusion_p", "out_scale"], [73, 1, 1, "c.i16_expfusion_p", "scale"], [73, 1, 1, "c.i16_expfusion_p", "src_data"]], "i16_expfusion_s": [[73, 1, 1, "c.i16_expfusion_s", "core_mask"], [73, 1, 1, "c.i16_expfusion_s", "dst_data"], [73, 1, 1, "c.i16_expfusion_s", "in_scale"], [73, 1, 1, "c.i16_expfusion_s", "length"], [73, 1, 1, "c.i16_expfusion_s", "out_scale"], [73, 1, 1, "c.i16_expfusion_s", "scale"], [73, 1, 1, "c.i16_expfusion_s", "src_data"]], "i16_extract_features_p": [[55, 1, 1, "c.i16_extract_features_p", "num_strings"], [55, 1, 1, "c.i16_extract_features_p", "output_labels"], [55, 1, 1, "c.i16_extract_features_p", "output_weights"], [55, 1, 1, "c.i16_extract_features_p", "string_lengths"], [55, 1, 1, "c.i16_extract_features_p", "string_pointers"]], "i16_extract_features_s": [[55, 1, 1, "c.i16_extract_features_s", "core_mask"], [55, 1, 1, "c.i16_extract_features_s", "num_strings"], [55, 1, 1, "c.i16_extract_features_s", "output_labels"], [55, 1, 1, "c.i16_extract_features_s", "output_weights"], [55, 1, 1, "c.i16_extract_features_s", "string_lengths"], [55, 1, 1, "c.i16_extract_features_s", "string_pointers"]], "i16_fill_p": [[78, 1, 1, "c.i16_fill_p", "output"], [78, 1, 1, "c.i16_fill_p", "param"], [78, 1, 1, "c.i16_fill_p", "value"]], "i16_fill_s": [[78, 1, 1, "c.i16_fill_s", "core_mask"], [78, 1, 1, "c.i16_fill_s", "output"], [78, 1, 1, "c.i16_fill_s", "param"], [78, 1, 1, "c.i16_fill_s", "value"]], "i16_formattranspose_p": [[85, 1, 1, "c.i16_formattranspose_p", "batch"], [85, 1, 1, "c.i16_formattranspose_p", "channel"], [85, 1, 1, "c.i16_formattranspose_p", "dst_data"], [85, 1, 1, "c.i16_formattranspose_p", "dst_format"], [85, 1, 1, "c.i16_formattranspose_p", "plane"], [85, 1, 1, "c.i16_formattranspose_p", "src_data"], [85, 1, 1, "c.i16_formattranspose_p", "src_format"]], "i16_formattranspose_s": [[85, 1, 1, "c.i16_formattranspose_s", "batch"], [85, 1, 1, "c.i16_formattranspose_s", "channel"], [85, 1, 1, "c.i16_formattranspose_s", "core_mask"], [85, 1, 1, "c.i16_formattranspose_s", "dst_data"], [85, 1, 1, "c.i16_formattranspose_s", "dst_format"], [85, 1, 1, "c.i16_formattranspose_s", "plane"], [85, 1, 1, "c.i16_formattranspose_s", "src_data"], [85, 1, 1, "c.i16_formattranspose_s", "src_format"]], "i16_gather_nd_p": [[89, 1, 1, "c.i16_gather_nd_p", "indices"], [89, 1, 1, "c.i16_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.i16_gather_nd_p", "indices_shape"], [89, 1, 1, "c.i16_gather_nd_p", "input"], [89, 1, 1, "c.i16_gather_nd_p", "input_ndim"], [89, 1, 1, "c.i16_gather_nd_p", "input_shape"], [89, 1, 1, "c.i16_gather_nd_p", "output"]], "i16_gather_nd_s": [[89, 1, 1, "c.i16_gather_nd_s", "core_mask"], [89, 1, 1, "c.i16_gather_nd_s", "indices"], [89, 1, 1, "c.i16_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.i16_gather_nd_s", "indices_shape"], [89, 1, 1, "c.i16_gather_nd_s", "input"], [89, 1, 1, "c.i16_gather_nd_s", "input_ndim"], [89, 1, 1, "c.i16_gather_nd_s", "input_shape"], [89, 1, 1, "c.i16_gather_nd_s", "output"]], "i16_gather_p": [[88, 1, 1, "c.i16_gather_p", "axis"], [88, 1, 1, "c.i16_gather_p", "batch_dims"], [88, 1, 1, "c.i16_gather_p", "indices"], [88, 1, 1, "c.i16_gather_p", "indices_ndim"], [88, 1, 1, "c.i16_gather_p", "indices_shape"], [88, 1, 1, "c.i16_gather_p", "input"], [88, 1, 1, "c.i16_gather_p", "input_ndim"], [88, 1, 1, "c.i16_gather_p", "input_shape"], [88, 1, 1, "c.i16_gather_p", "output"]], "i16_gather_s": [[88, 1, 1, "c.i16_gather_s", "axis"], [88, 1, 1, "c.i16_gather_s", "batch_dims"], [88, 1, 1, "c.i16_gather_s", "core_mask"], [88, 1, 1, "c.i16_gather_s", "indices"], [88, 1, 1, "c.i16_gather_s", "indices_ndim"], [88, 1, 1, "c.i16_gather_s", "indices_shape"], [88, 1, 1, "c.i16_gather_s", "input"], [88, 1, 1, "c.i16_gather_s", "input_ndim"], [88, 1, 1, "c.i16_gather_s", "input_shape"], [88, 1, 1, "c.i16_gather_s", "output"]], "i16_gatherd_p": [[90, 1, 1, "c.i16_gatherd_p", "dim"], [90, 1, 1, "c.i16_gatherd_p", "index"], [90, 1, 1, "c.i16_gatherd_p", "index_shape"], [90, 1, 1, "c.i16_gatherd_p", "input_shape"], [90, 1, 1, "c.i16_gatherd_p", "input_shape_size"], [90, 1, 1, "c.i16_gatherd_p", "input_x"], [90, 1, 1, "c.i16_gatherd_p", "output"]], "i16_gatherd_s": [[90, 1, 1, "c.i16_gatherd_s", "core_mask"], [90, 1, 1, "c.i16_gatherd_s", "dim"], [90, 1, 1, "c.i16_gatherd_s", "index"], [90, 1, 1, "c.i16_gatherd_s", "index_shape"], [90, 1, 1, "c.i16_gatherd_s", "input_shape"], [90, 1, 1, "c.i16_gatherd_s", "input_shape_size"], [90, 1, 1, "c.i16_gatherd_s", "input_x"], [90, 1, 1, "c.i16_gatherd_s", "output"]], "i16_greater_p": [[92, 1, 1, "c.i16_greater_p", "element_num"], [92, 1, 1, "c.i16_greater_p", "in_elements_num0"], [92, 1, 1, "c.i16_greater_p", "input1"], [92, 1, 1, "c.i16_greater_p", "input2"], [92, 1, 1, "c.i16_greater_p", "optimize"], [92, 1, 1, "c.i16_greater_p", "output"]], "i16_greater_s": [[92, 1, 1, "c.i16_greater_s", "core_mask"], [92, 1, 1, "c.i16_greater_s", "element_num"], [92, 1, 1, "c.i16_greater_s", "in_elements_num0"], [92, 1, 1, "c.i16_greater_s", "input1"], [92, 1, 1, "c.i16_greater_s", "input2"], [92, 1, 1, "c.i16_greater_s", "optimize"], [92, 1, 1, "c.i16_greater_s", "output"]], "i16_greaterequal_p": [[93, 1, 1, "c.i16_greaterequal_p", "element_num"], [93, 1, 1, "c.i16_greaterequal_p", "in_elements_num0"], [93, 1, 1, "c.i16_greaterequal_p", "input1"], [93, 1, 1, "c.i16_greaterequal_p", "input2"], [93, 1, 1, "c.i16_greaterequal_p", "optimize"], [93, 1, 1, "c.i16_greaterequal_p", "output"]], "i16_greaterequal_s": [[93, 1, 1, "c.i16_greaterequal_s", "core_mask"], [93, 1, 1, "c.i16_greaterequal_s", "element_num"], [93, 1, 1, "c.i16_greaterequal_s", "in_elements_num0"], [93, 1, 1, "c.i16_greaterequal_s", "input1"], [93, 1, 1, "c.i16_greaterequal_s", "input2"], [93, 1, 1, "c.i16_greaterequal_s", "optimize"], [93, 1, 1, "c.i16_greaterequal_s", "output"]], "i16_invertpermutation_p": [[98, 1, 1, "c.i16_invertpermutation_p", "input"], [98, 1, 1, "c.i16_invertpermutation_p", "num"], [98, 1, 1, "c.i16_invertpermutation_p", "output"]], "i16_invertpermutation_s": [[98, 1, 1, "c.i16_invertpermutation_s", "core_mask"], [98, 1, 1, "c.i16_invertpermutation_s", "input"], [98, 1, 1, "c.i16_invertpermutation_s", "num"], [98, 1, 1, "c.i16_invertpermutation_s", "output"]], "i16_isfinite_p": [[99, 1, 1, "c.i16_isfinite_p", "Input"], [99, 1, 1, "c.i16_isfinite_p", "length"], [99, 1, 1, "c.i16_isfinite_p", "output"]], "i16_isfinite_s": [[99, 1, 1, "c.i16_isfinite_s", "Input"], [99, 1, 1, "c.i16_isfinite_s", "core_mask"], [99, 1, 1, "c.i16_isfinite_s", "length"], [99, 1, 1, "c.i16_isfinite_s", "output"]], "i16_less_p": [[104, 1, 1, "c.i16_less_p", "Input0"], [104, 1, 1, "c.i16_less_p", "Input1"], [104, 1, 1, "c.i16_less_p", "in_elements_num0"], [104, 1, 1, "c.i16_less_p", "length"], [104, 1, 1, "c.i16_less_p", "optimize"], [104, 1, 1, "c.i16_less_p", "output"]], "i16_less_s": [[104, 1, 1, "c.i16_less_s", "Input0"], [104, 1, 1, "c.i16_less_s", "Input1"], [104, 1, 1, "c.i16_less_s", "core_mask"], [104, 1, 1, "c.i16_less_s", "in_elements_num0"], [104, 1, 1, "c.i16_less_s", "length"], [104, 1, 1, "c.i16_less_s", "optimize"], [104, 1, 1, "c.i16_less_s", "output"]], "i16_lessequal_p": [[105, 1, 1, "c.i16_lessequal_p", "Input0"], [105, 1, 1, "c.i16_lessequal_p", "Input1"], [105, 1, 1, "c.i16_lessequal_p", "in_elements_num0"], [105, 1, 1, "c.i16_lessequal_p", "length"], [105, 1, 1, "c.i16_lessequal_p", "optimize"], [105, 1, 1, "c.i16_lessequal_p", "output"]], "i16_lessequal_s": [[105, 1, 1, "c.i16_lessequal_s", "Input0"], [105, 1, 1, "c.i16_lessequal_s", "Input1"], [105, 1, 1, "c.i16_lessequal_s", "core_mask"], [105, 1, 1, "c.i16_lessequal_s", "in_elements_num0"], [105, 1, 1, "c.i16_lessequal_s", "length"], [105, 1, 1, "c.i16_lessequal_s", "optimize"], [105, 1, 1, "c.i16_lessequal_s", "output"]], "i16_log1p_p": [[108, 1, 1, "c.i16_log1p_p", "Input"], [108, 1, 1, "c.i16_log1p_p", "length"], [108, 1, 1, "c.i16_log1p_p", "output"]], "i16_log1p_s": [[108, 1, 1, "c.i16_log1p_s", "Input"], [108, 1, 1, "c.i16_log1p_s", "core_mask"], [108, 1, 1, "c.i16_log1p_s", "length"], [108, 1, 1, "c.i16_log1p_s", "output"]], "i16_log_p": [[107, 1, 1, "c.i16_log_p", "input"], [107, 1, 1, "c.i16_log_p", "length"], [107, 1, 1, "c.i16_log_p", "output"]], "i16_log_s": [[107, 1, 1, "c.i16_log_s", "core_mask"], [107, 1, 1, "c.i16_log_s", "input"], [107, 1, 1, "c.i16_log_s", "length"], [107, 1, 1, "c.i16_log_s", "output"]], "i16_logical_not_p": [[110, 1, 1, "c.i16_logical_not_p", "input"], [110, 1, 1, "c.i16_logical_not_p", "length"], [110, 1, 1, "c.i16_logical_not_p", "output"]], "i16_logical_not_s": [[110, 1, 1, "c.i16_logical_not_s", "core_mask"], [110, 1, 1, "c.i16_logical_not_s", "input"], [110, 1, 1, "c.i16_logical_not_s", "length"], [110, 1, 1, "c.i16_logical_not_s", "output"]], "i16_logical_or_p": [[111, 1, 1, "c.i16_logical_or_p", "input0"], [111, 1, 1, "c.i16_logical_or_p", "input1"], [111, 1, 1, "c.i16_logical_or_p", "length"], [111, 1, 1, "c.i16_logical_or_p", "output"]], "i16_logical_or_s": [[111, 1, 1, "c.i16_logical_or_s", "core_mask"], [111, 1, 1, "c.i16_logical_or_s", "input0"], [111, 1, 1, "c.i16_logical_or_s", "input1"], [111, 1, 1, "c.i16_logical_or_s", "length"], [111, 1, 1, "c.i16_logical_or_s", "output"]], "i16_lsh_projection_p": [[116, 1, 1, "c.i16_lsh_projection_p", "bits_per_hash"], [116, 1, 1, "c.i16_lsh_projection_p", "feature"], [116, 1, 1, "c.i16_lsh_projection_p", "feature_num"], [116, 1, 1, "c.i16_lsh_projection_p", "hash_group_num"], [116, 1, 1, "c.i16_lsh_projection_p", "hash_seed"], [116, 1, 1, "c.i16_lsh_projection_p", "output"], [116, 1, 1, "c.i16_lsh_projection_p", "weight"]], "i16_lsh_projection_s": [[116, 1, 1, "c.i16_lsh_projection_s", "bits_per_hash"], [116, 1, 1, "c.i16_lsh_projection_s", "core_mask"], [116, 1, 1, "c.i16_lsh_projection_s", "feature"], [116, 1, 1, "c.i16_lsh_projection_s", "feature_num"], [116, 1, 1, "c.i16_lsh_projection_s", "hash_group_num"], [116, 1, 1, "c.i16_lsh_projection_s", "hash_seed"], [116, 1, 1, "c.i16_lsh_projection_s", "output"], [116, 1, 1, "c.i16_lsh_projection_s", "weight"]], "i16_matmulfusion_p": [[121, 1, 1, "c.i16_matmulfusion_p", "A"], [121, 1, 1, "c.i16_matmulfusion_p", "B"], [121, 1, 1, "c.i16_matmulfusion_p", "C"], [121, 1, 1, "c.i16_matmulfusion_p", "K"], [121, 1, 1, "c.i16_matmulfusion_p", "M"], [121, 1, 1, "c.i16_matmulfusion_p", "N"], [121, 1, 1, "c.i16_matmulfusion_p", "activation_type"], [121, 1, 1, "c.i16_matmulfusion_p", "bias"]], "i16_matmulfusion_s": [[121, 1, 1, "c.i16_matmulfusion_s", "A"], [121, 1, 1, "c.i16_matmulfusion_s", "B"], [121, 1, 1, "c.i16_matmulfusion_s", "C"], [121, 1, 1, "c.i16_matmulfusion_s", "K"], [121, 1, 1, "c.i16_matmulfusion_s", "M"], [121, 1, 1, "c.i16_matmulfusion_s", "N"], [121, 1, 1, "c.i16_matmulfusion_s", "activation_type"], [121, 1, 1, "c.i16_matmulfusion_s", "bias"], [121, 1, 1, "c.i16_matmulfusion_s", "core_mask"]], "i16_maximum_p": [[122, 1, 1, "c.i16_maximum_p", "input0"], [122, 1, 1, "c.i16_maximum_p", "input1"], [122, 1, 1, "c.i16_maximum_p", "length"], [122, 1, 1, "c.i16_maximum_p", "output"]], "i16_maximum_s": [[122, 1, 1, "c.i16_maximum_s", "core_mask"], [122, 1, 1, "c.i16_maximum_s", "input0"], [122, 1, 1, "c.i16_maximum_s", "input1"], [122, 1, 1, "c.i16_maximum_s", "length"], [122, 1, 1, "c.i16_maximum_s", "output"]], "i16_minimum_p": [[127, 1, 1, "c.i16_minimum_p", "input0"], [127, 1, 1, "c.i16_minimum_p", "input1"], [127, 1, 1, "c.i16_minimum_p", "length"], [127, 1, 1, "c.i16_minimum_p", "output"]], "i16_minimum_s": [[127, 1, 1, "c.i16_minimum_s", "core_mask"], [127, 1, 1, "c.i16_minimum_s", "input0"], [127, 1, 1, "c.i16_minimum_s", "input1"], [127, 1, 1, "c.i16_minimum_s", "length"], [127, 1, 1, "c.i16_minimum_s", "output"]], "i16_mod_p": [[129, 1, 1, "c.i16_mod_p", "input0"], [129, 1, 1, "c.i16_mod_p", "input1"], [129, 1, 1, "c.i16_mod_p", "length"], [129, 1, 1, "c.i16_mod_p", "output"]], "i16_mod_s": [[129, 1, 1, "c.i16_mod_s", "core_mask"], [129, 1, 1, "c.i16_mod_s", "input0"], [129, 1, 1, "c.i16_mod_s", "input1"], [129, 1, 1, "c.i16_mod_s", "length"], [129, 1, 1, "c.i16_mod_s", "output"]], "i16_mul_p": [[130, 1, 1, "c.i16_mul_p", "input0"], [130, 1, 1, "c.i16_mul_p", "input1"], [130, 1, 1, "c.i16_mul_p", "length"], [130, 1, 1, "c.i16_mul_p", "output"]], "i16_mul_s": [[130, 1, 1, "c.i16_mul_s", "core_mask"], [130, 1, 1, "c.i16_mul_s", "input0"], [130, 1, 1, "c.i16_mul_s", "input1"], [130, 1, 1, "c.i16_mul_s", "length"], [130, 1, 1, "c.i16_mul_s", "output"]], "i16_neg_grad_p": [[133, 1, 1, "c.i16_neg_grad_p", "Input"], [133, 1, 1, "c.i16_neg_grad_p", "length"], [133, 1, 1, "c.i16_neg_grad_p", "output"]], "i16_neg_grad_s": [[133, 1, 1, "c.i16_neg_grad_s", "Input"], [133, 1, 1, "c.i16_neg_grad_s", "core_mask"], [133, 1, 1, "c.i16_neg_grad_s", "length"], [133, 1, 1, "c.i16_neg_grad_s", "output"]], "i16_neg_p": [[132, 1, 1, "c.i16_neg_p", "Input"], [132, 1, 1, "c.i16_neg_p", "length"], [132, 1, 1, "c.i16_neg_p", "output"]], "i16_neg_s": [[132, 1, 1, "c.i16_neg_s", "Input"], [132, 1, 1, "c.i16_neg_s", "core_mask"], [132, 1, 1, "c.i16_neg_s", "length"], [132, 1, 1, "c.i16_neg_s", "output"]], "i16_nonzero_p": [[137, 1, 1, "c.i16_nonzero_p", "dim_strides"], [137, 1, 1, "c.i16_nonzero_p", "input"], [137, 1, 1, "c.i16_nonzero_p", "input_rank"], [137, 1, 1, "c.i16_nonzero_p", "length"], [137, 1, 1, "c.i16_nonzero_p", "non_zero_num"], [137, 1, 1, "c.i16_nonzero_p", "output"], [137, 1, 1, "c.i16_nonzero_p", "shape"]], "i16_nonzero_s": [[137, 1, 1, "c.i16_nonzero_s", "core_mask"], [137, 1, 1, "c.i16_nonzero_s", "dim_strides"], [137, 1, 1, "c.i16_nonzero_s", "input"], [137, 1, 1, "c.i16_nonzero_s", "input_rank"], [137, 1, 1, "c.i16_nonzero_s", "length"], [137, 1, 1, "c.i16_nonzero_s", "non_zero_num"], [137, 1, 1, "c.i16_nonzero_s", "output"], [137, 1, 1, "c.i16_nonzero_s", "shape"]], "i16_not_equal_p": [[138, 1, 1, "c.i16_not_equal_p", "Input0"], [138, 1, 1, "c.i16_not_equal_p", "Input1"], [138, 1, 1, "c.i16_not_equal_p", "length"], [138, 1, 1, "c.i16_not_equal_p", "output"]], "i16_not_equal_s": [[138, 1, 1, "c.i16_not_equal_s", "Input0"], [138, 1, 1, "c.i16_not_equal_s", "Input1"], [138, 1, 1, "c.i16_not_equal_s", "core_mask"], [138, 1, 1, "c.i16_not_equal_s", "length"], [138, 1, 1, "c.i16_not_equal_s", "output"]], "i16_onehot_p": [[139, 1, 1, "c.i16_onehot_p", "axis"], [139, 1, 1, "c.i16_onehot_p", "depth"], [139, 1, 1, "c.i16_onehot_p", "indices"], [139, 1, 1, "c.i16_onehot_p", "indices_shape"], [139, 1, 1, "c.i16_onehot_p", "indices_shape_size"], [139, 1, 1, "c.i16_onehot_p", "on_off"], [139, 1, 1, "c.i16_onehot_p", "output"], [139, 1, 1, "c.i16_onehot_p", "support_neg_index"]], "i16_onehot_s": [[139, 1, 1, "c.i16_onehot_s", "axis"], [139, 1, 1, "c.i16_onehot_s", "core_mask"], [139, 1, 1, "c.i16_onehot_s", "depth"], [139, 1, 1, "c.i16_onehot_s", "indices"], [139, 1, 1, "c.i16_onehot_s", "indices_shape"], [139, 1, 1, "c.i16_onehot_s", "indices_shape_size"], [139, 1, 1, "c.i16_onehot_s", "on_off"], [139, 1, 1, "c.i16_onehot_s", "output"], [139, 1, 1, "c.i16_onehot_s", "support_neg_index"]], "i16_ones_like_p": [[140, 1, 1, "c.i16_ones_like_p", "length"], [140, 1, 1, "c.i16_ones_like_p", "output"]], "i16_ones_like_s": [[140, 1, 1, "c.i16_ones_like_s", "core_mask"], [140, 1, 1, "c.i16_ones_like_s", "length"], [140, 1, 1, "c.i16_ones_like_s", "output"]], "i16_padfusion_p": [[141, 1, 1, "c.i16_padfusion_p", "params"]], "i16_padfusion_s": [[141, 1, 1, "c.i16_padfusion_s", "core_mask"], [141, 1, 1, "c.i16_padfusion_s", "params"]], "i16_pow_fusion_p": [[142, 1, 1, "c.i16_pow_fusion_p", "Input"], [142, 1, 1, "c.i16_pow_fusion_p", "broadcast"], [142, 1, 1, "c.i16_pow_fusion_p", "exponent"], [142, 1, 1, "c.i16_pow_fusion_p", "length_in"], [142, 1, 1, "c.i16_pow_fusion_p", "output"], [142, 1, 1, "c.i16_pow_fusion_p", "scale"], [142, 1, 1, "c.i16_pow_fusion_p", "shift"]], "i16_pow_fusion_s": [[142, 1, 1, "c.i16_pow_fusion_s", "Input"], [142, 1, 1, "c.i16_pow_fusion_s", "broadcast"], [142, 1, 1, "c.i16_pow_fusion_s", "core_mask"], [142, 1, 1, "c.i16_pow_fusion_s", "exponent"], [142, 1, 1, "c.i16_pow_fusion_s", "length_in"], [142, 1, 1, "c.i16_pow_fusion_s", "output"], [142, 1, 1, "c.i16_pow_fusion_s", "scale"], [142, 1, 1, "c.i16_pow_fusion_s", "shift"]], "i16_raggedrange_p": [[147, 1, 1, "c.i16_raggedrange_p", "deltas"], [147, 1, 1, "c.i16_raggedrange_p", "limits"], [147, 1, 1, "c.i16_raggedrange_p", "range_count"], [147, 1, 1, "c.i16_raggedrange_p", "splits"], [147, 1, 1, "c.i16_raggedrange_p", "starts"], [147, 1, 1, "c.i16_raggedrange_p", "values"]], "i16_raggedrange_s": [[147, 1, 1, "c.i16_raggedrange_s", "core_mask"], [147, 1, 1, "c.i16_raggedrange_s", "deltas"], [147, 1, 1, "c.i16_raggedrange_s", "limits"], [147, 1, 1, "c.i16_raggedrange_s", "range_count"], [147, 1, 1, "c.i16_raggedrange_s", "splits"], [147, 1, 1, "c.i16_raggedrange_s", "starts"], [147, 1, 1, "c.i16_raggedrange_s", "values"]], "i16_range_p": [[150, 1, 1, "c.i16_range_p", "delta"], [150, 1, 1, "c.i16_range_p", "length"], [150, 1, 1, "c.i16_range_p", "output"], [150, 1, 1, "c.i16_range_p", "start"]], "i16_range_s": [[150, 1, 1, "c.i16_range_s", "core_mask"], [150, 1, 1, "c.i16_range_s", "delta"], [150, 1, 1, "c.i16_range_s", "length"], [150, 1, 1, "c.i16_range_s", "output"], [150, 1, 1, "c.i16_range_s", "start"]], "i16_real_div_p": [[152, 1, 1, "c.i16_real_div_p", "input0"], [152, 1, 1, "c.i16_real_div_p", "input1"], [152, 1, 1, "c.i16_real_div_p", "length"], [152, 1, 1, "c.i16_real_div_p", "output"]], "i16_real_div_s": [[152, 1, 1, "c.i16_real_div_s", "core_mask"], [152, 1, 1, "c.i16_real_div_s", "input0"], [152, 1, 1, "c.i16_real_div_s", "input1"], [152, 1, 1, "c.i16_real_div_s", "length"], [152, 1, 1, "c.i16_real_div_s", "output"]], "i16_reciprocal_p": [[153, 1, 1, "c.i16_reciprocal_p", "Input"], [153, 1, 1, "c.i16_reciprocal_p", "length"], [153, 1, 1, "c.i16_reciprocal_p", "output"]], "i16_reciprocal_s": [[153, 1, 1, "c.i16_reciprocal_s", "Input"], [153, 1, 1, "c.i16_reciprocal_s", "core_mask"], [153, 1, 1, "c.i16_reciprocal_s", "length"], [153, 1, 1, "c.i16_reciprocal_s", "output"]], "i16_reduce_p": [[154, 1, 1, "c.i16_reduce_p", "core_mask"], [154, 1, 1, "c.i16_reduce_p", "dst_data"], [154, 1, 1, "c.i16_reduce_p", "param"], [154, 1, 1, "c.i16_reduce_p", "src_data"], [154, 1, 1, "c.i16_reduce_p", "tmp_dst_data"], [154, 1, 1, "c.i16_reduce_p", "tmp_src_data"]], "i16_reduce_s": [[154, 1, 1, "c.i16_reduce_s", "core_mask"], [154, 1, 1, "c.i16_reduce_s", "dst_data"], [154, 1, 1, "c.i16_reduce_s", "param"], [154, 1, 1, "c.i16_reduce_s", "src_data"]], "i16_reduceall_p": [[21, 1, 1, "c.i16_reduceall_p", "axis_size"], [21, 1, 1, "c.i16_reduceall_p", "dst_data"], [21, 1, 1, "c.i16_reduceall_p", "inner_size"], [21, 1, 1, "c.i16_reduceall_p", "outer_size"], [21, 1, 1, "c.i16_reduceall_p", "src_data"]], "i16_reduceall_s": [[21, 1, 1, "c.i16_reduceall_s", "axis_size"], [21, 1, 1, "c.i16_reduceall_s", "core_mask"], [21, 1, 1, "c.i16_reduceall_s", "dst_data"], [21, 1, 1, "c.i16_reduceall_s", "inner_size"], [21, 1, 1, "c.i16_reduceall_s", "outer_size"], [21, 1, 1, "c.i16_reduceall_s", "src_data"]], "i16_reducescatter_p": [[155, 1, 1, "c.i16_reducescatter_p", "data_size"], [155, 1, 1, "c.i16_reducescatter_p", "input_data"], [155, 1, 1, "c.i16_reducescatter_p", "output_data"], [155, 1, 1, "c.i16_reducescatter_p", "reduce_type"]], "i16_reducescatter_s": [[155, 1, 1, "c.i16_reducescatter_s", "core_mask"], [155, 1, 1, "c.i16_reducescatter_s", "data_size"], [155, 1, 1, "c.i16_reducescatter_s", "input_data"], [155, 1, 1, "c.i16_reducescatter_s", "output_data"], [155, 1, 1, "c.i16_reducescatter_s", "reduce_type"]], "i16_reshape_p": [[156, 1, 1, "c.i16_reshape_p", "input"], [156, 1, 1, "c.i16_reshape_p", "length"], [156, 1, 1, "c.i16_reshape_p", "output"]], "i16_reshape_s": [[156, 1, 1, "c.i16_reshape_s", "core_mask"], [156, 1, 1, "c.i16_reshape_s", "input"], [156, 1, 1, "c.i16_reshape_s", "length"], [156, 1, 1, "c.i16_reshape_s", "output"]], "i16_rsqrt_p": [[164, 1, 1, "c.i16_rsqrt_p", "dst"], [164, 1, 1, "c.i16_rsqrt_p", "length"], [164, 1, 1, "c.i16_rsqrt_p", "src"]], "i16_rsqrt_s": [[164, 1, 1, "c.i16_rsqrt_s", "core_mask"], [164, 1, 1, "c.i16_rsqrt_s", "dst"], [164, 1, 1, "c.i16_rsqrt_s", "length"], [164, 1, 1, "c.i16_rsqrt_s", "src"]], "i16_scalefusion_p": [[166, 1, 1, "c.i16_scalefusion_p", "bias"], [166, 1, 1, "c.i16_scalefusion_p", "dst_data"], [166, 1, 1, "c.i16_scalefusion_p", "length"], [166, 1, 1, "c.i16_scalefusion_p", "scale"], [166, 1, 1, "c.i16_scalefusion_p", "src_data"]], "i16_scalefusion_s": [[166, 1, 1, "c.i16_scalefusion_s", "bias"], [166, 1, 1, "c.i16_scalefusion_s", "core_mask"], [166, 1, 1, "c.i16_scalefusion_s", "dst_data"], [166, 1, 1, "c.i16_scalefusion_s", "length"], [166, 1, 1, "c.i16_scalefusion_s", "scale"], [166, 1, 1, "c.i16_scalefusion_s", "src_data"]], "i16_scatter_elements_p": [[167, 1, 1, "c.i16_scatter_elements_p", "core_mask"], [167, 1, 1, "c.i16_scatter_elements_p", "indices"], [167, 1, 1, "c.i16_scatter_elements_p", "input"], [167, 1, 1, "c.i16_scatter_elements_p", "output"], [167, 1, 1, "c.i16_scatter_elements_p", "param"], [167, 1, 1, "c.i16_scatter_elements_p", "updates"]], "i16_scatter_elements_s": [[167, 1, 1, "c.i16_scatter_elements_s", "core_mask"], [167, 1, 1, "c.i16_scatter_elements_s", "indices"], [167, 1, 1, "c.i16_scatter_elements_s", "input"], [167, 1, 1, "c.i16_scatter_elements_s", "output"], [167, 1, 1, "c.i16_scatter_elements_s", "param"], [167, 1, 1, "c.i16_scatter_elements_s", "updates"]], "i16_scatter_nd_p": [[168, 1, 1, "c.i16_scatter_nd_p", "indices"], [168, 1, 1, "c.i16_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.i16_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.i16_scatter_nd_p", "output"], [168, 1, 1, "c.i16_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.i16_scatter_nd_p", "output_shape"], [168, 1, 1, "c.i16_scatter_nd_p", "updates"]], "i16_scatter_nd_s": [[168, 1, 1, "c.i16_scatter_nd_s", "core_mask"], [168, 1, 1, "c.i16_scatter_nd_s", "indices"], [168, 1, 1, "c.i16_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.i16_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.i16_scatter_nd_s", "output"], [168, 1, 1, "c.i16_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.i16_scatter_nd_s", "output_shape"], [168, 1, 1, "c.i16_scatter_nd_s", "updates"]], "i16_scatter_nd_update_p": [[169, 1, 1, "c.i16_scatter_nd_update_p", "indices"], [169, 1, 1, "c.i16_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.i16_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.i16_scatter_nd_update_p", "output"], [169, 1, 1, "c.i16_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.i16_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.i16_scatter_nd_update_p", "updates"]], "i16_scatter_nd_update_s": [[169, 1, 1, "c.i16_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.i16_scatter_nd_update_s", "indices"], [169, 1, 1, "c.i16_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.i16_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.i16_scatter_nd_update_s", "output"], [169, 1, 1, "c.i16_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.i16_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.i16_scatter_nd_update_s", "updates"]], "i16_select_p": [[170, 1, 1, "c.i16_select_p", "condition"], [170, 1, 1, "c.i16_select_p", "index_list1"], [170, 1, 1, "c.i16_select_p", "index_list2"], [170, 1, 1, "c.i16_select_p", "index_list3"], [170, 1, 1, "c.i16_select_p", "input0"], [170, 1, 1, "c.i16_select_p", "input1"], [170, 1, 1, "c.i16_select_p", "is_broadcast"], [170, 1, 1, "c.i16_select_p", "output"], [170, 1, 1, "c.i16_select_p", "output_dims"], [170, 1, 1, "c.i16_select_p", "output_dims_num"]], "i16_select_s": [[170, 1, 1, "c.i16_select_s", "condition"], [170, 1, 1, "c.i16_select_s", "core_mask"], [170, 1, 1, "c.i16_select_s", "index_list1"], [170, 1, 1, "c.i16_select_s", "index_list2"], [170, 1, 1, "c.i16_select_s", "index_list3"], [170, 1, 1, "c.i16_select_s", "input0"], [170, 1, 1, "c.i16_select_s", "input1"], [170, 1, 1, "c.i16_select_s", "is_broadcast"], [170, 1, 1, "c.i16_select_s", "output"], [170, 1, 1, "c.i16_select_s", "output_dims"], [170, 1, 1, "c.i16_select_s", "output_dims_num"]], "i16_sin_p": [[175, 1, 1, "c.i16_sin_p", "dst_data"], [175, 1, 1, "c.i16_sin_p", "length"], [175, 1, 1, "c.i16_sin_p", "src_data"]], "i16_sin_s": [[175, 1, 1, "c.i16_sin_s", "core_mask"], [175, 1, 1, "c.i16_sin_s", "dst_data"], [175, 1, 1, "c.i16_sin_s", "length"], [175, 1, 1, "c.i16_sin_s", "src_data"]], "i16_slice_p": [[178, 1, 1, "c.i16_slice_p", "begin"], [178, 1, 1, "c.i16_slice_p", "input"], [178, 1, 1, "c.i16_slice_p", "input_shape"], [178, 1, 1, "c.i16_slice_p", "ndim"], [178, 1, 1, "c.i16_slice_p", "output"], [178, 1, 1, "c.i16_slice_p", "size"]], "i16_slice_s": [[178, 1, 1, "c.i16_slice_s", "begin"], [178, 1, 1, "c.i16_slice_s", "core_mask"], [178, 1, 1, "c.i16_slice_s", "input"], [178, 1, 1, "c.i16_slice_s", "input_shape"], [178, 1, 1, "c.i16_slice_s", "ndim"], [178, 1, 1, "c.i16_slice_s", "output"], [178, 1, 1, "c.i16_slice_s", "size"]], "i16_spacetobatch_p": [[183, 1, 1, "c.i16_spacetobatch_p", "block_size"], [183, 1, 1, "c.i16_spacetobatch_p", "data_size"], [183, 1, 1, "c.i16_spacetobatch_p", "input"], [183, 1, 1, "c.i16_spacetobatch_p", "input_shape"], [183, 1, 1, "c.i16_spacetobatch_p", "output"], [183, 1, 1, "c.i16_spacetobatch_p", "paddings"]], "i16_spacetobatch_s": [[183, 1, 1, "c.i16_spacetobatch_s", "block_size"], [183, 1, 1, "c.i16_spacetobatch_s", "core_mask"], [183, 1, 1, "c.i16_spacetobatch_s", "data_size"], [183, 1, 1, "c.i16_spacetobatch_s", "input"], [183, 1, 1, "c.i16_spacetobatch_s", "input_shape"], [183, 1, 1, "c.i16_spacetobatch_s", "output"], [183, 1, 1, "c.i16_spacetobatch_s", "paddings"]], "i16_spacetobatchnd_p": [[184, 1, 1, "c.i16_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.i16_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.i16_spacetobatchnd_p", "input"], [184, 1, 1, "c.i16_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.i16_spacetobatchnd_p", "output"], [184, 1, 1, "c.i16_spacetobatchnd_p", "paddings"]], "i16_spacetobatchnd_s": [[184, 1, 1, "c.i16_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.i16_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.i16_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.i16_spacetobatchnd_s", "input"], [184, 1, 1, "c.i16_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.i16_spacetobatchnd_s", "output"], [184, 1, 1, "c.i16_spacetobatchnd_s", "paddings"]], "i16_spacetodepth_p": [[185, 1, 1, "c.i16_spacetodepth_p", "block"], [185, 1, 1, "c.i16_spacetodepth_p", "data_size"], [185, 1, 1, "c.i16_spacetodepth_p", "in_shape"], [185, 1, 1, "c.i16_spacetodepth_p", "input"], [185, 1, 1, "c.i16_spacetodepth_p", "output"]], "i16_spacetodepth_s": [[185, 1, 1, "c.i16_spacetodepth_s", "block"], [185, 1, 1, "c.i16_spacetodepth_s", "core_mask"], [185, 1, 1, "c.i16_spacetodepth_s", "data_size"], [185, 1, 1, "c.i16_spacetodepth_s", "in_shape"], [185, 1, 1, "c.i16_spacetodepth_s", "input"], [185, 1, 1, "c.i16_spacetodepth_s", "output"]], "i16_sparsefillemptyrows_p": [[187, 1, 1, "c.i16_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_p", "values_ptr"]], "i16_sparsefillemptyrows_s": [[187, 1, 1, "c.i16_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.i16_sparsefillemptyrows_s", "values_ptr"]], "i16_sparsesegmentsum_p": [[189, 1, 1, "c.i16_sparsesegmentsum_p", "in_data"], [189, 1, 1, "c.i16_sparsesegmentsum_p", "in_data_shape"], [189, 1, 1, "c.i16_sparsesegmentsum_p", "in_data_shape_size"], [189, 1, 1, "c.i16_sparsesegmentsum_p", "in_indices"], [189, 1, 1, "c.i16_sparsesegmentsum_p", "in_indices_size"], [189, 1, 1, "c.i16_sparsesegmentsum_p", "in_segment_ids"], [189, 1, 1, "c.i16_sparsesegmentsum_p", "out_data"], [189, 1, 1, "c.i16_sparsesegmentsum_p", "out_data_shape"]], "i16_sparsesegmentsum_s": [[189, 1, 1, "c.i16_sparsesegmentsum_s", "core_mask"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "in_data"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "in_data_shape"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "in_data_shape_size"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "in_indices"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "in_indices_size"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "in_segment_ids"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "out_data"], [189, 1, 1, "c.i16_sparsesegmentsum_s", "out_data_shape"]], "i16_sparsetodense_p": [[190, 1, 1, "c.i16_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.i16_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.i16_sparsetodense_p", "output"], [190, 1, 1, "c.i16_sparsetodense_p", "output_strides"], [190, 1, 1, "c.i16_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.i16_sparsetodense_p", "sparse_values"]], "i16_sparsetodense_s": [[190, 1, 1, "c.i16_sparsetodense_s", "core_mask"], [190, 1, 1, "c.i16_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.i16_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.i16_sparsetodense_s", "output"], [190, 1, 1, "c.i16_sparsetodense_s", "output_strides"], [190, 1, 1, "c.i16_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.i16_sparsetodense_s", "sparse_values"]], "i16_splice_p": [[191, 1, 1, "c.i16_splice_p", "context_dim"], [191, 1, 1, "c.i16_splice_p", "dst_col"], [191, 1, 1, "c.i16_splice_p", "dst_data"], [191, 1, 1, "c.i16_splice_p", "dst_row"], [191, 1, 1, "c.i16_splice_p", "forward_indexes"], [191, 1, 1, "c.i16_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.i16_splice_p", "src_col"], [191, 1, 1, "c.i16_splice_p", "src_data"], [191, 1, 1, "c.i16_splice_p", "src_row"]], "i16_splice_s": [[191, 1, 1, "c.i16_splice_s", "context_dim"], [191, 1, 1, "c.i16_splice_s", "core_mask"], [191, 1, 1, "c.i16_splice_s", "dst_col"], [191, 1, 1, "c.i16_splice_s", "dst_data"], [191, 1, 1, "c.i16_splice_s", "dst_row"], [191, 1, 1, "c.i16_splice_s", "forward_indexes"], [191, 1, 1, "c.i16_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.i16_splice_s", "src_col"], [191, 1, 1, "c.i16_splice_s", "src_data"], [191, 1, 1, "c.i16_splice_s", "src_row"]], "i16_split_p": [[192, 1, 1, "c.i16_split_p", "axis"], [192, 1, 1, "c.i16_split_p", "input"], [192, 1, 1, "c.i16_split_p", "input_ndim"], [192, 1, 1, "c.i16_split_p", "input_shape"], [192, 1, 1, "c.i16_split_p", "num_split"], [192, 1, 1, "c.i16_split_p", "outputs"], [192, 1, 1, "c.i16_split_p", "split_sizes"]], "i16_split_s": [[192, 1, 1, "c.i16_split_s", "axis"], [192, 1, 1, "c.i16_split_s", "core_mask"], [192, 1, 1, "c.i16_split_s", "input"], [192, 1, 1, "c.i16_split_s", "input_ndim"], [192, 1, 1, "c.i16_split_s", "input_shape"], [192, 1, 1, "c.i16_split_s", "num_split"], [192, 1, 1, "c.i16_split_s", "outputs"], [192, 1, 1, "c.i16_split_s", "split_sizes"]], "i16_split_with_overlap_p": [[193, 1, 1, "c.i16_split_with_overlap_p", "axis"], [193, 1, 1, "c.i16_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.i16_split_with_overlap_p", "input"], [193, 1, 1, "c.i16_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.i16_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.i16_split_with_overlap_p", "num_split"], [193, 1, 1, "c.i16_split_with_overlap_p", "outputs"], [193, 1, 1, "c.i16_split_with_overlap_p", "start_indices"]], "i16_split_with_overlap_s": [[193, 1, 1, "c.i16_split_with_overlap_s", "axis"], [193, 1, 1, "c.i16_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.i16_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.i16_split_with_overlap_s", "input"], [193, 1, 1, "c.i16_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.i16_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.i16_split_with_overlap_s", "num_split"], [193, 1, 1, "c.i16_split_with_overlap_s", "outputs"], [193, 1, 1, "c.i16_split_with_overlap_s", "start_indices"]], "i16_sqrt_p": [[194, 1, 1, "c.i16_sqrt_p", "dst_data"], [194, 1, 1, "c.i16_sqrt_p", "length"], [194, 1, 1, "c.i16_sqrt_p", "src_data"]], "i16_sqrt_s": [[194, 1, 1, "c.i16_sqrt_s", "core_mask"], [194, 1, 1, "c.i16_sqrt_s", "dst_data"], [194, 1, 1, "c.i16_sqrt_s", "length"], [194, 1, 1, "c.i16_sqrt_s", "src_data"]], "i16_sqrtgrad_p": [[195, 1, 1, "c.i16_sqrtgrad_p", "input1"], [195, 1, 1, "c.i16_sqrtgrad_p", "input2"], [195, 1, 1, "c.i16_sqrtgrad_p", "output"], [195, 1, 1, "c.i16_sqrtgrad_p", "size"]], "i16_sqrtgrad_s": [[195, 1, 1, "c.i16_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.i16_sqrtgrad_s", "input1"], [195, 1, 1, "c.i16_sqrtgrad_s", "input2"], [195, 1, 1, "c.i16_sqrtgrad_s", "output"], [195, 1, 1, "c.i16_sqrtgrad_s", "size"]], "i16_square_p": [[196, 1, 1, "c.i16_square_p", "dst"], [196, 1, 1, "c.i16_square_p", "length"], [196, 1, 1, "c.i16_square_p", "src"]], "i16_square_s": [[196, 1, 1, "c.i16_square_s", "core_mask"], [196, 1, 1, "c.i16_square_s", "dst"], [196, 1, 1, "c.i16_square_s", "length"], [196, 1, 1, "c.i16_square_s", "src"]], "i16_squaredifference_p": [[197, 1, 1, "c.i16_squaredifference_p", "input0"], [197, 1, 1, "c.i16_squaredifference_p", "input1"], [197, 1, 1, "c.i16_squaredifference_p", "length"], [197, 1, 1, "c.i16_squaredifference_p", "output"]], "i16_squaredifference_s": [[197, 1, 1, "c.i16_squaredifference_s", "core_mask"], [197, 1, 1, "c.i16_squaredifference_s", "input0"], [197, 1, 1, "c.i16_squaredifference_s", "input1"], [197, 1, 1, "c.i16_squaredifference_s", "length"], [197, 1, 1, "c.i16_squaredifference_s", "output"]], "i16_stack_p": [[199, 1, 1, "c.i16_stack_p", "axis"], [199, 1, 1, "c.i16_stack_p", "input_ndim"], [199, 1, 1, "c.i16_stack_p", "input_shape"], [199, 1, 1, "c.i16_stack_p", "inputs"], [199, 1, 1, "c.i16_stack_p", "num_inputs"], [199, 1, 1, "c.i16_stack_p", "output"]], "i16_stack_s": [[199, 1, 1, "c.i16_stack_s", "axis"], [199, 1, 1, "c.i16_stack_s", "core_mask"], [199, 1, 1, "c.i16_stack_s", "input_ndim"], [199, 1, 1, "c.i16_stack_s", "input_shape"], [199, 1, 1, "c.i16_stack_s", "inputs"], [199, 1, 1, "c.i16_stack_s", "num_inputs"], [199, 1, 1, "c.i16_stack_s", "output"]], "i16_subrelu6_p": [[202, 1, 1, "c.i16_subrelu6_p", "input0"], [202, 1, 1, "c.i16_subrelu6_p", "input1"], [202, 1, 1, "c.i16_subrelu6_p", "output"], [202, 1, 1, "c.i16_subrelu6_p", "size"]], "i16_subrelu6_s": [[202, 1, 1, "c.i16_subrelu6_s", "core_mask"], [202, 1, 1, "c.i16_subrelu6_s", "input0"], [202, 1, 1, "c.i16_subrelu6_s", "input1"], [202, 1, 1, "c.i16_subrelu6_s", "output"], [202, 1, 1, "c.i16_subrelu6_s", "size"]], "i16_subrelu_p": [[202, 1, 1, "c.i16_subrelu_p", "input0"], [202, 1, 1, "c.i16_subrelu_p", "input1"], [202, 1, 1, "c.i16_subrelu_p", "output"], [202, 1, 1, "c.i16_subrelu_p", "size"]], "i16_subrelu_s": [[202, 1, 1, "c.i16_subrelu_s", "core_mask"], [202, 1, 1, "c.i16_subrelu_s", "input0"], [202, 1, 1, "c.i16_subrelu_s", "input1"], [202, 1, 1, "c.i16_subrelu_s", "output"], [202, 1, 1, "c.i16_subrelu_s", "size"]], "i16_tensor_scatter_add_p": [[206, 1, 1, "c.i16_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "input"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "output"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.i16_tensor_scatter_add_p", "updates"]], "i16_tensor_scatter_add_s": [[206, 1, 1, "c.i16_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "input"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "output"], [206, 1, 1, "c.i16_tensor_scatter_add_s", "updates"]], "i16_tensorarrayread_p": [[208, 1, 1, "c.i16_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.i16_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.i16_tensorarrayread_p", "index"], [208, 1, 1, "c.i16_tensorarrayread_p", "output_data"], [208, 1, 1, "c.i16_tensorarrayread_p", "output_size"]], "i16_tensorarrayread_s": [[208, 1, 1, "c.i16_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.i16_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.i16_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.i16_tensorarrayread_s", "index"], [208, 1, 1, "c.i16_tensorarrayread_s", "output_data"], [208, 1, 1, "c.i16_tensorarrayread_s", "output_size"]], "i16_tensorlistfromtensor_p": [[210, 1, 1, "c.i16_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.i16_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.i16_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.i16_tensorlistfromtensor_p", "output_tensors"]], "i16_tensorlistfromtensor_s": [[210, 1, 1, "c.i16_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.i16_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.i16_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.i16_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.i16_tensorlistfromtensor_s", "output_tensors"]], "i16_tile_p": [[215, 1, 1, "c.i16_tile_p", "input"], [215, 1, 1, "c.i16_tile_p", "input_shape"], [215, 1, 1, "c.i16_tile_p", "output"], [215, 1, 1, "c.i16_tile_p", "stride"], [215, 1, 1, "c.i16_tile_p", "tile_dim"], [215, 1, 1, "c.i16_tile_p", "tile_num"]], "i16_tile_s": [[215, 1, 1, "c.i16_tile_s", "core_mask"], [215, 1, 1, "c.i16_tile_s", "input"], [215, 1, 1, "c.i16_tile_s", "input_shape"], [215, 1, 1, "c.i16_tile_s", "output"], [215, 1, 1, "c.i16_tile_s", "stride"], [215, 1, 1, "c.i16_tile_s", "tile_dim"], [215, 1, 1, "c.i16_tile_s", "tile_num"]], "i16_topk_fusion_p": [[216, 1, 1, "c.i16_topk_fusion_p", "input"], [216, 1, 1, "c.i16_topk_fusion_p", "output"], [216, 1, 1, "c.i16_topk_fusion_p", "output_index"], [216, 1, 1, "c.i16_topk_fusion_p", "parameter"]], "i16_topk_fusion_s": [[216, 1, 1, "c.i16_topk_fusion_s", "core_mask"], [216, 1, 1, "c.i16_topk_fusion_s", "input"], [216, 1, 1, "c.i16_topk_fusion_s", "output"], [216, 1, 1, "c.i16_topk_fusion_s", "output_index"], [216, 1, 1, "c.i16_topk_fusion_s", "parameter"]], "i16_transpose_p": [[217, 1, 1, "c.i16_transpose_p", "in_data"], [217, 1, 1, "c.i16_transpose_p", "num_axes"], [217, 1, 1, "c.i16_transpose_p", "out_data"], [217, 1, 1, "c.i16_transpose_p", "out_strides"], [217, 1, 1, "c.i16_transpose_p", "output_shape"], [217, 1, 1, "c.i16_transpose_p", "perm"], [217, 1, 1, "c.i16_transpose_p", "strides"]], "i16_transpose_s": [[217, 1, 1, "c.i16_transpose_s", "core_mask"], [217, 1, 1, "c.i16_transpose_s", "in_data"], [217, 1, 1, "c.i16_transpose_s", "num_axes"], [217, 1, 1, "c.i16_transpose_s", "out_data"], [217, 1, 1, "c.i16_transpose_s", "out_strides"], [217, 1, 1, "c.i16_transpose_s", "output_shape"], [217, 1, 1, "c.i16_transpose_s", "perm"], [217, 1, 1, "c.i16_transpose_s", "strides"]], "i16_tril_p": [[218, 1, 1, "c.i16_tril_p", "dst"], [218, 1, 1, "c.i16_tril_p", "height"], [218, 1, 1, "c.i16_tril_p", "k"], [218, 1, 1, "c.i16_tril_p", "out_elems"], [218, 1, 1, "c.i16_tril_p", "src"], [218, 1, 1, "c.i16_tril_p", "width"]], "i16_tril_s": [[218, 1, 1, "c.i16_tril_s", "core_mask"], [218, 1, 1, "c.i16_tril_s", "dst"], [218, 1, 1, "c.i16_tril_s", "height"], [218, 1, 1, "c.i16_tril_s", "k"], [218, 1, 1, "c.i16_tril_s", "out_elems"], [218, 1, 1, "c.i16_tril_s", "src"], [218, 1, 1, "c.i16_tril_s", "width"]], "i16_triu_p": [[219, 1, 1, "c.i16_triu_p", "dst"], [219, 1, 1, "c.i16_triu_p", "height"], [219, 1, 1, "c.i16_triu_p", "k"], [219, 1, 1, "c.i16_triu_p", "out_elems"], [219, 1, 1, "c.i16_triu_p", "src"], [219, 1, 1, "c.i16_triu_p", "width"]], "i16_triu_s": [[219, 1, 1, "c.i16_triu_s", "core_mask"], [219, 1, 1, "c.i16_triu_s", "dst"], [219, 1, 1, "c.i16_triu_s", "height"], [219, 1, 1, "c.i16_triu_s", "k"], [219, 1, 1, "c.i16_triu_s", "out_elems"], [219, 1, 1, "c.i16_triu_s", "src"], [219, 1, 1, "c.i16_triu_s", "width"]], "i16_unsorted_segment_sum_p": [[222, 1, 1, "c.i16_unsorted_segment_sum_p", "dim0"], [222, 1, 1, "c.i16_unsorted_segment_sum_p", "dim1"], [222, 1, 1, "c.i16_unsorted_segment_sum_p", "id_max"], [222, 1, 1, "c.i16_unsorted_segment_sum_p", "index"], [222, 1, 1, "c.i16_unsorted_segment_sum_p", "input"], [222, 1, 1, "c.i16_unsorted_segment_sum_p", "output"]], "i16_unsorted_segment_sum_s": [[222, 1, 1, "c.i16_unsorted_segment_sum_s", "core_mask"], [222, 1, 1, "c.i16_unsorted_segment_sum_s", "dim0"], [222, 1, 1, "c.i16_unsorted_segment_sum_s", "dim1"], [222, 1, 1, "c.i16_unsorted_segment_sum_s", "id_max"], [222, 1, 1, "c.i16_unsorted_segment_sum_s", "index"], [222, 1, 1, "c.i16_unsorted_segment_sum_s", "input"], [222, 1, 1, "c.i16_unsorted_segment_sum_s", "output"]], "i16_where_p": [[225, 1, 1, "c.i16_where_p", "condition"], [225, 1, 1, "c.i16_where_p", "input0"], [225, 1, 1, "c.i16_where_p", "input1"], [225, 1, 1, "c.i16_where_p", "length"], [225, 1, 1, "c.i16_where_p", "output"]], "i16_where_s": [[225, 1, 1, "c.i16_where_s", "condition"], [225, 1, 1, "c.i16_where_s", "core_mask"], [225, 1, 1, "c.i16_where_s", "input0"], [225, 1, 1, "c.i16_where_s", "input1"], [225, 1, 1, "c.i16_where_s", "length"], [225, 1, 1, "c.i16_where_s", "output"]], "i16_zerolike_p": [[226, 1, 1, "c.i16_zerolike_p", "length"], [226, 1, 1, "c.i16_zerolike_p", "output"]], "i16_zerolike_s": [[226, 1, 1, "c.i16_zerolike_s", "core_mask"], [226, 1, 1, "c.i16_zerolike_s", "length"], [226, 1, 1, "c.i16_zerolike_s", "output"]], "i32_Unique_p": [[221, 1, 1, "c.i32_Unique_p", "input"], [221, 1, 1, "c.i32_Unique_p", "input_len"], [221, 1, 1, "c.i32_Unique_p", "output0"], [221, 1, 1, "c.i32_Unique_p", "output0_len"]], "i32_Unique_s": [[221, 1, 1, "c.i32_Unique_s", "core_mask"], [221, 1, 1, "c.i32_Unique_s", "input"], [221, 1, 1, "c.i32_Unique_s", "input_len"], [221, 1, 1, "c.i32_Unique_s", "output0"], [221, 1, 1, "c.i32_Unique_s", "output0_len"]], "i32_abs_p": [[10, 1, 1, "c.i32_abs_p", "dst_data"], [10, 1, 1, "c.i32_abs_p", "length"], [10, 1, 1, "c.i32_abs_p", "src_data"]], "i32_abs_s": [[10, 1, 1, "c.i32_abs_s", "core_mask"], [10, 1, 1, "c.i32_abs_s", "dst_data"], [10, 1, 1, "c.i32_abs_s", "length"], [10, 1, 1, "c.i32_abs_s", "src_data"]], "i32_addext_p": [[17, 1, 1, "c.i32_addext_p", "alpha"], [17, 1, 1, "c.i32_addext_p", "in0"], [17, 1, 1, "c.i32_addext_p", "in1"], [17, 1, 1, "c.i32_addext_p", "out"], [17, 1, 1, "c.i32_addext_p", "size"]], "i32_addext_s": [[17, 1, 1, "c.i32_addext_s", "alpha"], [17, 1, 1, "c.i32_addext_s", "core_mask"], [17, 1, 1, "c.i32_addext_s", "in0"], [17, 1, 1, "c.i32_addext_s", "in1"], [17, 1, 1, "c.i32_addext_s", "out"], [17, 1, 1, "c.i32_addext_s", "size"]], "i32_addn_p": [[19, 1, 1, "c.i32_addn_p", "input0"], [19, 1, 1, "c.i32_addn_p", "input1"], [19, 1, 1, "c.i32_addn_p", "length"], [19, 1, 1, "c.i32_addn_p", "output"]], "i32_addn_s": [[19, 1, 1, "c.i32_addn_s", "core_mask"], [19, 1, 1, "c.i32_addn_s", "input0"], [19, 1, 1, "c.i32_addn_s", "input1"], [19, 1, 1, "c.i32_addn_s", "length"], [19, 1, 1, "c.i32_addn_s", "output"]], "i32_addrelu6_p": [[17, 1, 1, "c.i32_addrelu6_p", "in0"], [17, 1, 1, "c.i32_addrelu6_p", "in1"], [17, 1, 1, "c.i32_addrelu6_p", "out"], [17, 1, 1, "c.i32_addrelu6_p", "size"]], "i32_addrelu6_s": [[17, 1, 1, "c.i32_addrelu6_s", "core_mask"], [17, 1, 1, "c.i32_addrelu6_s", "in0"], [17, 1, 1, "c.i32_addrelu6_s", "in1"], [17, 1, 1, "c.i32_addrelu6_s", "out"], [17, 1, 1, "c.i32_addrelu6_s", "size"]], "i32_addrelu_p": [[17, 1, 1, "c.i32_addrelu_p", "in0"], [17, 1, 1, "c.i32_addrelu_p", "in1"], [17, 1, 1, "c.i32_addrelu_p", "out"], [17, 1, 1, "c.i32_addrelu_p", "size"]], "i32_addrelu_s": [[17, 1, 1, "c.i32_addrelu_s", "core_mask"], [17, 1, 1, "c.i32_addrelu_s", "in0"], [17, 1, 1, "c.i32_addrelu_s", "in1"], [17, 1, 1, "c.i32_addrelu_s", "out"], [17, 1, 1, "c.i32_addrelu_s", "size"]], "i32_allgather_p": [[22, 1, 1, "c.i32_allgather_p", "data_size"], [22, 1, 1, "c.i32_allgather_p", "input"], [22, 1, 1, "c.i32_allgather_p", "input_rank"], [22, 1, 1, "c.i32_allgather_p", "output"], [22, 1, 1, "c.i32_allgather_p", "output_rank"]], "i32_allgather_s": [[22, 1, 1, "c.i32_allgather_s", "core_mask"], [22, 1, 1, "c.i32_allgather_s", "data_size"], [22, 1, 1, "c.i32_allgather_s", "input"], [22, 1, 1, "c.i32_allgather_s", "input_rank"], [22, 1, 1, "c.i32_allgather_s", "output"], [22, 1, 1, "c.i32_allgather_s", "output_rank"]], "i32_and_p": [[112, 1, 1, "c.i32_and_p", "input0"], [112, 1, 1, "c.i32_and_p", "input1"], [112, 1, 1, "c.i32_and_p", "length"], [112, 1, 1, "c.i32_and_p", "output"]], "i32_and_s": [[112, 1, 1, "c.i32_and_s", "core_mask"], [112, 1, 1, "c.i32_and_s", "input0"], [112, 1, 1, "c.i32_and_s", "input1"], [112, 1, 1, "c.i32_and_s", "length"], [112, 1, 1, "c.i32_and_s", "output"]], "i32_assign_p": [[27, 1, 1, "c.i32_assign_p", "dst"], [27, 1, 1, "c.i32_assign_p", "length"], [27, 1, 1, "c.i32_assign_p", "src"]], "i32_assign_s": [[27, 1, 1, "c.i32_assign_s", "core_mask"], [27, 1, 1, "c.i32_assign_s", "dst"], [27, 1, 1, "c.i32_assign_s", "length"], [27, 1, 1, "c.i32_assign_s", "src"]], "i32_assignadd_p": [[28, 1, 1, "c.i32_assignadd_p", "input"], [28, 1, 1, "c.i32_assignadd_p", "length"], [28, 1, 1, "c.i32_assignadd_p", "output"]], "i32_assignadd_s": [[28, 1, 1, "c.i32_assignadd_s", "core_mask"], [28, 1, 1, "c.i32_assignadd_s", "input"], [28, 1, 1, "c.i32_assignadd_s", "length"], [28, 1, 1, "c.i32_assignadd_s", "output"]], "i32_batchtospace_p": [[35, 1, 1, "c.i32_batchtospace_p", "block_size"], [35, 1, 1, "c.i32_batchtospace_p", "crops"], [35, 1, 1, "c.i32_batchtospace_p", "data_size"], [35, 1, 1, "c.i32_batchtospace_p", "input"], [35, 1, 1, "c.i32_batchtospace_p", "input_shape"], [35, 1, 1, "c.i32_batchtospace_p", "output"]], "i32_batchtospace_s": [[35, 1, 1, "c.i32_batchtospace_s", "block_size"], [35, 1, 1, "c.i32_batchtospace_s", "core_mask"], [35, 1, 1, "c.i32_batchtospace_s", "crops"], [35, 1, 1, "c.i32_batchtospace_s", "data_size"], [35, 1, 1, "c.i32_batchtospace_s", "input"], [35, 1, 1, "c.i32_batchtospace_s", "input_shape"], [35, 1, 1, "c.i32_batchtospace_s", "output"]], "i32_batchtospacend_p": [[36, 1, 1, "c.i32_batchtospacend_p", "block_size"], [36, 1, 1, "c.i32_batchtospacend_p", "crops"], [36, 1, 1, "c.i32_batchtospacend_p", "data_size"], [36, 1, 1, "c.i32_batchtospacend_p", "input"], [36, 1, 1, "c.i32_batchtospacend_p", "input_shape"], [36, 1, 1, "c.i32_batchtospacend_p", "output"]], "i32_batchtospacend_s": [[36, 1, 1, "c.i32_batchtospacend_s", "block_size"], [36, 1, 1, "c.i32_batchtospacend_s", "core_mask"], [36, 1, 1, "c.i32_batchtospacend_s", "crops"], [36, 1, 1, "c.i32_batchtospacend_s", "data_size"], [36, 1, 1, "c.i32_batchtospacend_s", "input"], [36, 1, 1, "c.i32_batchtospacend_s", "input_shape"], [36, 1, 1, "c.i32_batchtospacend_s", "output"]], "i32_biasadd_p": [[37, 1, 1, "c.i32_biasadd_p", "data_format"], [37, 1, 1, "c.i32_biasadd_p", "dims"], [37, 1, 1, "c.i32_biasadd_p", "input_bias"], [37, 1, 1, "c.i32_biasadd_p", "input_x"], [37, 1, 1, "c.i32_biasadd_p", "length"], [37, 1, 1, "c.i32_biasadd_p", "output"], [37, 1, 1, "c.i32_biasadd_p", "shape_size"]], "i32_biasadd_s": [[37, 1, 1, "c.i32_biasadd_s", "core_mask"], [37, 1, 1, "c.i32_biasadd_s", "data_format"], [37, 1, 1, "c.i32_biasadd_s", "dims"], [37, 1, 1, "c.i32_biasadd_s", "input_bias"], [37, 1, 1, "c.i32_biasadd_s", "input_x"], [37, 1, 1, "c.i32_biasadd_s", "length"], [37, 1, 1, "c.i32_biasadd_s", "output"], [37, 1, 1, "c.i32_biasadd_s", "shape_size"]], "i32_broadcastto_p": [[41, 1, 1, "c.i32_broadcastto_p", "data_size"], [41, 1, 1, "c.i32_broadcastto_p", "input"], [41, 1, 1, "c.i32_broadcastto_p", "input_shape"], [41, 1, 1, "c.i32_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.i32_broadcastto_p", "output"], [41, 1, 1, "c.i32_broadcastto_p", "output_shape"], [41, 1, 1, "c.i32_broadcastto_p", "output_shape_size"]], "i32_broadcastto_s": [[41, 1, 1, "c.i32_broadcastto_s", "core_mask"], [41, 1, 1, "c.i32_broadcastto_s", "data_size"], [41, 1, 1, "c.i32_broadcastto_s", "input"], [41, 1, 1, "c.i32_broadcastto_s", "input_shape"], [41, 1, 1, "c.i32_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.i32_broadcastto_s", "output"], [41, 1, 1, "c.i32_broadcastto_s", "output_shape"], [41, 1, 1, "c.i32_broadcastto_s", "output_shape_size"]], "i32_clip_p": [[44, 1, 1, "c.i32_clip_p", "dst"], [44, 1, 1, "c.i32_clip_p", "length"], [44, 1, 1, "c.i32_clip_p", "max"], [44, 1, 1, "c.i32_clip_p", "min"], [44, 1, 1, "c.i32_clip_p", "src"]], "i32_clip_s": [[44, 1, 1, "c.i32_clip_s", "core_mask"], [44, 1, 1, "c.i32_clip_s", "dst"], [44, 1, 1, "c.i32_clip_s", "length"], [44, 1, 1, "c.i32_clip_s", "max"], [44, 1, 1, "c.i32_clip_s", "min"], [44, 1, 1, "c.i32_clip_s", "src"]], "i32_concat_p": [[45, 1, 1, "c.i32_concat_p", "axis"], [45, 1, 1, "c.i32_concat_p", "input_ndim"], [45, 1, 1, "c.i32_concat_p", "input_shapes"], [45, 1, 1, "c.i32_concat_p", "inputs"], [45, 1, 1, "c.i32_concat_p", "num_inputs"], [45, 1, 1, "c.i32_concat_p", "output"]], "i32_concat_s": [[45, 1, 1, "c.i32_concat_s", "axis"], [45, 1, 1, "c.i32_concat_s", "core_mask"], [45, 1, 1, "c.i32_concat_s", "input_ndim"], [45, 1, 1, "c.i32_concat_s", "input_shapes"], [45, 1, 1, "c.i32_concat_s", "inputs"], [45, 1, 1, "c.i32_concat_s", "num_inputs"], [45, 1, 1, "c.i32_concat_s", "output"]], "i32_constant_of_shape_p": [[46, 1, 1, "c.i32_constant_of_shape_p", "end"], [46, 1, 1, "c.i32_constant_of_shape_p", "output"], [46, 1, 1, "c.i32_constant_of_shape_p", "start"], [46, 1, 1, "c.i32_constant_of_shape_p", "value"]], "i32_constant_of_shape_s": [[46, 1, 1, "c.i32_constant_of_shape_s", "core_mask"], [46, 1, 1, "c.i32_constant_of_shape_s", "end"], [46, 1, 1, "c.i32_constant_of_shape_s", "output"], [46, 1, 1, "c.i32_constant_of_shape_s", "start"], [46, 1, 1, "c.i32_constant_of_shape_s", "value"]], "i32_cos_p": [[51, 1, 1, "c.i32_cos_p", "dst_data"], [51, 1, 1, "c.i32_cos_p", "length"], [51, 1, 1, "c.i32_cos_p", "src_data"]], "i32_cos_s": [[51, 1, 1, "c.i32_cos_s", "core_mask"], [51, 1, 1, "c.i32_cos_s", "dst_data"], [51, 1, 1, "c.i32_cos_s", "length"], [51, 1, 1, "c.i32_cos_s", "src_data"]], "i32_cumsum_p": [[54, 1, 1, "c.i32_cumsum_p", "axis_dim"], [54, 1, 1, "c.i32_cumsum_p", "exclusive"], [54, 1, 1, "c.i32_cumsum_p", "inner_dim"], [54, 1, 1, "c.i32_cumsum_p", "input"], [54, 1, 1, "c.i32_cumsum_p", "out_dim"], [54, 1, 1, "c.i32_cumsum_p", "output"]], "i32_cumsum_s": [[54, 1, 1, "c.i32_cumsum_s", "axis_dim"], [54, 1, 1, "c.i32_cumsum_s", "core_mask"], [54, 1, 1, "c.i32_cumsum_s", "exclusive"], [54, 1, 1, "c.i32_cumsum_s", "inner_dim"], [54, 1, 1, "c.i32_cumsum_s", "input"], [54, 1, 1, "c.i32_cumsum_s", "out_dim"], [54, 1, 1, "c.i32_cumsum_s", "output"]], "i32_depthtospace_p": [[59, 1, 1, "c.i32_depthtospace_p", "block_size"], [59, 1, 1, "c.i32_depthtospace_p", "data_size"], [59, 1, 1, "c.i32_depthtospace_p", "in_shape"], [59, 1, 1, "c.i32_depthtospace_p", "input"], [59, 1, 1, "c.i32_depthtospace_p", "output"]], "i32_depthtospace_s": [[59, 1, 1, "c.i32_depthtospace_s", "block_size"], [59, 1, 1, "c.i32_depthtospace_s", "core_mask"], [59, 1, 1, "c.i32_depthtospace_s", "data_size"], [59, 1, 1, "c.i32_depthtospace_s", "in_shape"], [59, 1, 1, "c.i32_depthtospace_s", "input"], [59, 1, 1, "c.i32_depthtospace_s", "output"]], "i32_div_fusion_p": [[61, 1, 1, "c.i32_div_fusion_p", "input0"], [61, 1, 1, "c.i32_div_fusion_p", "input1"], [61, 1, 1, "c.i32_div_fusion_p", "length"], [61, 1, 1, "c.i32_div_fusion_p", "output"]], "i32_div_fusion_s": [[61, 1, 1, "c.i32_div_fusion_s", "core_mask"], [61, 1, 1, "c.i32_div_fusion_s", "input0"], [61, 1, 1, "c.i32_div_fusion_s", "input1"], [61, 1, 1, "c.i32_div_fusion_s", "length"], [61, 1, 1, "c.i32_div_fusion_s", "output"]], "i32_eltwise_p": [[67, 1, 1, "c.i32_eltwise_p", "Input0"], [67, 1, 1, "c.i32_eltwise_p", "Input1"], [67, 1, 1, "c.i32_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.i32_eltwise_p", "length"], [67, 1, 1, "c.i32_eltwise_p", "output"]], "i32_eltwise_s": [[67, 1, 1, "c.i32_eltwise_s", "Input0"], [67, 1, 1, "c.i32_eltwise_s", "Input1"], [67, 1, 1, "c.i32_eltwise_s", "core_mask"], [67, 1, 1, "c.i32_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.i32_eltwise_s", "length"], [67, 1, 1, "c.i32_eltwise_s", "output"]], "i32_equal_p": [[70, 1, 1, "c.i32_equal_p", "Input0"], [70, 1, 1, "c.i32_equal_p", "Input1"], [70, 1, 1, "c.i32_equal_p", "length"], [70, 1, 1, "c.i32_equal_p", "output"]], "i32_equal_s": [[70, 1, 1, "c.i32_equal_s", "Input0"], [70, 1, 1, "c.i32_equal_s", "Input1"], [70, 1, 1, "c.i32_equal_s", "core_mask"], [70, 1, 1, "c.i32_equal_s", "length"], [70, 1, 1, "c.i32_equal_s", "output"]], "i32_expfusion_p": [[73, 1, 1, "c.i32_expfusion_p", "dst_data"], [73, 1, 1, "c.i32_expfusion_p", "in_scale"], [73, 1, 1, "c.i32_expfusion_p", "length"], [73, 1, 1, "c.i32_expfusion_p", "out_scale"], [73, 1, 1, "c.i32_expfusion_p", "scale"], [73, 1, 1, "c.i32_expfusion_p", "src_data"]], "i32_expfusion_s": [[73, 1, 1, "c.i32_expfusion_s", "core_mask"], [73, 1, 1, "c.i32_expfusion_s", "dst_data"], [73, 1, 1, "c.i32_expfusion_s", "in_scale"], [73, 1, 1, "c.i32_expfusion_s", "length"], [73, 1, 1, "c.i32_expfusion_s", "out_scale"], [73, 1, 1, "c.i32_expfusion_s", "scale"], [73, 1, 1, "c.i32_expfusion_s", "src_data"]], "i32_extract_features_p": [[55, 1, 1, "c.i32_extract_features_p", "num_strings"], [55, 1, 1, "c.i32_extract_features_p", "output_labels"], [55, 1, 1, "c.i32_extract_features_p", "output_weights"], [55, 1, 1, "c.i32_extract_features_p", "string_lengths"], [55, 1, 1, "c.i32_extract_features_p", "string_pointers"]], "i32_extract_features_s": [[55, 1, 1, "c.i32_extract_features_s", "core_mask"], [55, 1, 1, "c.i32_extract_features_s", "num_strings"], [55, 1, 1, "c.i32_extract_features_s", "output_labels"], [55, 1, 1, "c.i32_extract_features_s", "output_weights"], [55, 1, 1, "c.i32_extract_features_s", "string_lengths"], [55, 1, 1, "c.i32_extract_features_s", "string_pointers"]], "i32_fill_p": [[78, 1, 1, "c.i32_fill_p", "output"], [78, 1, 1, "c.i32_fill_p", "param"], [78, 1, 1, "c.i32_fill_p", "value"]], "i32_fill_s": [[78, 1, 1, "c.i32_fill_s", "core_mask"], [78, 1, 1, "c.i32_fill_s", "output"], [78, 1, 1, "c.i32_fill_s", "param"], [78, 1, 1, "c.i32_fill_s", "value"]], "i32_formattranspose_p": [[85, 1, 1, "c.i32_formattranspose_p", "batch"], [85, 1, 1, "c.i32_formattranspose_p", "channel"], [85, 1, 1, "c.i32_formattranspose_p", "dst_data"], [85, 1, 1, "c.i32_formattranspose_p", "dst_format"], [85, 1, 1, "c.i32_formattranspose_p", "plane"], [85, 1, 1, "c.i32_formattranspose_p", "src_data"], [85, 1, 1, "c.i32_formattranspose_p", "src_format"]], "i32_formattranspose_s": [[85, 1, 1, "c.i32_formattranspose_s", "batch"], [85, 1, 1, "c.i32_formattranspose_s", "channel"], [85, 1, 1, "c.i32_formattranspose_s", "core_mask"], [85, 1, 1, "c.i32_formattranspose_s", "dst_data"], [85, 1, 1, "c.i32_formattranspose_s", "dst_format"], [85, 1, 1, "c.i32_formattranspose_s", "plane"], [85, 1, 1, "c.i32_formattranspose_s", "src_data"], [85, 1, 1, "c.i32_formattranspose_s", "src_format"]], "i32_gather_nd_p": [[89, 1, 1, "c.i32_gather_nd_p", "indices"], [89, 1, 1, "c.i32_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.i32_gather_nd_p", "indices_shape"], [89, 1, 1, "c.i32_gather_nd_p", "input"], [89, 1, 1, "c.i32_gather_nd_p", "input_ndim"], [89, 1, 1, "c.i32_gather_nd_p", "input_shape"], [89, 1, 1, "c.i32_gather_nd_p", "output"]], "i32_gather_nd_s": [[89, 1, 1, "c.i32_gather_nd_s", "core_mask"], [89, 1, 1, "c.i32_gather_nd_s", "indices"], [89, 1, 1, "c.i32_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.i32_gather_nd_s", "indices_shape"], [89, 1, 1, "c.i32_gather_nd_s", "input"], [89, 1, 1, "c.i32_gather_nd_s", "input_ndim"], [89, 1, 1, "c.i32_gather_nd_s", "input_shape"], [89, 1, 1, "c.i32_gather_nd_s", "output"]], "i32_gather_p": [[88, 1, 1, "c.i32_gather_p", "axis"], [88, 1, 1, "c.i32_gather_p", "batch_dims"], [88, 1, 1, "c.i32_gather_p", "indices"], [88, 1, 1, "c.i32_gather_p", "indices_ndim"], [88, 1, 1, "c.i32_gather_p", "indices_shape"], [88, 1, 1, "c.i32_gather_p", "input"], [88, 1, 1, "c.i32_gather_p", "input_ndim"], [88, 1, 1, "c.i32_gather_p", "input_shape"], [88, 1, 1, "c.i32_gather_p", "output"]], "i32_gather_s": [[88, 1, 1, "c.i32_gather_s", "axis"], [88, 1, 1, "c.i32_gather_s", "batch_dims"], [88, 1, 1, "c.i32_gather_s", "core_mask"], [88, 1, 1, "c.i32_gather_s", "indices"], [88, 1, 1, "c.i32_gather_s", "indices_ndim"], [88, 1, 1, "c.i32_gather_s", "indices_shape"], [88, 1, 1, "c.i32_gather_s", "input"], [88, 1, 1, "c.i32_gather_s", "input_ndim"], [88, 1, 1, "c.i32_gather_s", "input_shape"], [88, 1, 1, "c.i32_gather_s", "output"]], "i32_gatherd_p": [[90, 1, 1, "c.i32_gatherd_p", "dim"], [90, 1, 1, "c.i32_gatherd_p", "index"], [90, 1, 1, "c.i32_gatherd_p", "index_shape"], [90, 1, 1, "c.i32_gatherd_p", "input_shape"], [90, 1, 1, "c.i32_gatherd_p", "input_shape_size"], [90, 1, 1, "c.i32_gatherd_p", "input_x"], [90, 1, 1, "c.i32_gatherd_p", "output"]], "i32_gatherd_s": [[90, 1, 1, "c.i32_gatherd_s", "core_mask"], [90, 1, 1, "c.i32_gatherd_s", "dim"], [90, 1, 1, "c.i32_gatherd_s", "index"], [90, 1, 1, "c.i32_gatherd_s", "index_shape"], [90, 1, 1, "c.i32_gatherd_s", "input_shape"], [90, 1, 1, "c.i32_gatherd_s", "input_shape_size"], [90, 1, 1, "c.i32_gatherd_s", "input_x"], [90, 1, 1, "c.i32_gatherd_s", "output"]], "i32_greater_p": [[92, 1, 1, "c.i32_greater_p", "element_num"], [92, 1, 1, "c.i32_greater_p", "in_elements_num0"], [92, 1, 1, "c.i32_greater_p", "input1"], [92, 1, 1, "c.i32_greater_p", "input2"], [92, 1, 1, "c.i32_greater_p", "optimize"], [92, 1, 1, "c.i32_greater_p", "output"]], "i32_greater_s": [[92, 1, 1, "c.i32_greater_s", "core_mask"], [92, 1, 1, "c.i32_greater_s", "element_num"], [92, 1, 1, "c.i32_greater_s", "in_elements_num0"], [92, 1, 1, "c.i32_greater_s", "input1"], [92, 1, 1, "c.i32_greater_s", "input2"], [92, 1, 1, "c.i32_greater_s", "optimize"], [92, 1, 1, "c.i32_greater_s", "output"]], "i32_greaterequal_p": [[93, 1, 1, "c.i32_greaterequal_p", "element_num"], [93, 1, 1, "c.i32_greaterequal_p", "in_elements_num0"], [93, 1, 1, "c.i32_greaterequal_p", "input1"], [93, 1, 1, "c.i32_greaterequal_p", "input2"], [93, 1, 1, "c.i32_greaterequal_p", "optimize"], [93, 1, 1, "c.i32_greaterequal_p", "output"]], "i32_greaterequal_s": [[93, 1, 1, "c.i32_greaterequal_s", "core_mask"], [93, 1, 1, "c.i32_greaterequal_s", "element_num"], [93, 1, 1, "c.i32_greaterequal_s", "in_elements_num0"], [93, 1, 1, "c.i32_greaterequal_s", "input1"], [93, 1, 1, "c.i32_greaterequal_s", "input2"], [93, 1, 1, "c.i32_greaterequal_s", "optimize"], [93, 1, 1, "c.i32_greaterequal_s", "output"]], "i32_hashtablelookup_p": [[96, 1, 1, "c.i32_hashtablelookup_p", "hits_tensor"], [96, 1, 1, "c.i32_hashtablelookup_p", "input_size"], [96, 1, 1, "c.i32_hashtablelookup_p", "input_tensor"], [96, 1, 1, "c.i32_hashtablelookup_p", "keys_size"], [96, 1, 1, "c.i32_hashtablelookup_p", "keys_tensor"], [96, 1, 1, "c.i32_hashtablelookup_p", "output_tensor"], [96, 1, 1, "c.i32_hashtablelookup_p", "values_tensor"]], "i32_hashtablelookup_s": [[96, 1, 1, "c.i32_hashtablelookup_s", "core_mask"], [96, 1, 1, "c.i32_hashtablelookup_s", "hits_tensor"], [96, 1, 1, "c.i32_hashtablelookup_s", "input_size"], [96, 1, 1, "c.i32_hashtablelookup_s", "input_tensor"], [96, 1, 1, "c.i32_hashtablelookup_s", "keys_size"], [96, 1, 1, "c.i32_hashtablelookup_s", "keys_tensor"], [96, 1, 1, "c.i32_hashtablelookup_s", "output_tensor"], [96, 1, 1, "c.i32_hashtablelookup_s", "values_tensor"]], "i32_invertpermutation_p": [[98, 1, 1, "c.i32_invertpermutation_p", "input"], [98, 1, 1, "c.i32_invertpermutation_p", "num"], [98, 1, 1, "c.i32_invertpermutation_p", "output"]], "i32_invertpermutation_s": [[98, 1, 1, "c.i32_invertpermutation_s", "core_mask"], [98, 1, 1, "c.i32_invertpermutation_s", "input"], [98, 1, 1, "c.i32_invertpermutation_s", "num"], [98, 1, 1, "c.i32_invertpermutation_s", "output"]], "i32_isfinite_p": [[99, 1, 1, "c.i32_isfinite_p", "Input"], [99, 1, 1, "c.i32_isfinite_p", "length"], [99, 1, 1, "c.i32_isfinite_p", "output"]], "i32_isfinite_s": [[99, 1, 1, "c.i32_isfinite_s", "Input"], [99, 1, 1, "c.i32_isfinite_s", "core_mask"], [99, 1, 1, "c.i32_isfinite_s", "length"], [99, 1, 1, "c.i32_isfinite_s", "output"]], "i32_less_p": [[104, 1, 1, "c.i32_less_p", "Input0"], [104, 1, 1, "c.i32_less_p", "Input1"], [104, 1, 1, "c.i32_less_p", "in_elements_num0"], [104, 1, 1, "c.i32_less_p", "length"], [104, 1, 1, "c.i32_less_p", "optimize"], [104, 1, 1, "c.i32_less_p", "output"]], "i32_less_s": [[104, 1, 1, "c.i32_less_s", "Input0"], [104, 1, 1, "c.i32_less_s", "Input1"], [104, 1, 1, "c.i32_less_s", "core_mask"], [104, 1, 1, "c.i32_less_s", "in_elements_num0"], [104, 1, 1, "c.i32_less_s", "length"], [104, 1, 1, "c.i32_less_s", "optimize"], [104, 1, 1, "c.i32_less_s", "output"]], "i32_lessequal_p": [[105, 1, 1, "c.i32_lessequal_p", "Input0"], [105, 1, 1, "c.i32_lessequal_p", "Input1"], [105, 1, 1, "c.i32_lessequal_p", "in_elements_num0"], [105, 1, 1, "c.i32_lessequal_p", "length"], [105, 1, 1, "c.i32_lessequal_p", "optimize"], [105, 1, 1, "c.i32_lessequal_p", "output"]], "i32_lessequal_s": [[105, 1, 1, "c.i32_lessequal_s", "Input0"], [105, 1, 1, "c.i32_lessequal_s", "Input1"], [105, 1, 1, "c.i32_lessequal_s", "core_mask"], [105, 1, 1, "c.i32_lessequal_s", "in_elements_num0"], [105, 1, 1, "c.i32_lessequal_s", "length"], [105, 1, 1, "c.i32_lessequal_s", "optimize"], [105, 1, 1, "c.i32_lessequal_s", "output"]], "i32_log1p_p": [[108, 1, 1, "c.i32_log1p_p", "Input"], [108, 1, 1, "c.i32_log1p_p", "length"], [108, 1, 1, "c.i32_log1p_p", "output"]], "i32_log1p_s": [[108, 1, 1, "c.i32_log1p_s", "Input"], [108, 1, 1, "c.i32_log1p_s", "core_mask"], [108, 1, 1, "c.i32_log1p_s", "length"], [108, 1, 1, "c.i32_log1p_s", "output"]], "i32_log_p": [[107, 1, 1, "c.i32_log_p", "input"], [107, 1, 1, "c.i32_log_p", "length"], [107, 1, 1, "c.i32_log_p", "output"]], "i32_log_s": [[107, 1, 1, "c.i32_log_s", "core_mask"], [107, 1, 1, "c.i32_log_s", "input"], [107, 1, 1, "c.i32_log_s", "length"], [107, 1, 1, "c.i32_log_s", "output"]], "i32_logical_not_p": [[110, 1, 1, "c.i32_logical_not_p", "input"], [110, 1, 1, "c.i32_logical_not_p", "length"], [110, 1, 1, "c.i32_logical_not_p", "output"]], "i32_logical_not_s": [[110, 1, 1, "c.i32_logical_not_s", "core_mask"], [110, 1, 1, "c.i32_logical_not_s", "input"], [110, 1, 1, "c.i32_logical_not_s", "length"], [110, 1, 1, "c.i32_logical_not_s", "output"]], "i32_logical_or_p": [[111, 1, 1, "c.i32_logical_or_p", "input0"], [111, 1, 1, "c.i32_logical_or_p", "input1"], [111, 1, 1, "c.i32_logical_or_p", "length"], [111, 1, 1, "c.i32_logical_or_p", "output"]], "i32_logical_or_s": [[111, 1, 1, "c.i32_logical_or_s", "core_mask"], [111, 1, 1, "c.i32_logical_or_s", "input0"], [111, 1, 1, "c.i32_logical_or_s", "input1"], [111, 1, 1, "c.i32_logical_or_s", "length"], [111, 1, 1, "c.i32_logical_or_s", "output"]], "i32_lsh_projection_p": [[116, 1, 1, "c.i32_lsh_projection_p", "bits_per_hash"], [116, 1, 1, "c.i32_lsh_projection_p", "feature"], [116, 1, 1, "c.i32_lsh_projection_p", "feature_num"], [116, 1, 1, "c.i32_lsh_projection_p", "hash_group_num"], [116, 1, 1, "c.i32_lsh_projection_p", "hash_seed"], [116, 1, 1, "c.i32_lsh_projection_p", "output"], [116, 1, 1, "c.i32_lsh_projection_p", "weight"]], "i32_lsh_projection_s": [[116, 1, 1, "c.i32_lsh_projection_s", "bits_per_hash"], [116, 1, 1, "c.i32_lsh_projection_s", "core_mask"], [116, 1, 1, "c.i32_lsh_projection_s", "feature"], [116, 1, 1, "c.i32_lsh_projection_s", "feature_num"], [116, 1, 1, "c.i32_lsh_projection_s", "hash_group_num"], [116, 1, 1, "c.i32_lsh_projection_s", "hash_seed"], [116, 1, 1, "c.i32_lsh_projection_s", "output"], [116, 1, 1, "c.i32_lsh_projection_s", "weight"]], "i32_matmulfusion_p": [[121, 1, 1, "c.i32_matmulfusion_p", "A"], [121, 1, 1, "c.i32_matmulfusion_p", "B"], [121, 1, 1, "c.i32_matmulfusion_p", "C"], [121, 1, 1, "c.i32_matmulfusion_p", "K"], [121, 1, 1, "c.i32_matmulfusion_p", "M"], [121, 1, 1, "c.i32_matmulfusion_p", "N"], [121, 1, 1, "c.i32_matmulfusion_p", "activation_type"], [121, 1, 1, "c.i32_matmulfusion_p", "bias"]], "i32_matmulfusion_s": [[121, 1, 1, "c.i32_matmulfusion_s", "A"], [121, 1, 1, "c.i32_matmulfusion_s", "B"], [121, 1, 1, "c.i32_matmulfusion_s", "C"], [121, 1, 1, "c.i32_matmulfusion_s", "K"], [121, 1, 1, "c.i32_matmulfusion_s", "M"], [121, 1, 1, "c.i32_matmulfusion_s", "N"], [121, 1, 1, "c.i32_matmulfusion_s", "activation_type"], [121, 1, 1, "c.i32_matmulfusion_s", "bias"], [121, 1, 1, "c.i32_matmulfusion_s", "core_mask"]], "i32_maximum_p": [[122, 1, 1, "c.i32_maximum_p", "input0"], [122, 1, 1, "c.i32_maximum_p", "input1"], [122, 1, 1, "c.i32_maximum_p", "length"], [122, 1, 1, "c.i32_maximum_p", "output"]], "i32_maximum_s": [[122, 1, 1, "c.i32_maximum_s", "core_mask"], [122, 1, 1, "c.i32_maximum_s", "input0"], [122, 1, 1, "c.i32_maximum_s", "input1"], [122, 1, 1, "c.i32_maximum_s", "length"], [122, 1, 1, "c.i32_maximum_s", "output"]], "i32_minimum_p": [[127, 1, 1, "c.i32_minimum_p", "input0"], [127, 1, 1, "c.i32_minimum_p", "input1"], [127, 1, 1, "c.i32_minimum_p", "length"], [127, 1, 1, "c.i32_minimum_p", "output"]], "i32_minimum_s": [[127, 1, 1, "c.i32_minimum_s", "core_mask"], [127, 1, 1, "c.i32_minimum_s", "input0"], [127, 1, 1, "c.i32_minimum_s", "input1"], [127, 1, 1, "c.i32_minimum_s", "length"], [127, 1, 1, "c.i32_minimum_s", "output"]], "i32_mod_p": [[129, 1, 1, "c.i32_mod_p", "input0"], [129, 1, 1, "c.i32_mod_p", "input1"], [129, 1, 1, "c.i32_mod_p", "length"], [129, 1, 1, "c.i32_mod_p", "output"]], "i32_mod_s": [[129, 1, 1, "c.i32_mod_s", "core_mask"], [129, 1, 1, "c.i32_mod_s", "input0"], [129, 1, 1, "c.i32_mod_s", "input1"], [129, 1, 1, "c.i32_mod_s", "length"], [129, 1, 1, "c.i32_mod_s", "output"]], "i32_mul_p": [[130, 1, 1, "c.i32_mul_p", "input0"], [130, 1, 1, "c.i32_mul_p", "input1"], [130, 1, 1, "c.i32_mul_p", "length"], [130, 1, 1, "c.i32_mul_p", "output"]], "i32_mul_s": [[130, 1, 1, "c.i32_mul_s", "core_mask"], [130, 1, 1, "c.i32_mul_s", "input0"], [130, 1, 1, "c.i32_mul_s", "input1"], [130, 1, 1, "c.i32_mul_s", "length"], [130, 1, 1, "c.i32_mul_s", "output"]], "i32_neg_grad_p": [[133, 1, 1, "c.i32_neg_grad_p", "Input"], [133, 1, 1, "c.i32_neg_grad_p", "length"], [133, 1, 1, "c.i32_neg_grad_p", "output"]], "i32_neg_grad_s": [[133, 1, 1, "c.i32_neg_grad_s", "Input"], [133, 1, 1, "c.i32_neg_grad_s", "core_mask"], [133, 1, 1, "c.i32_neg_grad_s", "length"], [133, 1, 1, "c.i32_neg_grad_s", "output"]], "i32_neg_p": [[132, 1, 1, "c.i32_neg_p", "Input"], [132, 1, 1, "c.i32_neg_p", "length"], [132, 1, 1, "c.i32_neg_p", "output"]], "i32_neg_s": [[132, 1, 1, "c.i32_neg_s", "Input"], [132, 1, 1, "c.i32_neg_s", "core_mask"], [132, 1, 1, "c.i32_neg_s", "length"], [132, 1, 1, "c.i32_neg_s", "output"]], "i32_nonzero_p": [[137, 1, 1, "c.i32_nonzero_p", "dim_strides"], [137, 1, 1, "c.i32_nonzero_p", "input"], [137, 1, 1, "c.i32_nonzero_p", "input_rank"], [137, 1, 1, "c.i32_nonzero_p", "length"], [137, 1, 1, "c.i32_nonzero_p", "non_zero_num"], [137, 1, 1, "c.i32_nonzero_p", "output"], [137, 1, 1, "c.i32_nonzero_p", "shape"]], "i32_nonzero_s": [[137, 1, 1, "c.i32_nonzero_s", "core_mask"], [137, 1, 1, "c.i32_nonzero_s", "dim_strides"], [137, 1, 1, "c.i32_nonzero_s", "input"], [137, 1, 1, "c.i32_nonzero_s", "input_rank"], [137, 1, 1, "c.i32_nonzero_s", "length"], [137, 1, 1, "c.i32_nonzero_s", "non_zero_num"], [137, 1, 1, "c.i32_nonzero_s", "output"], [137, 1, 1, "c.i32_nonzero_s", "shape"]], "i32_not_equal_p": [[138, 1, 1, "c.i32_not_equal_p", "Input0"], [138, 1, 1, "c.i32_not_equal_p", "Input1"], [138, 1, 1, "c.i32_not_equal_p", "length"], [138, 1, 1, "c.i32_not_equal_p", "output"]], "i32_not_equal_s": [[138, 1, 1, "c.i32_not_equal_s", "Input0"], [138, 1, 1, "c.i32_not_equal_s", "Input1"], [138, 1, 1, "c.i32_not_equal_s", "core_mask"], [138, 1, 1, "c.i32_not_equal_s", "length"], [138, 1, 1, "c.i32_not_equal_s", "output"]], "i32_onehot_p": [[139, 1, 1, "c.i32_onehot_p", "axis"], [139, 1, 1, "c.i32_onehot_p", "depth"], [139, 1, 1, "c.i32_onehot_p", "indices"], [139, 1, 1, "c.i32_onehot_p", "indices_shape"], [139, 1, 1, "c.i32_onehot_p", "indices_shape_size"], [139, 1, 1, "c.i32_onehot_p", "on_off"], [139, 1, 1, "c.i32_onehot_p", "output"], [139, 1, 1, "c.i32_onehot_p", "support_neg_index"]], "i32_onehot_s": [[139, 1, 1, "c.i32_onehot_s", "axis"], [139, 1, 1, "c.i32_onehot_s", "core_mask"], [139, 1, 1, "c.i32_onehot_s", "depth"], [139, 1, 1, "c.i32_onehot_s", "indices"], [139, 1, 1, "c.i32_onehot_s", "indices_shape"], [139, 1, 1, "c.i32_onehot_s", "indices_shape_size"], [139, 1, 1, "c.i32_onehot_s", "on_off"], [139, 1, 1, "c.i32_onehot_s", "output"], [139, 1, 1, "c.i32_onehot_s", "support_neg_index"]], "i32_ones_like_p": [[140, 1, 1, "c.i32_ones_like_p", "length"], [140, 1, 1, "c.i32_ones_like_p", "output"]], "i32_ones_like_s": [[140, 1, 1, "c.i32_ones_like_s", "core_mask"], [140, 1, 1, "c.i32_ones_like_s", "length"], [140, 1, 1, "c.i32_ones_like_s", "output"]], "i32_padfusion_p": [[141, 1, 1, "c.i32_padfusion_p", "params"]], "i32_padfusion_s": [[141, 1, 1, "c.i32_padfusion_s", "core_mask"], [141, 1, 1, "c.i32_padfusion_s", "params"]], "i32_pow_fusion_p": [[142, 1, 1, "c.i32_pow_fusion_p", "Input"], [142, 1, 1, "c.i32_pow_fusion_p", "broadcast"], [142, 1, 1, "c.i32_pow_fusion_p", "exponent"], [142, 1, 1, "c.i32_pow_fusion_p", "length_in"], [142, 1, 1, "c.i32_pow_fusion_p", "output"], [142, 1, 1, "c.i32_pow_fusion_p", "scale"], [142, 1, 1, "c.i32_pow_fusion_p", "shift"]], "i32_pow_fusion_s": [[142, 1, 1, "c.i32_pow_fusion_s", "Input"], [142, 1, 1, "c.i32_pow_fusion_s", "broadcast"], [142, 1, 1, "c.i32_pow_fusion_s", "core_mask"], [142, 1, 1, "c.i32_pow_fusion_s", "exponent"], [142, 1, 1, "c.i32_pow_fusion_s", "length_in"], [142, 1, 1, "c.i32_pow_fusion_s", "output"], [142, 1, 1, "c.i32_pow_fusion_s", "scale"], [142, 1, 1, "c.i32_pow_fusion_s", "shift"]], "i32_raggedrange_p": [[147, 1, 1, "c.i32_raggedrange_p", "deltas"], [147, 1, 1, "c.i32_raggedrange_p", "limits"], [147, 1, 1, "c.i32_raggedrange_p", "range_count"], [147, 1, 1, "c.i32_raggedrange_p", "splits"], [147, 1, 1, "c.i32_raggedrange_p", "starts"], [147, 1, 1, "c.i32_raggedrange_p", "values"]], "i32_raggedrange_s": [[147, 1, 1, "c.i32_raggedrange_s", "core_mask"], [147, 1, 1, "c.i32_raggedrange_s", "deltas"], [147, 1, 1, "c.i32_raggedrange_s", "limits"], [147, 1, 1, "c.i32_raggedrange_s", "range_count"], [147, 1, 1, "c.i32_raggedrange_s", "splits"], [147, 1, 1, "c.i32_raggedrange_s", "starts"], [147, 1, 1, "c.i32_raggedrange_s", "values"]], "i32_range_p": [[150, 1, 1, "c.i32_range_p", "delta"], [150, 1, 1, "c.i32_range_p", "length"], [150, 1, 1, "c.i32_range_p", "output"], [150, 1, 1, "c.i32_range_p", "start"]], "i32_range_s": [[150, 1, 1, "c.i32_range_s", "core_mask"], [150, 1, 1, "c.i32_range_s", "delta"], [150, 1, 1, "c.i32_range_s", "length"], [150, 1, 1, "c.i32_range_s", "output"], [150, 1, 1, "c.i32_range_s", "start"]], "i32_real_div_p": [[152, 1, 1, "c.i32_real_div_p", "input0"], [152, 1, 1, "c.i32_real_div_p", "input1"], [152, 1, 1, "c.i32_real_div_p", "length"], [152, 1, 1, "c.i32_real_div_p", "output"]], "i32_real_div_s": [[152, 1, 1, "c.i32_real_div_s", "core_mask"], [152, 1, 1, "c.i32_real_div_s", "input0"], [152, 1, 1, "c.i32_real_div_s", "input1"], [152, 1, 1, "c.i32_real_div_s", "length"], [152, 1, 1, "c.i32_real_div_s", "output"]], "i32_reciprocal_p": [[153, 1, 1, "c.i32_reciprocal_p", "Input"], [153, 1, 1, "c.i32_reciprocal_p", "length"], [153, 1, 1, "c.i32_reciprocal_p", "output"]], "i32_reciprocal_s": [[153, 1, 1, "c.i32_reciprocal_s", "Input"], [153, 1, 1, "c.i32_reciprocal_s", "core_mask"], [153, 1, 1, "c.i32_reciprocal_s", "length"], [153, 1, 1, "c.i32_reciprocal_s", "output"]], "i32_reduce_p": [[154, 1, 1, "c.i32_reduce_p", "core_mask"], [154, 1, 1, "c.i32_reduce_p", "dst_data"], [154, 1, 1, "c.i32_reduce_p", "param"], [154, 1, 1, "c.i32_reduce_p", "src_data"], [154, 1, 1, "c.i32_reduce_p", "tmp_dst_data"], [154, 1, 1, "c.i32_reduce_p", "tmp_src_data"]], "i32_reduce_s": [[154, 1, 1, "c.i32_reduce_s", "core_mask"], [154, 1, 1, "c.i32_reduce_s", "dst_data"], [154, 1, 1, "c.i32_reduce_s", "param"], [154, 1, 1, "c.i32_reduce_s", "src_data"]], "i32_reduceall_p": [[21, 1, 1, "c.i32_reduceall_p", "axis_size"], [21, 1, 1, "c.i32_reduceall_p", "dst_data"], [21, 1, 1, "c.i32_reduceall_p", "inner_size"], [21, 1, 1, "c.i32_reduceall_p", "outer_size"], [21, 1, 1, "c.i32_reduceall_p", "src_data"]], "i32_reduceall_s": [[21, 1, 1, "c.i32_reduceall_s", "axis_size"], [21, 1, 1, "c.i32_reduceall_s", "core_mask"], [21, 1, 1, "c.i32_reduceall_s", "dst_data"], [21, 1, 1, "c.i32_reduceall_s", "inner_size"], [21, 1, 1, "c.i32_reduceall_s", "outer_size"], [21, 1, 1, "c.i32_reduceall_s", "src_data"]], "i32_reducescatter_p": [[155, 1, 1, "c.i32_reducescatter_p", "data_size"], [155, 1, 1, "c.i32_reducescatter_p", "input_data"], [155, 1, 1, "c.i32_reducescatter_p", "output_data"], [155, 1, 1, "c.i32_reducescatter_p", "reduce_type"]], "i32_reducescatter_s": [[155, 1, 1, "c.i32_reducescatter_s", "core_mask"], [155, 1, 1, "c.i32_reducescatter_s", "data_size"], [155, 1, 1, "c.i32_reducescatter_s", "input_data"], [155, 1, 1, "c.i32_reducescatter_s", "output_data"], [155, 1, 1, "c.i32_reducescatter_s", "reduce_type"]], "i32_reshape_p": [[156, 1, 1, "c.i32_reshape_p", "input"], [156, 1, 1, "c.i32_reshape_p", "length"], [156, 1, 1, "c.i32_reshape_p", "output"]], "i32_reshape_s": [[156, 1, 1, "c.i32_reshape_s", "core_mask"], [156, 1, 1, "c.i32_reshape_s", "input"], [156, 1, 1, "c.i32_reshape_s", "length"], [156, 1, 1, "c.i32_reshape_s", "output"]], "i32_rsqrt_p": [[164, 1, 1, "c.i32_rsqrt_p", "dst"], [164, 1, 1, "c.i32_rsqrt_p", "length"], [164, 1, 1, "c.i32_rsqrt_p", "src"]], "i32_rsqrt_s": [[164, 1, 1, "c.i32_rsqrt_s", "core_mask"], [164, 1, 1, "c.i32_rsqrt_s", "dst"], [164, 1, 1, "c.i32_rsqrt_s", "length"], [164, 1, 1, "c.i32_rsqrt_s", "src"]], "i32_scalefusion_p": [[166, 1, 1, "c.i32_scalefusion_p", "bias"], [166, 1, 1, "c.i32_scalefusion_p", "dst_data"], [166, 1, 1, "c.i32_scalefusion_p", "length"], [166, 1, 1, "c.i32_scalefusion_p", "scale"], [166, 1, 1, "c.i32_scalefusion_p", "src_data"]], "i32_scalefusion_s": [[166, 1, 1, "c.i32_scalefusion_s", "bias"], [166, 1, 1, "c.i32_scalefusion_s", "core_mask"], [166, 1, 1, "c.i32_scalefusion_s", "dst_data"], [166, 1, 1, "c.i32_scalefusion_s", "length"], [166, 1, 1, "c.i32_scalefusion_s", "scale"], [166, 1, 1, "c.i32_scalefusion_s", "src_data"]], "i32_scatter_elements_p": [[167, 1, 1, "c.i32_scatter_elements_p", "core_mask"], [167, 1, 1, "c.i32_scatter_elements_p", "indices"], [167, 1, 1, "c.i32_scatter_elements_p", "input"], [167, 1, 1, "c.i32_scatter_elements_p", "output"], [167, 1, 1, "c.i32_scatter_elements_p", "param"], [167, 1, 1, "c.i32_scatter_elements_p", "updates"]], "i32_scatter_elements_s": [[167, 1, 1, "c.i32_scatter_elements_s", "core_mask"], [167, 1, 1, "c.i32_scatter_elements_s", "indices"], [167, 1, 1, "c.i32_scatter_elements_s", "input"], [167, 1, 1, "c.i32_scatter_elements_s", "output"], [167, 1, 1, "c.i32_scatter_elements_s", "param"], [167, 1, 1, "c.i32_scatter_elements_s", "updates"]], "i32_scatter_nd_p": [[168, 1, 1, "c.i32_scatter_nd_p", "indices"], [168, 1, 1, "c.i32_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.i32_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.i32_scatter_nd_p", "output"], [168, 1, 1, "c.i32_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.i32_scatter_nd_p", "output_shape"], [168, 1, 1, "c.i32_scatter_nd_p", "updates"]], "i32_scatter_nd_s": [[168, 1, 1, "c.i32_scatter_nd_s", "core_mask"], [168, 1, 1, "c.i32_scatter_nd_s", "indices"], [168, 1, 1, "c.i32_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.i32_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.i32_scatter_nd_s", "output"], [168, 1, 1, "c.i32_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.i32_scatter_nd_s", "output_shape"], [168, 1, 1, "c.i32_scatter_nd_s", "updates"]], "i32_scatter_nd_update_p": [[169, 1, 1, "c.i32_scatter_nd_update_p", "indices"], [169, 1, 1, "c.i32_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.i32_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.i32_scatter_nd_update_p", "output"], [169, 1, 1, "c.i32_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.i32_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.i32_scatter_nd_update_p", "updates"]], "i32_scatter_nd_update_s": [[169, 1, 1, "c.i32_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.i32_scatter_nd_update_s", "indices"], [169, 1, 1, "c.i32_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.i32_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.i32_scatter_nd_update_s", "output"], [169, 1, 1, "c.i32_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.i32_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.i32_scatter_nd_update_s", "updates"]], "i32_select_p": [[170, 1, 1, "c.i32_select_p", "condition"], [170, 1, 1, "c.i32_select_p", "index_list1"], [170, 1, 1, "c.i32_select_p", "index_list2"], [170, 1, 1, "c.i32_select_p", "index_list3"], [170, 1, 1, "c.i32_select_p", "input0"], [170, 1, 1, "c.i32_select_p", "input1"], [170, 1, 1, "c.i32_select_p", "is_broadcast"], [170, 1, 1, "c.i32_select_p", "output"], [170, 1, 1, "c.i32_select_p", "output_dims"], [170, 1, 1, "c.i32_select_p", "output_dims_num"]], "i32_select_s": [[170, 1, 1, "c.i32_select_s", "condition"], [170, 1, 1, "c.i32_select_s", "core_mask"], [170, 1, 1, "c.i32_select_s", "index_list1"], [170, 1, 1, "c.i32_select_s", "index_list2"], [170, 1, 1, "c.i32_select_s", "index_list3"], [170, 1, 1, "c.i32_select_s", "input0"], [170, 1, 1, "c.i32_select_s", "input1"], [170, 1, 1, "c.i32_select_s", "is_broadcast"], [170, 1, 1, "c.i32_select_s", "output"], [170, 1, 1, "c.i32_select_s", "output_dims"], [170, 1, 1, "c.i32_select_s", "output_dims_num"]], "i32_sin_p": [[175, 1, 1, "c.i32_sin_p", "dst_data"], [175, 1, 1, "c.i32_sin_p", "length"], [175, 1, 1, "c.i32_sin_p", "src_data"]], "i32_sin_s": [[175, 1, 1, "c.i32_sin_s", "core_mask"], [175, 1, 1, "c.i32_sin_s", "dst_data"], [175, 1, 1, "c.i32_sin_s", "length"], [175, 1, 1, "c.i32_sin_s", "src_data"]], "i32_slice_p": [[178, 1, 1, "c.i32_slice_p", "begin"], [178, 1, 1, "c.i32_slice_p", "input"], [178, 1, 1, "c.i32_slice_p", "input_shape"], [178, 1, 1, "c.i32_slice_p", "ndim"], [178, 1, 1, "c.i32_slice_p", "output"], [178, 1, 1, "c.i32_slice_p", "size"]], "i32_slice_s": [[178, 1, 1, "c.i32_slice_s", "begin"], [178, 1, 1, "c.i32_slice_s", "core_mask"], [178, 1, 1, "c.i32_slice_s", "input"], [178, 1, 1, "c.i32_slice_s", "input_shape"], [178, 1, 1, "c.i32_slice_s", "ndim"], [178, 1, 1, "c.i32_slice_s", "output"], [178, 1, 1, "c.i32_slice_s", "size"]], "i32_spacetobatch_p": [[183, 1, 1, "c.i32_spacetobatch_p", "block_size"], [183, 1, 1, "c.i32_spacetobatch_p", "data_size"], [183, 1, 1, "c.i32_spacetobatch_p", "input"], [183, 1, 1, "c.i32_spacetobatch_p", "input_shape"], [183, 1, 1, "c.i32_spacetobatch_p", "output"], [183, 1, 1, "c.i32_spacetobatch_p", "paddings"]], "i32_spacetobatch_s": [[183, 1, 1, "c.i32_spacetobatch_s", "block_size"], [183, 1, 1, "c.i32_spacetobatch_s", "core_mask"], [183, 1, 1, "c.i32_spacetobatch_s", "data_size"], [183, 1, 1, "c.i32_spacetobatch_s", "input"], [183, 1, 1, "c.i32_spacetobatch_s", "input_shape"], [183, 1, 1, "c.i32_spacetobatch_s", "output"], [183, 1, 1, "c.i32_spacetobatch_s", "paddings"]], "i32_spacetobatchnd_p": [[184, 1, 1, "c.i32_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.i32_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.i32_spacetobatchnd_p", "input"], [184, 1, 1, "c.i32_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.i32_spacetobatchnd_p", "output"], [184, 1, 1, "c.i32_spacetobatchnd_p", "paddings"]], "i32_spacetobatchnd_s": [[184, 1, 1, "c.i32_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.i32_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.i32_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.i32_spacetobatchnd_s", "input"], [184, 1, 1, "c.i32_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.i32_spacetobatchnd_s", "output"], [184, 1, 1, "c.i32_spacetobatchnd_s", "paddings"]], "i32_spacetodepth_p": [[185, 1, 1, "c.i32_spacetodepth_p", "block"], [185, 1, 1, "c.i32_spacetodepth_p", "data_size"], [185, 1, 1, "c.i32_spacetodepth_p", "in_shape"], [185, 1, 1, "c.i32_spacetodepth_p", "input"], [185, 1, 1, "c.i32_spacetodepth_p", "output"]], "i32_spacetodepth_s": [[185, 1, 1, "c.i32_spacetodepth_s", "block"], [185, 1, 1, "c.i32_spacetodepth_s", "core_mask"], [185, 1, 1, "c.i32_spacetodepth_s", "data_size"], [185, 1, 1, "c.i32_spacetodepth_s", "in_shape"], [185, 1, 1, "c.i32_spacetodepth_s", "input"], [185, 1, 1, "c.i32_spacetodepth_s", "output"]], "i32_sparsefillemptyrows_p": [[187, 1, 1, "c.i32_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_p", "values_ptr"]], "i32_sparsefillemptyrows_s": [[187, 1, 1, "c.i32_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.i32_sparsefillemptyrows_s", "values_ptr"]], "i32_sparsesegmentsum_p": [[189, 1, 1, "c.i32_sparsesegmentsum_p", "in_data"], [189, 1, 1, "c.i32_sparsesegmentsum_p", "in_data_shape"], [189, 1, 1, "c.i32_sparsesegmentsum_p", "in_data_shape_size"], [189, 1, 1, "c.i32_sparsesegmentsum_p", "in_indices"], [189, 1, 1, "c.i32_sparsesegmentsum_p", "in_indices_size"], [189, 1, 1, "c.i32_sparsesegmentsum_p", "in_segment_ids"], [189, 1, 1, "c.i32_sparsesegmentsum_p", "out_data"], [189, 1, 1, "c.i32_sparsesegmentsum_p", "out_data_shape"]], "i32_sparsesegmentsum_s": [[189, 1, 1, "c.i32_sparsesegmentsum_s", "core_mask"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "in_data"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "in_data_shape"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "in_data_shape_size"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "in_indices"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "in_indices_size"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "in_segment_ids"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "out_data"], [189, 1, 1, "c.i32_sparsesegmentsum_s", "out_data_shape"]], "i32_sparsetodense_p": [[190, 1, 1, "c.i32_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.i32_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.i32_sparsetodense_p", "output"], [190, 1, 1, "c.i32_sparsetodense_p", "output_strides"], [190, 1, 1, "c.i32_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.i32_sparsetodense_p", "sparse_values"]], "i32_sparsetodense_s": [[190, 1, 1, "c.i32_sparsetodense_s", "core_mask"], [190, 1, 1, "c.i32_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.i32_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.i32_sparsetodense_s", "output"], [190, 1, 1, "c.i32_sparsetodense_s", "output_strides"], [190, 1, 1, "c.i32_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.i32_sparsetodense_s", "sparse_values"]], "i32_splice_p": [[191, 1, 1, "c.i32_splice_p", "context_dim"], [191, 1, 1, "c.i32_splice_p", "dst_col"], [191, 1, 1, "c.i32_splice_p", "dst_data"], [191, 1, 1, "c.i32_splice_p", "dst_row"], [191, 1, 1, "c.i32_splice_p", "forward_indexes"], [191, 1, 1, "c.i32_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.i32_splice_p", "src_col"], [191, 1, 1, "c.i32_splice_p", "src_data"], [191, 1, 1, "c.i32_splice_p", "src_row"]], "i32_splice_s": [[191, 1, 1, "c.i32_splice_s", "context_dim"], [191, 1, 1, "c.i32_splice_s", "core_mask"], [191, 1, 1, "c.i32_splice_s", "dst_col"], [191, 1, 1, "c.i32_splice_s", "dst_data"], [191, 1, 1, "c.i32_splice_s", "dst_row"], [191, 1, 1, "c.i32_splice_s", "forward_indexes"], [191, 1, 1, "c.i32_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.i32_splice_s", "src_col"], [191, 1, 1, "c.i32_splice_s", "src_data"], [191, 1, 1, "c.i32_splice_s", "src_row"]], "i32_split_p": [[192, 1, 1, "c.i32_split_p", "axis"], [192, 1, 1, "c.i32_split_p", "input"], [192, 1, 1, "c.i32_split_p", "input_ndim"], [192, 1, 1, "c.i32_split_p", "input_shape"], [192, 1, 1, "c.i32_split_p", "num_split"], [192, 1, 1, "c.i32_split_p", "outputs"], [192, 1, 1, "c.i32_split_p", "split_sizes"]], "i32_split_s": [[192, 1, 1, "c.i32_split_s", "axis"], [192, 1, 1, "c.i32_split_s", "core_mask"], [192, 1, 1, "c.i32_split_s", "input"], [192, 1, 1, "c.i32_split_s", "input_ndim"], [192, 1, 1, "c.i32_split_s", "input_shape"], [192, 1, 1, "c.i32_split_s", "num_split"], [192, 1, 1, "c.i32_split_s", "outputs"], [192, 1, 1, "c.i32_split_s", "split_sizes"]], "i32_split_with_overlap_p": [[193, 1, 1, "c.i32_split_with_overlap_p", "axis"], [193, 1, 1, "c.i32_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.i32_split_with_overlap_p", "input"], [193, 1, 1, "c.i32_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.i32_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.i32_split_with_overlap_p", "num_split"], [193, 1, 1, "c.i32_split_with_overlap_p", "outputs"], [193, 1, 1, "c.i32_split_with_overlap_p", "start_indices"]], "i32_split_with_overlap_s": [[193, 1, 1, "c.i32_split_with_overlap_s", "axis"], [193, 1, 1, "c.i32_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.i32_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.i32_split_with_overlap_s", "input"], [193, 1, 1, "c.i32_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.i32_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.i32_split_with_overlap_s", "num_split"], [193, 1, 1, "c.i32_split_with_overlap_s", "outputs"], [193, 1, 1, "c.i32_split_with_overlap_s", "start_indices"]], "i32_sqrt_p": [[194, 1, 1, "c.i32_sqrt_p", "dst_data"], [194, 1, 1, "c.i32_sqrt_p", "length"], [194, 1, 1, "c.i32_sqrt_p", "src_data"]], "i32_sqrt_s": [[194, 1, 1, "c.i32_sqrt_s", "core_mask"], [194, 1, 1, "c.i32_sqrt_s", "dst_data"], [194, 1, 1, "c.i32_sqrt_s", "length"], [194, 1, 1, "c.i32_sqrt_s", "src_data"]], "i32_sqrtgrad_p": [[195, 1, 1, "c.i32_sqrtgrad_p", "input1"], [195, 1, 1, "c.i32_sqrtgrad_p", "input2"], [195, 1, 1, "c.i32_sqrtgrad_p", "output"], [195, 1, 1, "c.i32_sqrtgrad_p", "size"]], "i32_sqrtgrad_s": [[195, 1, 1, "c.i32_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.i32_sqrtgrad_s", "input1"], [195, 1, 1, "c.i32_sqrtgrad_s", "input2"], [195, 1, 1, "c.i32_sqrtgrad_s", "output"], [195, 1, 1, "c.i32_sqrtgrad_s", "size"]], "i32_square_p": [[196, 1, 1, "c.i32_square_p", "dst"], [196, 1, 1, "c.i32_square_p", "length"], [196, 1, 1, "c.i32_square_p", "src"]], "i32_square_s": [[196, 1, 1, "c.i32_square_s", "core_mask"], [196, 1, 1, "c.i32_square_s", "dst"], [196, 1, 1, "c.i32_square_s", "length"], [196, 1, 1, "c.i32_square_s", "src"]], "i32_squaredifference_p": [[197, 1, 1, "c.i32_squaredifference_p", "input0"], [197, 1, 1, "c.i32_squaredifference_p", "input1"], [197, 1, 1, "c.i32_squaredifference_p", "length"], [197, 1, 1, "c.i32_squaredifference_p", "output"]], "i32_squaredifference_s": [[197, 1, 1, "c.i32_squaredifference_s", "core_mask"], [197, 1, 1, "c.i32_squaredifference_s", "input0"], [197, 1, 1, "c.i32_squaredifference_s", "input1"], [197, 1, 1, "c.i32_squaredifference_s", "length"], [197, 1, 1, "c.i32_squaredifference_s", "output"]], "i32_stack_p": [[199, 1, 1, "c.i32_stack_p", "axis"], [199, 1, 1, "c.i32_stack_p", "input_ndim"], [199, 1, 1, "c.i32_stack_p", "input_shape"], [199, 1, 1, "c.i32_stack_p", "inputs"], [199, 1, 1, "c.i32_stack_p", "num_inputs"], [199, 1, 1, "c.i32_stack_p", "output"]], "i32_stack_s": [[199, 1, 1, "c.i32_stack_s", "axis"], [199, 1, 1, "c.i32_stack_s", "core_mask"], [199, 1, 1, "c.i32_stack_s", "input_ndim"], [199, 1, 1, "c.i32_stack_s", "input_shape"], [199, 1, 1, "c.i32_stack_s", "inputs"], [199, 1, 1, "c.i32_stack_s", "num_inputs"], [199, 1, 1, "c.i32_stack_s", "output"]], "i32_subrelu6_p": [[202, 1, 1, "c.i32_subrelu6_p", "input0"], [202, 1, 1, "c.i32_subrelu6_p", "input1"], [202, 1, 1, "c.i32_subrelu6_p", "output"], [202, 1, 1, "c.i32_subrelu6_p", "size"]], "i32_subrelu6_s": [[202, 1, 1, "c.i32_subrelu6_s", "core_mask"], [202, 1, 1, "c.i32_subrelu6_s", "input0"], [202, 1, 1, "c.i32_subrelu6_s", "input1"], [202, 1, 1, "c.i32_subrelu6_s", "output"], [202, 1, 1, "c.i32_subrelu6_s", "size"]], "i32_subrelu_p": [[202, 1, 1, "c.i32_subrelu_p", "input0"], [202, 1, 1, "c.i32_subrelu_p", "input1"], [202, 1, 1, "c.i32_subrelu_p", "output"], [202, 1, 1, "c.i32_subrelu_p", "size"]], "i32_subrelu_s": [[202, 1, 1, "c.i32_subrelu_s", "core_mask"], [202, 1, 1, "c.i32_subrelu_s", "input0"], [202, 1, 1, "c.i32_subrelu_s", "input1"], [202, 1, 1, "c.i32_subrelu_s", "output"], [202, 1, 1, "c.i32_subrelu_s", "size"]], "i32_tensor_scatter_add_p": [[206, 1, 1, "c.i32_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "input"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "output"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.i32_tensor_scatter_add_p", "updates"]], "i32_tensor_scatter_add_s": [[206, 1, 1, "c.i32_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "input"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "output"], [206, 1, 1, "c.i32_tensor_scatter_add_s", "updates"]], "i32_tensorarrayread_p": [[208, 1, 1, "c.i32_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.i32_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.i32_tensorarrayread_p", "index"], [208, 1, 1, "c.i32_tensorarrayread_p", "output_data"], [208, 1, 1, "c.i32_tensorarrayread_p", "output_size"]], "i32_tensorarrayread_s": [[208, 1, 1, "c.i32_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.i32_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.i32_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.i32_tensorarrayread_s", "index"], [208, 1, 1, "c.i32_tensorarrayread_s", "output_data"], [208, 1, 1, "c.i32_tensorarrayread_s", "output_size"]], "i32_tensorlistfromtensor_p": [[210, 1, 1, "c.i32_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.i32_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.i32_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.i32_tensorlistfromtensor_p", "output_tensors"]], "i32_tensorlistfromtensor_s": [[210, 1, 1, "c.i32_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.i32_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.i32_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.i32_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.i32_tensorlistfromtensor_s", "output_tensors"]], "i32_tile_p": [[215, 1, 1, "c.i32_tile_p", "input"], [215, 1, 1, "c.i32_tile_p", "input_shape"], [215, 1, 1, "c.i32_tile_p", "output"], [215, 1, 1, "c.i32_tile_p", "stride"], [215, 1, 1, "c.i32_tile_p", "tile_dim"], [215, 1, 1, "c.i32_tile_p", "tile_num"]], "i32_tile_s": [[215, 1, 1, "c.i32_tile_s", "core_mask"], [215, 1, 1, "c.i32_tile_s", "input"], [215, 1, 1, "c.i32_tile_s", "input_shape"], [215, 1, 1, "c.i32_tile_s", "output"], [215, 1, 1, "c.i32_tile_s", "stride"], [215, 1, 1, "c.i32_tile_s", "tile_dim"], [215, 1, 1, "c.i32_tile_s", "tile_num"]], "i32_topk_fusion_p": [[216, 1, 1, "c.i32_topk_fusion_p", "input"], [216, 1, 1, "c.i32_topk_fusion_p", "output"], [216, 1, 1, "c.i32_topk_fusion_p", "output_index"], [216, 1, 1, "c.i32_topk_fusion_p", "parameter"]], "i32_topk_fusion_s": [[216, 1, 1, "c.i32_topk_fusion_s", "core_mask"], [216, 1, 1, "c.i32_topk_fusion_s", "input"], [216, 1, 1, "c.i32_topk_fusion_s", "output"], [216, 1, 1, "c.i32_topk_fusion_s", "output_index"], [216, 1, 1, "c.i32_topk_fusion_s", "parameter"]], "i32_transpose_p": [[217, 1, 1, "c.i32_transpose_p", "in_data"], [217, 1, 1, "c.i32_transpose_p", "num_axes"], [217, 1, 1, "c.i32_transpose_p", "out_data"], [217, 1, 1, "c.i32_transpose_p", "out_strides"], [217, 1, 1, "c.i32_transpose_p", "output_shape"], [217, 1, 1, "c.i32_transpose_p", "perm"], [217, 1, 1, "c.i32_transpose_p", "strides"]], "i32_transpose_s": [[217, 1, 1, "c.i32_transpose_s", "core_mask"], [217, 1, 1, "c.i32_transpose_s", "in_data"], [217, 1, 1, "c.i32_transpose_s", "num_axes"], [217, 1, 1, "c.i32_transpose_s", "out_data"], [217, 1, 1, "c.i32_transpose_s", "out_strides"], [217, 1, 1, "c.i32_transpose_s", "output_shape"], [217, 1, 1, "c.i32_transpose_s", "perm"], [217, 1, 1, "c.i32_transpose_s", "strides"]], "i32_tril_p": [[218, 1, 1, "c.i32_tril_p", "dst"], [218, 1, 1, "c.i32_tril_p", "height"], [218, 1, 1, "c.i32_tril_p", "k"], [218, 1, 1, "c.i32_tril_p", "out_elems"], [218, 1, 1, "c.i32_tril_p", "src"], [218, 1, 1, "c.i32_tril_p", "width"]], "i32_tril_s": [[218, 1, 1, "c.i32_tril_s", "core_mask"], [218, 1, 1, "c.i32_tril_s", "dst"], [218, 1, 1, "c.i32_tril_s", "height"], [218, 1, 1, "c.i32_tril_s", "k"], [218, 1, 1, "c.i32_tril_s", "out_elems"], [218, 1, 1, "c.i32_tril_s", "src"], [218, 1, 1, "c.i32_tril_s", "width"]], "i32_triu_p": [[219, 1, 1, "c.i32_triu_p", "dst"], [219, 1, 1, "c.i32_triu_p", "height"], [219, 1, 1, "c.i32_triu_p", "k"], [219, 1, 1, "c.i32_triu_p", "out_elems"], [219, 1, 1, "c.i32_triu_p", "src"], [219, 1, 1, "c.i32_triu_p", "width"]], "i32_triu_s": [[219, 1, 1, "c.i32_triu_s", "core_mask"], [219, 1, 1, "c.i32_triu_s", "dst"], [219, 1, 1, "c.i32_triu_s", "height"], [219, 1, 1, "c.i32_triu_s", "k"], [219, 1, 1, "c.i32_triu_s", "out_elems"], [219, 1, 1, "c.i32_triu_s", "src"], [219, 1, 1, "c.i32_triu_s", "width"]], "i32_unsorted_segment_sum_p": [[222, 1, 1, "c.i32_unsorted_segment_sum_p", "dim0"], [222, 1, 1, "c.i32_unsorted_segment_sum_p", "dim1"], [222, 1, 1, "c.i32_unsorted_segment_sum_p", "id_max"], [222, 1, 1, "c.i32_unsorted_segment_sum_p", "index"], [222, 1, 1, "c.i32_unsorted_segment_sum_p", "input"], [222, 1, 1, "c.i32_unsorted_segment_sum_p", "output"]], "i32_unsorted_segment_sum_s": [[222, 1, 1, "c.i32_unsorted_segment_sum_s", "core_mask"], [222, 1, 1, "c.i32_unsorted_segment_sum_s", "dim0"], [222, 1, 1, "c.i32_unsorted_segment_sum_s", "dim1"], [222, 1, 1, "c.i32_unsorted_segment_sum_s", "id_max"], [222, 1, 1, "c.i32_unsorted_segment_sum_s", "index"], [222, 1, 1, "c.i32_unsorted_segment_sum_s", "input"], [222, 1, 1, "c.i32_unsorted_segment_sum_s", "output"]], "i32_where_p": [[225, 1, 1, "c.i32_where_p", "condition"], [225, 1, 1, "c.i32_where_p", "input0"], [225, 1, 1, "c.i32_where_p", "input1"], [225, 1, 1, "c.i32_where_p", "length"], [225, 1, 1, "c.i32_where_p", "output"]], "i32_where_s": [[225, 1, 1, "c.i32_where_s", "condition"], [225, 1, 1, "c.i32_where_s", "core_mask"], [225, 1, 1, "c.i32_where_s", "input0"], [225, 1, 1, "c.i32_where_s", "input1"], [225, 1, 1, "c.i32_where_s", "length"], [225, 1, 1, "c.i32_where_s", "output"]], "i32_zerolike_p": [[226, 1, 1, "c.i32_zerolike_p", "length"], [226, 1, 1, "c.i32_zerolike_p", "output"]], "i32_zerolike_s": [[226, 1, 1, "c.i32_zerolike_s", "core_mask"], [226, 1, 1, "c.i32_zerolike_s", "length"], [226, 1, 1, "c.i32_zerolike_s", "output"]], "i8_Gru_p": [[95, 1, 1, "c.i8_Gru_p", "buffer"], [95, 1, 1, "c.i8_Gru_p", "core_mask"], [95, 1, 1, "c.i8_Gru_p", "gru_param"], [95, 1, 1, "c.i8_Gru_p", "hidden_state"], [95, 1, 1, "c.i8_Gru_p", "input"], [95, 1, 1, "c.i8_Gru_p", "input_bias"], [95, 1, 1, "c.i8_Gru_p", "output"], [95, 1, 1, "c.i8_Gru_p", "state_bias"], [95, 1, 1, "c.i8_Gru_p", "weight_g"], [95, 1, 1, "c.i8_Gru_p", "weight_r"]], "i8_Gru_s": [[95, 1, 1, "c.i8_Gru_s", "buffer"], [95, 1, 1, "c.i8_Gru_s", "core_mask"], [95, 1, 1, "c.i8_Gru_s", "gru_param"], [95, 1, 1, "c.i8_Gru_s", "hidden_state"], [95, 1, 1, "c.i8_Gru_s", "input"], [95, 1, 1, "c.i8_Gru_s", "input_bias"], [95, 1, 1, "c.i8_Gru_s", "output"], [95, 1, 1, "c.i8_Gru_s", "state_bias"], [95, 1, 1, "c.i8_Gru_s", "weight_g"], [95, 1, 1, "c.i8_Gru_s", "weight_r"]], "i8_Unique_p": [[221, 1, 1, "c.i8_Unique_p", "input"], [221, 1, 1, "c.i8_Unique_p", "input_len"], [221, 1, 1, "c.i8_Unique_p", "output0"], [221, 1, 1, "c.i8_Unique_p", "output0_len"]], "i8_Unique_s": [[221, 1, 1, "c.i8_Unique_s", "core_mask"], [221, 1, 1, "c.i8_Unique_s", "input"], [221, 1, 1, "c.i8_Unique_s", "input_len"], [221, 1, 1, "c.i8_Unique_s", "output0"], [221, 1, 1, "c.i8_Unique_s", "output0_len"]], "i8_abs_p": [[10, 1, 1, "c.i8_abs_p", "dst_data"], [10, 1, 1, "c.i8_abs_p", "length"], [10, 1, 1, "c.i8_abs_p", "src_data"]], "i8_abs_s": [[10, 1, 1, "c.i8_abs_s", "core_mask"], [10, 1, 1, "c.i8_abs_s", "dst_data"], [10, 1, 1, "c.i8_abs_s", "length"], [10, 1, 1, "c.i8_abs_s", "src_data"]], "i8_adder_p": [[16, 1, 1, "c.i8_adder_p", "bias"], [16, 1, 1, "c.i8_adder_p", "conv_param"], [16, 1, 1, "c.i8_adder_p", "core_mask"], [16, 1, 1, "c.i8_adder_p", "input_w"], [16, 1, 1, "c.i8_adder_p", "input_x"], [16, 1, 1, "c.i8_adder_p", "out_y"], [16, 1, 1, "c.i8_adder_p", "quant_param"]], "i8_adder_s": [[16, 1, 1, "c.i8_adder_s", "bias"], [16, 1, 1, "c.i8_adder_s", "core_mask"], [16, 1, 1, "c.i8_adder_s", "input_w"], [16, 1, 1, "c.i8_adder_s", "input_x"], [16, 1, 1, "c.i8_adder_s", "out_y"], [16, 1, 1, "c.i8_adder_s", "param"]], "i8_addext_p": [[17, 1, 1, "c.i8_addext_p", "alpha"], [17, 1, 1, "c.i8_addext_p", "in0"], [17, 1, 1, "c.i8_addext_p", "in1"], [17, 1, 1, "c.i8_addext_p", "out"], [17, 1, 1, "c.i8_addext_p", "size"]], "i8_addext_s": [[17, 1, 1, "c.i8_addext_s", "alpha"], [17, 1, 1, "c.i8_addext_s", "core_mask"], [17, 1, 1, "c.i8_addext_s", "in0"], [17, 1, 1, "c.i8_addext_s", "in1"], [17, 1, 1, "c.i8_addext_s", "out"], [17, 1, 1, "c.i8_addext_s", "size"]], "i8_addn_p": [[19, 1, 1, "c.i8_addn_p", "input0"], [19, 1, 1, "c.i8_addn_p", "input1"], [19, 1, 1, "c.i8_addn_p", "length"], [19, 1, 1, "c.i8_addn_p", "output"]], "i8_addn_s": [[19, 1, 1, "c.i8_addn_s", "core_mask"], [19, 1, 1, "c.i8_addn_s", "input0"], [19, 1, 1, "c.i8_addn_s", "input1"], [19, 1, 1, "c.i8_addn_s", "length"], [19, 1, 1, "c.i8_addn_s", "output"]], "i8_addrelu6_p": [[17, 1, 1, "c.i8_addrelu6_p", "in0"], [17, 1, 1, "c.i8_addrelu6_p", "in1"], [17, 1, 1, "c.i8_addrelu6_p", "out"], [17, 1, 1, "c.i8_addrelu6_p", "size"]], "i8_addrelu6_s": [[17, 1, 1, "c.i8_addrelu6_s", "core_mask"], [17, 1, 1, "c.i8_addrelu6_s", "in0"], [17, 1, 1, "c.i8_addrelu6_s", "in1"], [17, 1, 1, "c.i8_addrelu6_s", "out"], [17, 1, 1, "c.i8_addrelu6_s", "size"]], "i8_addrelu_p": [[17, 1, 1, "c.i8_addrelu_p", "in0"], [17, 1, 1, "c.i8_addrelu_p", "in1"], [17, 1, 1, "c.i8_addrelu_p", "out"], [17, 1, 1, "c.i8_addrelu_p", "size"]], "i8_addrelu_s": [[17, 1, 1, "c.i8_addrelu_s", "core_mask"], [17, 1, 1, "c.i8_addrelu_s", "in0"], [17, 1, 1, "c.i8_addrelu_s", "in1"], [17, 1, 1, "c.i8_addrelu_s", "out"], [17, 1, 1, "c.i8_addrelu_s", "size"]], "i8_affine_p": [[20, 1, 1, "c.i8_affine_p", "params"]], "i8_affine_s": [[20, 1, 1, "c.i8_affine_s", "core_mask"], [20, 1, 1, "c.i8_affine_s", "params"]], "i8_allgather_p": [[22, 1, 1, "c.i8_allgather_p", "data_size"], [22, 1, 1, "c.i8_allgather_p", "input"], [22, 1, 1, "c.i8_allgather_p", "input_rank"], [22, 1, 1, "c.i8_allgather_p", "output"], [22, 1, 1, "c.i8_allgather_p", "output_rank"]], "i8_allgather_s": [[22, 1, 1, "c.i8_allgather_s", "core_mask"], [22, 1, 1, "c.i8_allgather_s", "data_size"], [22, 1, 1, "c.i8_allgather_s", "input"], [22, 1, 1, "c.i8_allgather_s", "input_rank"], [22, 1, 1, "c.i8_allgather_s", "output"], [22, 1, 1, "c.i8_allgather_s", "output_rank"]], "i8_and_p": [[112, 1, 1, "c.i8_and_p", "input0"], [112, 1, 1, "c.i8_and_p", "input1"], [112, 1, 1, "c.i8_and_p", "length"], [112, 1, 1, "c.i8_and_p", "output"]], "i8_and_s": [[112, 1, 1, "c.i8_and_s", "core_mask"], [112, 1, 1, "c.i8_and_s", "input0"], [112, 1, 1, "c.i8_and_s", "input1"], [112, 1, 1, "c.i8_and_s", "length"], [112, 1, 1, "c.i8_and_s", "output"]], "i8_assign_p": [[27, 1, 1, "c.i8_assign_p", "dst"], [27, 1, 1, "c.i8_assign_p", "length"], [27, 1, 1, "c.i8_assign_p", "src"]], "i8_assign_s": [[27, 1, 1, "c.i8_assign_s", "core_mask"], [27, 1, 1, "c.i8_assign_s", "dst"], [27, 1, 1, "c.i8_assign_s", "length"], [27, 1, 1, "c.i8_assign_s", "src"]], "i8_assignadd_p": [[28, 1, 1, "c.i8_assignadd_p", "input"], [28, 1, 1, "c.i8_assignadd_p", "length"], [28, 1, 1, "c.i8_assignadd_p", "output"]], "i8_assignadd_s": [[28, 1, 1, "c.i8_assignadd_s", "core_mask"], [28, 1, 1, "c.i8_assignadd_s", "input"], [28, 1, 1, "c.i8_assignadd_s", "length"], [28, 1, 1, "c.i8_assignadd_s", "output"]], "i8_avgpool_fusion_p": [[31, 1, 1, "c.i8_avgpool_fusion_p", "batch"], [31, 1, 1, "c.i8_avgpool_fusion_p", "channel"], [31, 1, 1, "c.i8_avgpool_fusion_p", "in_h"], [31, 1, 1, "c.i8_avgpool_fusion_p", "in_w"], [31, 1, 1, "c.i8_avgpool_fusion_p", "input"], [31, 1, 1, "c.i8_avgpool_fusion_p", "max_val"], [31, 1, 1, "c.i8_avgpool_fusion_p", "min_val"], [31, 1, 1, "c.i8_avgpool_fusion_p", "output"], [31, 1, 1, "c.i8_avgpool_fusion_p", "pad_bottom"], [31, 1, 1, "c.i8_avgpool_fusion_p", "pad_left"], [31, 1, 1, "c.i8_avgpool_fusion_p", "pad_right"], [31, 1, 1, "c.i8_avgpool_fusion_p", "pad_top"], [31, 1, 1, "c.i8_avgpool_fusion_p", "stride_h"], [31, 1, 1, "c.i8_avgpool_fusion_p", "stride_w"], [31, 1, 1, "c.i8_avgpool_fusion_p", "win_h"], [31, 1, 1, "c.i8_avgpool_fusion_p", "win_w"]], "i8_avgpool_fusion_s": [[31, 1, 1, "c.i8_avgpool_fusion_s", "batch"], [31, 1, 1, "c.i8_avgpool_fusion_s", "channel"], [31, 1, 1, "c.i8_avgpool_fusion_s", "core_mask"], [31, 1, 1, "c.i8_avgpool_fusion_s", "in_h"], [31, 1, 1, "c.i8_avgpool_fusion_s", "in_w"], [31, 1, 1, "c.i8_avgpool_fusion_s", "input"], [31, 1, 1, "c.i8_avgpool_fusion_s", "max_val"], [31, 1, 1, "c.i8_avgpool_fusion_s", "min_val"], [31, 1, 1, "c.i8_avgpool_fusion_s", "output"], [31, 1, 1, "c.i8_avgpool_fusion_s", "pad_bottom"], [31, 1, 1, "c.i8_avgpool_fusion_s", "pad_left"], [31, 1, 1, "c.i8_avgpool_fusion_s", "pad_right"], [31, 1, 1, "c.i8_avgpool_fusion_s", "pad_top"], [31, 1, 1, "c.i8_avgpool_fusion_s", "stride_h"], [31, 1, 1, "c.i8_avgpool_fusion_s", "stride_w"], [31, 1, 1, "c.i8_avgpool_fusion_s", "win_h"], [31, 1, 1, "c.i8_avgpool_fusion_s", "win_w"]], "i8_batchnorm_p": [[33, 1, 1, "c.i8_batchnorm_p", "channel"], [33, 1, 1, "c.i8_batchnorm_p", "epsilon"], [33, 1, 1, "c.i8_batchnorm_p", "input"], [33, 1, 1, "c.i8_batchnorm_p", "mean"], [33, 1, 1, "c.i8_batchnorm_p", "output"], [33, 1, 1, "c.i8_batchnorm_p", "unit"], [33, 1, 1, "c.i8_batchnorm_p", "variance"]], "i8_batchnorm_s": [[33, 1, 1, "c.i8_batchnorm_s", "channel"], [33, 1, 1, "c.i8_batchnorm_s", "core_mask"], [33, 1, 1, "c.i8_batchnorm_s", "epsilon"], [33, 1, 1, "c.i8_batchnorm_s", "input"], [33, 1, 1, "c.i8_batchnorm_s", "mean"], [33, 1, 1, "c.i8_batchnorm_s", "output"], [33, 1, 1, "c.i8_batchnorm_s", "unit"], [33, 1, 1, "c.i8_batchnorm_s", "variance"]], "i8_batchtospace_p": [[35, 1, 1, "c.i8_batchtospace_p", "block_size"], [35, 1, 1, "c.i8_batchtospace_p", "crops"], [35, 1, 1, "c.i8_batchtospace_p", "data_size"], [35, 1, 1, "c.i8_batchtospace_p", "input"], [35, 1, 1, "c.i8_batchtospace_p", "input_shape"], [35, 1, 1, "c.i8_batchtospace_p", "output"]], "i8_batchtospace_s": [[35, 1, 1, "c.i8_batchtospace_s", "block_size"], [35, 1, 1, "c.i8_batchtospace_s", "core_mask"], [35, 1, 1, "c.i8_batchtospace_s", "crops"], [35, 1, 1, "c.i8_batchtospace_s", "data_size"], [35, 1, 1, "c.i8_batchtospace_s", "input"], [35, 1, 1, "c.i8_batchtospace_s", "input_shape"], [35, 1, 1, "c.i8_batchtospace_s", "output"]], "i8_batchtospacend_p": [[36, 1, 1, "c.i8_batchtospacend_p", "block_size"], [36, 1, 1, "c.i8_batchtospacend_p", "crops"], [36, 1, 1, "c.i8_batchtospacend_p", "data_size"], [36, 1, 1, "c.i8_batchtospacend_p", "input"], [36, 1, 1, "c.i8_batchtospacend_p", "input_shape"], [36, 1, 1, "c.i8_batchtospacend_p", "output"]], "i8_batchtospacend_s": [[36, 1, 1, "c.i8_batchtospacend_s", "block_size"], [36, 1, 1, "c.i8_batchtospacend_s", "core_mask"], [36, 1, 1, "c.i8_batchtospacend_s", "crops"], [36, 1, 1, "c.i8_batchtospacend_s", "data_size"], [36, 1, 1, "c.i8_batchtospacend_s", "input"], [36, 1, 1, "c.i8_batchtospacend_s", "input_shape"], [36, 1, 1, "c.i8_batchtospacend_s", "output"]], "i8_biasadd_p": [[37, 1, 1, "c.i8_biasadd_p", "data_format"], [37, 1, 1, "c.i8_biasadd_p", "dims"], [37, 1, 1, "c.i8_biasadd_p", "input_bias"], [37, 1, 1, "c.i8_biasadd_p", "input_x"], [37, 1, 1, "c.i8_biasadd_p", "length"], [37, 1, 1, "c.i8_biasadd_p", "output"], [37, 1, 1, "c.i8_biasadd_p", "shape_size"]], "i8_biasadd_s": [[37, 1, 1, "c.i8_biasadd_s", "core_mask"], [37, 1, 1, "c.i8_biasadd_s", "data_format"], [37, 1, 1, "c.i8_biasadd_s", "dims"], [37, 1, 1, "c.i8_biasadd_s", "input_bias"], [37, 1, 1, "c.i8_biasadd_s", "input_x"], [37, 1, 1, "c.i8_biasadd_s", "length"], [37, 1, 1, "c.i8_biasadd_s", "output"], [37, 1, 1, "c.i8_biasadd_s", "shape_size"]], "i8_binarycrossentropy_p": [[39, 1, 1, "c.i8_binarycrossentropy_p", "input_size"], [39, 1, 1, "c.i8_binarycrossentropy_p", "input_x"], [39, 1, 1, "c.i8_binarycrossentropy_p", "input_y"], [39, 1, 1, "c.i8_binarycrossentropy_p", "loss"], [39, 1, 1, "c.i8_binarycrossentropy_p", "reduction"], [39, 1, 1, "c.i8_binarycrossentropy_p", "tmp_loss"], [39, 1, 1, "c.i8_binarycrossentropy_p", "weight"], [39, 1, 1, "c.i8_binarycrossentropy_p", "weight_defined"]], "i8_binarycrossentropy_s": [[39, 1, 1, "c.i8_binarycrossentropy_s", "core_mask"], [39, 1, 1, "c.i8_binarycrossentropy_s", "input_size"], [39, 1, 1, "c.i8_binarycrossentropy_s", "input_x"], [39, 1, 1, "c.i8_binarycrossentropy_s", "input_y"], [39, 1, 1, "c.i8_binarycrossentropy_s", "loss"], [39, 1, 1, "c.i8_binarycrossentropy_s", "reduction"], [39, 1, 1, "c.i8_binarycrossentropy_s", "tmp_loss"], [39, 1, 1, "c.i8_binarycrossentropy_s", "weight"], [39, 1, 1, "c.i8_binarycrossentropy_s", "weight_defined"]], "i8_broadcastto_p": [[41, 1, 1, "c.i8_broadcastto_p", "data_size"], [41, 1, 1, "c.i8_broadcastto_p", "input"], [41, 1, 1, "c.i8_broadcastto_p", "input_shape"], [41, 1, 1, "c.i8_broadcastto_p", "input_shape_size"], [41, 1, 1, "c.i8_broadcastto_p", "output"], [41, 1, 1, "c.i8_broadcastto_p", "output_shape"], [41, 1, 1, "c.i8_broadcastto_p", "output_shape_size"]], "i8_broadcastto_s": [[41, 1, 1, "c.i8_broadcastto_s", "core_mask"], [41, 1, 1, "c.i8_broadcastto_s", "data_size"], [41, 1, 1, "c.i8_broadcastto_s", "input"], [41, 1, 1, "c.i8_broadcastto_s", "input_shape"], [41, 1, 1, "c.i8_broadcastto_s", "input_shape_size"], [41, 1, 1, "c.i8_broadcastto_s", "output"], [41, 1, 1, "c.i8_broadcastto_s", "output_shape"], [41, 1, 1, "c.i8_broadcastto_s", "output_shape_size"]], "i8_celu_p": [[12, 1, 1, "c.i8_celu_p", "Input0"], [12, 1, 1, "c.i8_celu_p", "alpha"], [12, 1, 1, "c.i8_celu_p", "length"], [12, 1, 1, "c.i8_celu_p", "output"]], "i8_celu_s": [[12, 1, 1, "c.i8_celu_s", "Input0"], [12, 1, 1, "c.i8_celu_s", "alpha"], [12, 1, 1, "c.i8_celu_s", "core_mask"], [12, 1, 1, "c.i8_celu_s", "length"], [12, 1, 1, "c.i8_celu_s", "output"]], "i8_clip_p": [[12, 1, 1, "c.i8_clip_p", "Input0"], [12, 1, 1, "c.i8_clip_p", "length"], [12, 1, 1, "c.i8_clip_p", "max_val"], [12, 1, 1, "c.i8_clip_p", "min_val"], [12, 1, 1, "c.i8_clip_p", "output"]], "i8_clip_s": [[12, 1, 1, "c.i8_clip_s", "Input0"], [12, 1, 1, "c.i8_clip_s", "core_mask"], [12, 1, 1, "c.i8_clip_s", "length"], [12, 1, 1, "c.i8_clip_s", "max_val"], [12, 1, 1, "c.i8_clip_s", "min_val"], [12, 1, 1, "c.i8_clip_s", "output"]], "i8_concat_p": [[45, 1, 1, "c.i8_concat_p", "axis"], [45, 1, 1, "c.i8_concat_p", "input_ndim"], [45, 1, 1, "c.i8_concat_p", "input_shapes"], [45, 1, 1, "c.i8_concat_p", "inputs"], [45, 1, 1, "c.i8_concat_p", "num_inputs"], [45, 1, 1, "c.i8_concat_p", "output"]], "i8_concat_s": [[45, 1, 1, "c.i8_concat_s", "axis"], [45, 1, 1, "c.i8_concat_s", "core_mask"], [45, 1, 1, "c.i8_concat_s", "input_ndim"], [45, 1, 1, "c.i8_concat_s", "input_shapes"], [45, 1, 1, "c.i8_concat_s", "inputs"], [45, 1, 1, "c.i8_concat_s", "num_inputs"], [45, 1, 1, "c.i8_concat_s", "output"]], "i8_constant_of_shape_p": [[46, 1, 1, "c.i8_constant_of_shape_p", "end"], [46, 1, 1, "c.i8_constant_of_shape_p", "output"], [46, 1, 1, "c.i8_constant_of_shape_p", "start"], [46, 1, 1, "c.i8_constant_of_shape_p", "value"]], "i8_constant_of_shape_s": [[46, 1, 1, "c.i8_constant_of_shape_s", "core_mask"], [46, 1, 1, "c.i8_constant_of_shape_s", "end"], [46, 1, 1, "c.i8_constant_of_shape_s", "output"], [46, 1, 1, "c.i8_constant_of_shape_s", "start"], [46, 1, 1, "c.i8_constant_of_shape_s", "value"]], "i8_conv2d_p": [[47, 1, 1, "c.i8_conv2d_p", "bias"], [47, 1, 1, "c.i8_conv2d_p", "conv_param"], [47, 1, 1, "c.i8_conv2d_p", "core_mask"], [47, 1, 1, "c.i8_conv2d_p", "input_w"], [47, 1, 1, "c.i8_conv2d_p", "input_x"], [47, 1, 1, "c.i8_conv2d_p", "out_y"], [47, 1, 1, "c.i8_conv2d_p", "quant_param"]], "i8_conv2d_s": [[47, 1, 1, "c.i8_conv2d_s", "bias"], [47, 1, 1, "c.i8_conv2d_s", "conv_param"], [47, 1, 1, "c.i8_conv2d_s", "core_mask"], [47, 1, 1, "c.i8_conv2d_s", "input_w"], [47, 1, 1, "c.i8_conv2d_s", "input_x"], [47, 1, 1, "c.i8_conv2d_s", "out_y"], [47, 1, 1, "c.i8_conv2d_s", "quant_param"]], "i8_convtranspose_p": [[48, 1, 1, "c.i8_convtranspose_p", "bias"], [48, 1, 1, "c.i8_convtranspose_p", "conv_param"], [48, 1, 1, "c.i8_convtranspose_p", "core_mask"], [48, 1, 1, "c.i8_convtranspose_p", "input_w"], [48, 1, 1, "c.i8_convtranspose_p", "input_x"], [48, 1, 1, "c.i8_convtranspose_p", "out_y"]], "i8_convtranspose_s": [[48, 1, 1, "c.i8_convtranspose_s", "bias"], [48, 1, 1, "c.i8_convtranspose_s", "conv_param"], [48, 1, 1, "c.i8_convtranspose_s", "core_mask"], [48, 1, 1, "c.i8_convtranspose_s", "input_w"], [48, 1, 1, "c.i8_convtranspose_s", "input_x"], [48, 1, 1, "c.i8_convtranspose_s", "out_y"]], "i8_cos_p": [[51, 1, 1, "c.i8_cos_p", "dst_data"], [51, 1, 1, "c.i8_cos_p", "length"], [51, 1, 1, "c.i8_cos_p", "src_data"]], "i8_cos_s": [[51, 1, 1, "c.i8_cos_s", "core_mask"], [51, 1, 1, "c.i8_cos_s", "dst_data"], [51, 1, 1, "c.i8_cos_s", "length"], [51, 1, 1, "c.i8_cos_s", "src_data"]], "i8_crop_and_resize_anycore": [[53, 1, 1, "c.i8_crop_and_resize_anycore", "box_idx"], [53, 1, 1, "c.i8_crop_and_resize_anycore", "boxes"], [53, 1, 1, "c.i8_crop_and_resize_anycore", "core_mask"], [53, 1, 1, "c.i8_crop_and_resize_anycore", "dst"], [53, 1, 1, "c.i8_crop_and_resize_anycore", "extrapolation_value"], [53, 1, 1, "c.i8_crop_and_resize_anycore", "param"], [53, 1, 1, "c.i8_crop_and_resize_anycore", "src"]], "i8_cumsum_p": [[54, 1, 1, "c.i8_cumsum_p", "axis_dim"], [54, 1, 1, "c.i8_cumsum_p", "exclusive"], [54, 1, 1, "c.i8_cumsum_p", "inner_dim"], [54, 1, 1, "c.i8_cumsum_p", "input"], [54, 1, 1, "c.i8_cumsum_p", "out_dim"], [54, 1, 1, "c.i8_cumsum_p", "output"]], "i8_cumsum_s": [[54, 1, 1, "c.i8_cumsum_s", "axis_dim"], [54, 1, 1, "c.i8_cumsum_s", "core_mask"], [54, 1, 1, "c.i8_cumsum_s", "exclusive"], [54, 1, 1, "c.i8_cumsum_s", "inner_dim"], [54, 1, 1, "c.i8_cumsum_s", "input"], [54, 1, 1, "c.i8_cumsum_s", "out_dim"], [54, 1, 1, "c.i8_cumsum_s", "output"]], "i8_depthtospace_p": [[59, 1, 1, "c.i8_depthtospace_p", "block_size"], [59, 1, 1, "c.i8_depthtospace_p", "data_size"], [59, 1, 1, "c.i8_depthtospace_p", "in_shape"], [59, 1, 1, "c.i8_depthtospace_p", "input"], [59, 1, 1, "c.i8_depthtospace_p", "output"]], "i8_depthtospace_s": [[59, 1, 1, "c.i8_depthtospace_s", "block_size"], [59, 1, 1, "c.i8_depthtospace_s", "core_mask"], [59, 1, 1, "c.i8_depthtospace_s", "data_size"], [59, 1, 1, "c.i8_depthtospace_s", "in_shape"], [59, 1, 1, "c.i8_depthtospace_s", "input"], [59, 1, 1, "c.i8_depthtospace_s", "output"]], "i8_detection_post_process_p": [[60, 1, 1, "c.i8_detection_post_process_p", "anchors"], [60, 1, 1, "c.i8_detection_post_process_p", "input_boxes"], [60, 1, 1, "c.i8_detection_post_process_p", "input_scores"], [60, 1, 1, "c.i8_detection_post_process_p", "output_boxes"], [60, 1, 1, "c.i8_detection_post_process_p", "output_classes"], [60, 1, 1, "c.i8_detection_post_process_p", "output_num"], [60, 1, 1, "c.i8_detection_post_process_p", "output_scores"], [60, 1, 1, "c.i8_detection_post_process_p", "param"]], "i8_detection_post_process_s": [[60, 1, 1, "c.i8_detection_post_process_s", "anchors"], [60, 1, 1, "c.i8_detection_post_process_s", "core_mask"], [60, 1, 1, "c.i8_detection_post_process_s", "input_boxes"], [60, 1, 1, "c.i8_detection_post_process_s", "input_scores"], [60, 1, 1, "c.i8_detection_post_process_s", "output_boxes"], [60, 1, 1, "c.i8_detection_post_process_s", "output_classes"], [60, 1, 1, "c.i8_detection_post_process_s", "output_num"], [60, 1, 1, "c.i8_detection_post_process_s", "output_scores"], [60, 1, 1, "c.i8_detection_post_process_s", "param"]], "i8_div_fusion_p": [[61, 1, 1, "c.i8_div_fusion_p", "input0"], [61, 1, 1, "c.i8_div_fusion_p", "input1"], [61, 1, 1, "c.i8_div_fusion_p", "length"], [61, 1, 1, "c.i8_div_fusion_p", "output"]], "i8_div_fusion_s": [[61, 1, 1, "c.i8_div_fusion_s", "core_mask"], [61, 1, 1, "c.i8_div_fusion_s", "input0"], [61, 1, 1, "c.i8_div_fusion_s", "input1"], [61, 1, 1, "c.i8_div_fusion_s", "length"], [61, 1, 1, "c.i8_div_fusion_s", "output"]], "i8_eltwise_p": [[67, 1, 1, "c.i8_eltwise_p", "Input0"], [67, 1, 1, "c.i8_eltwise_p", "Input1"], [67, 1, 1, "c.i8_eltwise_p", "eltwise_mode_"], [67, 1, 1, "c.i8_eltwise_p", "length"], [67, 1, 1, "c.i8_eltwise_p", "output"]], "i8_eltwise_s": [[67, 1, 1, "c.i8_eltwise_s", "Input0"], [67, 1, 1, "c.i8_eltwise_s", "Input1"], [67, 1, 1, "c.i8_eltwise_s", "core_mask"], [67, 1, 1, "c.i8_eltwise_s", "eltwise_mode_"], [67, 1, 1, "c.i8_eltwise_s", "length"], [67, 1, 1, "c.i8_eltwise_s", "output"]], "i8_elu_p": [[12, 1, 1, "c.i8_elu_p", "Input0"], [12, 1, 1, "c.i8_elu_p", "alpha"], [12, 1, 1, "c.i8_elu_p", "length"], [12, 1, 1, "c.i8_elu_p", "output"]], "i8_elu_s": [[12, 1, 1, "c.i8_elu_s", "Input0"], [12, 1, 1, "c.i8_elu_s", "alpha"], [12, 1, 1, "c.i8_elu_s", "core_mask"], [12, 1, 1, "c.i8_elu_s", "length"], [12, 1, 1, "c.i8_elu_s", "output"]], "i8_equal_p": [[70, 1, 1, "c.i8_equal_p", "Input0"], [70, 1, 1, "c.i8_equal_p", "Input1"], [70, 1, 1, "c.i8_equal_p", "length"], [70, 1, 1, "c.i8_equal_p", "output"]], "i8_equal_s": [[70, 1, 1, "c.i8_equal_s", "Input0"], [70, 1, 1, "c.i8_equal_s", "Input1"], [70, 1, 1, "c.i8_equal_s", "core_mask"], [70, 1, 1, "c.i8_equal_s", "length"], [70, 1, 1, "c.i8_equal_s", "output"]], "i8_expfusion_p": [[73, 1, 1, "c.i8_expfusion_p", "dst_data"], [73, 1, 1, "c.i8_expfusion_p", "in_scale"], [73, 1, 1, "c.i8_expfusion_p", "length"], [73, 1, 1, "c.i8_expfusion_p", "out_scale"], [73, 1, 1, "c.i8_expfusion_p", "scale"], [73, 1, 1, "c.i8_expfusion_p", "src_data"]], "i8_expfusion_s": [[73, 1, 1, "c.i8_expfusion_s", "core_mask"], [73, 1, 1, "c.i8_expfusion_s", "dst_data"], [73, 1, 1, "c.i8_expfusion_s", "in_scale"], [73, 1, 1, "c.i8_expfusion_s", "length"], [73, 1, 1, "c.i8_expfusion_s", "out_scale"], [73, 1, 1, "c.i8_expfusion_s", "scale"], [73, 1, 1, "c.i8_expfusion_s", "src_data"]], "i8_extract_features_p": [[55, 1, 1, "c.i8_extract_features_p", "num_strings"], [55, 1, 1, "c.i8_extract_features_p", "output_labels"], [55, 1, 1, "c.i8_extract_features_p", "output_weights"], [55, 1, 1, "c.i8_extract_features_p", "string_lengths"], [55, 1, 1, "c.i8_extract_features_p", "string_pointers"]], "i8_extract_features_s": [[55, 1, 1, "c.i8_extract_features_s", "core_mask"], [55, 1, 1, "c.i8_extract_features_s", "num_strings"], [55, 1, 1, "c.i8_extract_features_s", "output_labels"], [55, 1, 1, "c.i8_extract_features_s", "output_weights"], [55, 1, 1, "c.i8_extract_features_s", "string_lengths"], [55, 1, 1, "c.i8_extract_features_s", "string_pointers"]], "i8_fill_p": [[78, 1, 1, "c.i8_fill_p", "output"], [78, 1, 1, "c.i8_fill_p", "param"], [78, 1, 1, "c.i8_fill_p", "value"]], "i8_fill_s": [[78, 1, 1, "c.i8_fill_s", "core_mask"], [78, 1, 1, "c.i8_fill_s", "output"], [78, 1, 1, "c.i8_fill_s", "param"], [78, 1, 1, "c.i8_fill_s", "value"]], "i8_formattranspose_p": [[85, 1, 1, "c.i8_formattranspose_p", "batch"], [85, 1, 1, "c.i8_formattranspose_p", "channel"], [85, 1, 1, "c.i8_formattranspose_p", "dst_data"], [85, 1, 1, "c.i8_formattranspose_p", "dst_format"], [85, 1, 1, "c.i8_formattranspose_p", "plane"], [85, 1, 1, "c.i8_formattranspose_p", "src_data"], [85, 1, 1, "c.i8_formattranspose_p", "src_format"]], "i8_formattranspose_s": [[85, 1, 1, "c.i8_formattranspose_s", "batch"], [85, 1, 1, "c.i8_formattranspose_s", "channel"], [85, 1, 1, "c.i8_formattranspose_s", "core_mask"], [85, 1, 1, "c.i8_formattranspose_s", "dst_data"], [85, 1, 1, "c.i8_formattranspose_s", "dst_format"], [85, 1, 1, "c.i8_formattranspose_s", "plane"], [85, 1, 1, "c.i8_formattranspose_s", "src_data"], [85, 1, 1, "c.i8_formattranspose_s", "src_format"]], "i8_fullconnection_p": [[86, 1, 1, "c.i8_fullconnection_p", "A"], [86, 1, 1, "c.i8_fullconnection_p", "B"], [86, 1, 1, "c.i8_fullconnection_p", "C"], [86, 1, 1, "c.i8_fullconnection_p", "K"], [86, 1, 1, "c.i8_fullconnection_p", "M"], [86, 1, 1, "c.i8_fullconnection_p", "N"], [86, 1, 1, "c.i8_fullconnection_p", "activation_type"], [86, 1, 1, "c.i8_fullconnection_p", "bias"]], "i8_fullconnection_s": [[86, 1, 1, "c.i8_fullconnection_s", "A"], [86, 1, 1, "c.i8_fullconnection_s", "B"], [86, 1, 1, "c.i8_fullconnection_s", "C"], [86, 1, 1, "c.i8_fullconnection_s", "K"], [86, 1, 1, "c.i8_fullconnection_s", "M"], [86, 1, 1, "c.i8_fullconnection_s", "N"], [86, 1, 1, "c.i8_fullconnection_s", "activation_type"], [86, 1, 1, "c.i8_fullconnection_s", "bias"], [86, 1, 1, "c.i8_fullconnection_s", "core_mask"]], "i8_gather_nd_p": [[89, 1, 1, "c.i8_gather_nd_p", "indices"], [89, 1, 1, "c.i8_gather_nd_p", "indices_ndim"], [89, 1, 1, "c.i8_gather_nd_p", "indices_shape"], [89, 1, 1, "c.i8_gather_nd_p", "input"], [89, 1, 1, "c.i8_gather_nd_p", "input_ndim"], [89, 1, 1, "c.i8_gather_nd_p", "input_shape"], [89, 1, 1, "c.i8_gather_nd_p", "output"]], "i8_gather_nd_s": [[89, 1, 1, "c.i8_gather_nd_s", "core_mask"], [89, 1, 1, "c.i8_gather_nd_s", "indices"], [89, 1, 1, "c.i8_gather_nd_s", "indices_ndim"], [89, 1, 1, "c.i8_gather_nd_s", "indices_shape"], [89, 1, 1, "c.i8_gather_nd_s", "input"], [89, 1, 1, "c.i8_gather_nd_s", "input_ndim"], [89, 1, 1, "c.i8_gather_nd_s", "input_shape"], [89, 1, 1, "c.i8_gather_nd_s", "output"]], "i8_gather_p": [[88, 1, 1, "c.i8_gather_p", "axis"], [88, 1, 1, "c.i8_gather_p", "batch_dims"], [88, 1, 1, "c.i8_gather_p", "indices"], [88, 1, 1, "c.i8_gather_p", "indices_ndim"], [88, 1, 1, "c.i8_gather_p", "indices_shape"], [88, 1, 1, "c.i8_gather_p", "input"], [88, 1, 1, "c.i8_gather_p", "input_ndim"], [88, 1, 1, "c.i8_gather_p", "input_shape"], [88, 1, 1, "c.i8_gather_p", "output"]], "i8_gather_s": [[88, 1, 1, "c.i8_gather_s", "axis"], [88, 1, 1, "c.i8_gather_s", "batch_dims"], [88, 1, 1, "c.i8_gather_s", "core_mask"], [88, 1, 1, "c.i8_gather_s", "indices"], [88, 1, 1, "c.i8_gather_s", "indices_ndim"], [88, 1, 1, "c.i8_gather_s", "indices_shape"], [88, 1, 1, "c.i8_gather_s", "input"], [88, 1, 1, "c.i8_gather_s", "input_ndim"], [88, 1, 1, "c.i8_gather_s", "input_shape"], [88, 1, 1, "c.i8_gather_s", "output"]], "i8_gatherd_p": [[90, 1, 1, "c.i8_gatherd_p", "dim"], [90, 1, 1, "c.i8_gatherd_p", "index"], [90, 1, 1, "c.i8_gatherd_p", "index_shape"], [90, 1, 1, "c.i8_gatherd_p", "input_shape"], [90, 1, 1, "c.i8_gatherd_p", "input_shape_size"], [90, 1, 1, "c.i8_gatherd_p", "input_x"], [90, 1, 1, "c.i8_gatherd_p", "output"]], "i8_gatherd_s": [[90, 1, 1, "c.i8_gatherd_s", "core_mask"], [90, 1, 1, "c.i8_gatherd_s", "dim"], [90, 1, 1, "c.i8_gatherd_s", "index"], [90, 1, 1, "c.i8_gatherd_s", "index_shape"], [90, 1, 1, "c.i8_gatherd_s", "input_shape"], [90, 1, 1, "c.i8_gatherd_s", "input_shape_size"], [90, 1, 1, "c.i8_gatherd_s", "input_x"], [90, 1, 1, "c.i8_gatherd_s", "output"]], "i8_gelu_p": [[12, 1, 1, "c.i8_gelu_p", "Input0"], [12, 1, 1, "c.i8_gelu_p", "approximate"], [12, 1, 1, "c.i8_gelu_p", "length"], [12, 1, 1, "c.i8_gelu_p", "output"]], "i8_gelu_s": [[12, 1, 1, "c.i8_gelu_s", "Input0"], [12, 1, 1, "c.i8_gelu_s", "approximate"], [12, 1, 1, "c.i8_gelu_s", "core_mask"], [12, 1, 1, "c.i8_gelu_s", "length"], [12, 1, 1, "c.i8_gelu_s", "output"]], "i8_glu_p": [[91, 1, 1, "c.i8_glu_p", "in_data"], [91, 1, 1, "c.i8_glu_p", "input_shape"], [91, 1, 1, "c.i8_glu_p", "len"], [91, 1, 1, "c.i8_glu_p", "ndim"], [91, 1, 1, "c.i8_glu_p", "num_split"], [91, 1, 1, "c.i8_glu_p", "out_data"], [91, 1, 1, "c.i8_glu_p", "split_data"], [91, 1, 1, "c.i8_glu_p", "split_dim"], [91, 1, 1, "c.i8_glu_p", "split_sizes"], [91, 1, 1, "c.i8_glu_p", "strides"]], "i8_glu_s": [[91, 1, 1, "c.i8_glu_s", "core_mask"], [91, 1, 1, "c.i8_glu_s", "in_data"], [91, 1, 1, "c.i8_glu_s", "input_shape"], [91, 1, 1, "c.i8_glu_s", "len"], [91, 1, 1, "c.i8_glu_s", "ndim"], [91, 1, 1, "c.i8_glu_s", "num_split"], [91, 1, 1, "c.i8_glu_s", "out_data"], [91, 1, 1, "c.i8_glu_s", "split_data"], [91, 1, 1, "c.i8_glu_s", "split_dim"], [91, 1, 1, "c.i8_glu_s", "split_sizes"], [91, 1, 1, "c.i8_glu_s", "strides"]], "i8_greater_p": [[92, 1, 1, "c.i8_greater_p", "element_num"], [92, 1, 1, "c.i8_greater_p", "in_elements_num0"], [92, 1, 1, "c.i8_greater_p", "input1"], [92, 1, 1, "c.i8_greater_p", "input2"], [92, 1, 1, "c.i8_greater_p", "optimize"], [92, 1, 1, "c.i8_greater_p", "output"]], "i8_greater_s": [[92, 1, 1, "c.i8_greater_s", "core_mask"], [92, 1, 1, "c.i8_greater_s", "element_num"], [92, 1, 1, "c.i8_greater_s", "in_elements_num0"], [92, 1, 1, "c.i8_greater_s", "input1"], [92, 1, 1, "c.i8_greater_s", "input2"], [92, 1, 1, "c.i8_greater_s", "optimize"], [92, 1, 1, "c.i8_greater_s", "output"]], "i8_greaterequal_p": [[93, 1, 1, "c.i8_greaterequal_p", "element_num"], [93, 1, 1, "c.i8_greaterequal_p", "in_elements_num0"], [93, 1, 1, "c.i8_greaterequal_p", "input1"], [93, 1, 1, "c.i8_greaterequal_p", "input2"], [93, 1, 1, "c.i8_greaterequal_p", "optimize"], [93, 1, 1, "c.i8_greaterequal_p", "output"]], "i8_greaterequal_s": [[93, 1, 1, "c.i8_greaterequal_s", "core_mask"], [93, 1, 1, "c.i8_greaterequal_s", "element_num"], [93, 1, 1, "c.i8_greaterequal_s", "in_elements_num0"], [93, 1, 1, "c.i8_greaterequal_s", "input1"], [93, 1, 1, "c.i8_greaterequal_s", "input2"], [93, 1, 1, "c.i8_greaterequal_s", "optimize"], [93, 1, 1, "c.i8_greaterequal_s", "output"]], "i8_hardshrink_p": [[12, 1, 1, "c.i8_hardshrink_p", "Input0"], [12, 1, 1, "c.i8_hardshrink_p", "lambd"], [12, 1, 1, "c.i8_hardshrink_p", "length"], [12, 1, 1, "c.i8_hardshrink_p", "output"]], "i8_hardshrink_s": [[12, 1, 1, "c.i8_hardshrink_s", "Input0"], [12, 1, 1, "c.i8_hardshrink_s", "core_mask"], [12, 1, 1, "c.i8_hardshrink_s", "lambd"], [12, 1, 1, "c.i8_hardshrink_s", "length"], [12, 1, 1, "c.i8_hardshrink_s", "output"]], "i8_hardtanh_p": [[12, 1, 1, "c.i8_hardtanh_p", "Input0"], [12, 1, 1, "c.i8_hardtanh_p", "length"], [12, 1, 1, "c.i8_hardtanh_p", "max_val"], [12, 1, 1, "c.i8_hardtanh_p", "min_val"], [12, 1, 1, "c.i8_hardtanh_p", "output"]], "i8_hardtanh_s": [[12, 1, 1, "c.i8_hardtanh_s", "Input0"], [12, 1, 1, "c.i8_hardtanh_s", "core_mask"], [12, 1, 1, "c.i8_hardtanh_s", "length"], [12, 1, 1, "c.i8_hardtanh_s", "max_val"], [12, 1, 1, "c.i8_hardtanh_s", "min_val"], [12, 1, 1, "c.i8_hardtanh_s", "output"]], "i8_hsigmoid_p": [[12, 1, 1, "c.i8_hsigmoid_p", "Input0"], [12, 1, 1, "c.i8_hsigmoid_p", "length"], [12, 1, 1, "c.i8_hsigmoid_p", "output"]], "i8_hsigmoid_s": [[12, 1, 1, "c.i8_hsigmoid_s", "Input0"], [12, 1, 1, "c.i8_hsigmoid_s", "core_mask"], [12, 1, 1, "c.i8_hsigmoid_s", "length"], [12, 1, 1, "c.i8_hsigmoid_s", "output"]], "i8_hswish_p": [[12, 1, 1, "c.i8_hswish_p", "Input0"], [12, 1, 1, "c.i8_hswish_p", "length"], [12, 1, 1, "c.i8_hswish_p", "output"]], "i8_hswish_s": [[12, 1, 1, "c.i8_hswish_s", "Input0"], [12, 1, 1, "c.i8_hswish_s", "core_mask"], [12, 1, 1, "c.i8_hswish_s", "length"], [12, 1, 1, "c.i8_hswish_s", "output"]], "i8_invertpermutation_p": [[98, 1, 1, "c.i8_invertpermutation_p", "input"], [98, 1, 1, "c.i8_invertpermutation_p", "num"], [98, 1, 1, "c.i8_invertpermutation_p", "output"]], "i8_invertpermutation_s": [[98, 1, 1, "c.i8_invertpermutation_s", "core_mask"], [98, 1, 1, "c.i8_invertpermutation_s", "input"], [98, 1, 1, "c.i8_invertpermutation_s", "num"], [98, 1, 1, "c.i8_invertpermutation_s", "output"]], "i8_isfinite_p": [[99, 1, 1, "c.i8_isfinite_p", "Input"], [99, 1, 1, "c.i8_isfinite_p", "length"], [99, 1, 1, "c.i8_isfinite_p", "output"]], "i8_isfinite_s": [[99, 1, 1, "c.i8_isfinite_s", "Input"], [99, 1, 1, "c.i8_isfinite_s", "core_mask"], [99, 1, 1, "c.i8_isfinite_s", "length"], [99, 1, 1, "c.i8_isfinite_s", "output"]], "i8_layernormfusion_p": [[101, 1, 1, "c.i8_layernormfusion_p", "beta_data"], [101, 1, 1, "c.i8_layernormfusion_p", "dst_data"], [101, 1, 1, "c.i8_layernormfusion_p", "epsilon"], [101, 1, 1, "c.i8_layernormfusion_p", "gamma_data"], [101, 1, 1, "c.i8_layernormfusion_p", "norm_inner_size"], [101, 1, 1, "c.i8_layernormfusion_p", "norm_outer_size"], [101, 1, 1, "c.i8_layernormfusion_p", "out_mean"], [101, 1, 1, "c.i8_layernormfusion_p", "out_variance"], [101, 1, 1, "c.i8_layernormfusion_p", "param_inner_size"], [101, 1, 1, "c.i8_layernormfusion_p", "param_outer_size"], [101, 1, 1, "c.i8_layernormfusion_p", "src_data"]], "i8_layernormfusion_s": [[101, 1, 1, "c.i8_layernormfusion_s", "beta_data"], [101, 1, 1, "c.i8_layernormfusion_s", "core_mask"], [101, 1, 1, "c.i8_layernormfusion_s", "dst_data"], [101, 1, 1, "c.i8_layernormfusion_s", "epsilon"], [101, 1, 1, "c.i8_layernormfusion_s", "gamma_data"], [101, 1, 1, "c.i8_layernormfusion_s", "length"], [101, 1, 1, "c.i8_layernormfusion_s", "norm_inner_size"], [101, 1, 1, "c.i8_layernormfusion_s", "norm_outer_size"], [101, 1, 1, "c.i8_layernormfusion_s", "out_mean"], [101, 1, 1, "c.i8_layernormfusion_s", "out_variance"], [101, 1, 1, "c.i8_layernormfusion_s", "param_inner_size"], [101, 1, 1, "c.i8_layernormfusion_s", "param_outer_size"], [101, 1, 1, "c.i8_layernormfusion_s", "src_data"]], "i8_leaky_relu_p": [[103, 1, 1, "c.i8_leaky_relu_p", "alpha"], [103, 1, 1, "c.i8_leaky_relu_p", "core_mask"], [103, 1, 1, "c.i8_leaky_relu_p", "elem_cnt"], [103, 1, 1, "c.i8_leaky_relu_p", "input"], [103, 1, 1, "c.i8_leaky_relu_p", "output"]], "i8_leaky_relu_s": [[103, 1, 1, "c.i8_leaky_relu_s", "alpha"], [103, 1, 1, "c.i8_leaky_relu_s", "core_mask"], [103, 1, 1, "c.i8_leaky_relu_s", "elem_cnt"], [103, 1, 1, "c.i8_leaky_relu_s", "input"], [103, 1, 1, "c.i8_leaky_relu_s", "output"]], "i8_less_p": [[104, 1, 1, "c.i8_less_p", "Input0"], [104, 1, 1, "c.i8_less_p", "Input1"], [104, 1, 1, "c.i8_less_p", "in_elements_num0"], [104, 1, 1, "c.i8_less_p", "length"], [104, 1, 1, "c.i8_less_p", "optimize"], [104, 1, 1, "c.i8_less_p", "output"]], "i8_less_s": [[104, 1, 1, "c.i8_less_s", "Input0"], [104, 1, 1, "c.i8_less_s", "Input1"], [104, 1, 1, "c.i8_less_s", "core_mask"], [104, 1, 1, "c.i8_less_s", "in_elements_num0"], [104, 1, 1, "c.i8_less_s", "length"], [104, 1, 1, "c.i8_less_s", "optimize"], [104, 1, 1, "c.i8_less_s", "output"]], "i8_lessequal_p": [[105, 1, 1, "c.i8_lessequal_p", "Input0"], [105, 1, 1, "c.i8_lessequal_p", "Input1"], [105, 1, 1, "c.i8_lessequal_p", "in_elements_num0"], [105, 1, 1, "c.i8_lessequal_p", "length"], [105, 1, 1, "c.i8_lessequal_p", "optimize"], [105, 1, 1, "c.i8_lessequal_p", "output"]], "i8_lessequal_s": [[105, 1, 1, "c.i8_lessequal_s", "Input0"], [105, 1, 1, "c.i8_lessequal_s", "Input1"], [105, 1, 1, "c.i8_lessequal_s", "core_mask"], [105, 1, 1, "c.i8_lessequal_s", "in_elements_num0"], [105, 1, 1, "c.i8_lessequal_s", "length"], [105, 1, 1, "c.i8_lessequal_s", "optimize"], [105, 1, 1, "c.i8_lessequal_s", "output"]], "i8_log1p_p": [[108, 1, 1, "c.i8_log1p_p", "Input"], [108, 1, 1, "c.i8_log1p_p", "length"], [108, 1, 1, "c.i8_log1p_p", "output"]], "i8_log1p_s": [[108, 1, 1, "c.i8_log1p_s", "Input"], [108, 1, 1, "c.i8_log1p_s", "core_mask"], [108, 1, 1, "c.i8_log1p_s", "length"], [108, 1, 1, "c.i8_log1p_s", "output"]], "i8_logical_not_p": [[110, 1, 1, "c.i8_logical_not_p", "input"], [110, 1, 1, "c.i8_logical_not_p", "length"], [110, 1, 1, "c.i8_logical_not_p", "output"]], "i8_logical_not_s": [[110, 1, 1, "c.i8_logical_not_s", "core_mask"], [110, 1, 1, "c.i8_logical_not_s", "input"], [110, 1, 1, "c.i8_logical_not_s", "length"], [110, 1, 1, "c.i8_logical_not_s", "output"]], "i8_logical_or_p": [[111, 1, 1, "c.i8_logical_or_p", "input0"], [111, 1, 1, "c.i8_logical_or_p", "input1"], [111, 1, 1, "c.i8_logical_or_p", "length"], [111, 1, 1, "c.i8_logical_or_p", "output"]], "i8_logical_or_s": [[111, 1, 1, "c.i8_logical_or_s", "core_mask"], [111, 1, 1, "c.i8_logical_or_s", "input0"], [111, 1, 1, "c.i8_logical_or_s", "input1"], [111, 1, 1, "c.i8_logical_or_s", "length"], [111, 1, 1, "c.i8_logical_or_s", "output"]], "i8_logsoftmax_p": [[113, 1, 1, "c.i8_logsoftmax_p", "axis"], [113, 1, 1, "c.i8_logsoftmax_p", "axis_size"], [113, 1, 1, "c.i8_logsoftmax_p", "inner_size"], [113, 1, 1, "c.i8_logsoftmax_p", "input_ptr"], [113, 1, 1, "c.i8_logsoftmax_p", "n_dim"], [113, 1, 1, "c.i8_logsoftmax_p", "output_ptr"], [113, 1, 1, "c.i8_logsoftmax_p", "outter_size"], [113, 1, 1, "c.i8_logsoftmax_p", "sum_data"]], "i8_logsoftmax_s": [[113, 1, 1, "c.i8_logsoftmax_s", "axis"], [113, 1, 1, "c.i8_logsoftmax_s", "axis_size"], [113, 1, 1, "c.i8_logsoftmax_s", "core_mask"], [113, 1, 1, "c.i8_logsoftmax_s", "inner_size"], [113, 1, 1, "c.i8_logsoftmax_s", "input_ptr"], [113, 1, 1, "c.i8_logsoftmax_s", "n_dim"], [113, 1, 1, "c.i8_logsoftmax_s", "output_ptr"], [113, 1, 1, "c.i8_logsoftmax_s", "outter_size"], [113, 1, 1, "c.i8_logsoftmax_s", "sum_data"]], "i8_lrelu_p": [[12, 1, 1, "c.i8_lrelu_p", "Input0"], [12, 1, 1, "c.i8_lrelu_p", "alpha"], [12, 1, 1, "c.i8_lrelu_p", "length"], [12, 1, 1, "c.i8_lrelu_p", "output"]], "i8_lrelu_s": [[12, 1, 1, "c.i8_lrelu_s", "Input0"], [12, 1, 1, "c.i8_lrelu_s", "alpha"], [12, 1, 1, "c.i8_lrelu_s", "core_mask"], [12, 1, 1, "c.i8_lrelu_s", "length"], [12, 1, 1, "c.i8_lrelu_s", "output"]], "i8_lsh_projection_p": [[116, 1, 1, "c.i8_lsh_projection_p", "bits_per_hash"], [116, 1, 1, "c.i8_lsh_projection_p", "feature"], [116, 1, 1, "c.i8_lsh_projection_p", "feature_num"], [116, 1, 1, "c.i8_lsh_projection_p", "hash_group_num"], [116, 1, 1, "c.i8_lsh_projection_p", "hash_seed"], [116, 1, 1, "c.i8_lsh_projection_p", "output"], [116, 1, 1, "c.i8_lsh_projection_p", "weight"]], "i8_lsh_projection_s": [[116, 1, 1, "c.i8_lsh_projection_s", "bits_per_hash"], [116, 1, 1, "c.i8_lsh_projection_s", "core_mask"], [116, 1, 1, "c.i8_lsh_projection_s", "feature"], [116, 1, 1, "c.i8_lsh_projection_s", "feature_num"], [116, 1, 1, "c.i8_lsh_projection_s", "hash_group_num"], [116, 1, 1, "c.i8_lsh_projection_s", "hash_seed"], [116, 1, 1, "c.i8_lsh_projection_s", "output"], [116, 1, 1, "c.i8_lsh_projection_s", "weight"]], "i8_maximum_p": [[122, 1, 1, "c.i8_maximum_p", "input0"], [122, 1, 1, "c.i8_maximum_p", "input1"], [122, 1, 1, "c.i8_maximum_p", "length"], [122, 1, 1, "c.i8_maximum_p", "output"]], "i8_maximum_s": [[122, 1, 1, "c.i8_maximum_s", "core_mask"], [122, 1, 1, "c.i8_maximum_s", "input0"], [122, 1, 1, "c.i8_maximum_s", "input1"], [122, 1, 1, "c.i8_maximum_s", "length"], [122, 1, 1, "c.i8_maximum_s", "output"]], "i8_minimum_p": [[127, 1, 1, "c.i8_minimum_p", "input0"], [127, 1, 1, "c.i8_minimum_p", "input1"], [127, 1, 1, "c.i8_minimum_p", "length"], [127, 1, 1, "c.i8_minimum_p", "output"]], "i8_minimum_s": [[127, 1, 1, "c.i8_minimum_s", "core_mask"], [127, 1, 1, "c.i8_minimum_s", "input0"], [127, 1, 1, "c.i8_minimum_s", "input1"], [127, 1, 1, "c.i8_minimum_s", "length"], [127, 1, 1, "c.i8_minimum_s", "output"]], "i8_mod_p": [[129, 1, 1, "c.i8_mod_p", "input0"], [129, 1, 1, "c.i8_mod_p", "input1"], [129, 1, 1, "c.i8_mod_p", "length"], [129, 1, 1, "c.i8_mod_p", "output"]], "i8_mod_s": [[129, 1, 1, "c.i8_mod_s", "core_mask"], [129, 1, 1, "c.i8_mod_s", "input0"], [129, 1, 1, "c.i8_mod_s", "input1"], [129, 1, 1, "c.i8_mod_s", "length"], [129, 1, 1, "c.i8_mod_s", "output"]], "i8_mul_p": [[130, 1, 1, "c.i8_mul_p", "input0"], [130, 1, 1, "c.i8_mul_p", "input1"], [130, 1, 1, "c.i8_mul_p", "length"], [130, 1, 1, "c.i8_mul_p", "output"]], "i8_mul_s": [[130, 1, 1, "c.i8_mul_s", "core_mask"], [130, 1, 1, "c.i8_mul_s", "input0"], [130, 1, 1, "c.i8_mul_s", "input1"], [130, 1, 1, "c.i8_mul_s", "length"], [130, 1, 1, "c.i8_mul_s", "output"]], "i8_neg_grad_p": [[133, 1, 1, "c.i8_neg_grad_p", "Input"], [133, 1, 1, "c.i8_neg_grad_p", "length"], [133, 1, 1, "c.i8_neg_grad_p", "output"]], "i8_neg_grad_s": [[133, 1, 1, "c.i8_neg_grad_s", "Input"], [133, 1, 1, "c.i8_neg_grad_s", "core_mask"], [133, 1, 1, "c.i8_neg_grad_s", "length"], [133, 1, 1, "c.i8_neg_grad_s", "output"]], "i8_neg_p": [[132, 1, 1, "c.i8_neg_p", "Input"], [132, 1, 1, "c.i8_neg_p", "length"], [132, 1, 1, "c.i8_neg_p", "output"]], "i8_neg_s": [[132, 1, 1, "c.i8_neg_s", "Input"], [132, 1, 1, "c.i8_neg_s", "core_mask"], [132, 1, 1, "c.i8_neg_s", "length"], [132, 1, 1, "c.i8_neg_s", "output"]], "i8_nllloss_p": [[134, 1, 1, "c.i8_nllloss_p", "batch_size"], [134, 1, 1, "c.i8_nllloss_p", "class_num"], [134, 1, 1, "c.i8_nllloss_p", "labels"], [134, 1, 1, "c.i8_nllloss_p", "log_probs"], [134, 1, 1, "c.i8_nllloss_p", "loss"], [134, 1, 1, "c.i8_nllloss_p", "reduction_type"], [134, 1, 1, "c.i8_nllloss_p", "total_weight"], [134, 1, 1, "c.i8_nllloss_p", "weight"]], "i8_nllloss_s": [[134, 1, 1, "c.i8_nllloss_s", "batch_size"], [134, 1, 1, "c.i8_nllloss_s", "class_num"], [134, 1, 1, "c.i8_nllloss_s", "core_mask"], [134, 1, 1, "c.i8_nllloss_s", "labels"], [134, 1, 1, "c.i8_nllloss_s", "log_probs"], [134, 1, 1, "c.i8_nllloss_s", "loss"], [134, 1, 1, "c.i8_nllloss_s", "reduction_type"], [134, 1, 1, "c.i8_nllloss_s", "total_weight"], [134, 1, 1, "c.i8_nllloss_s", "weight"]], "i8_non_max_suppression_p": [[136, 1, 1, "c.i8_non_max_suppression_p", "param"]], "i8_non_max_suppression_s": [[136, 1, 1, "c.i8_non_max_suppression_s", "core_mask"], [136, 1, 1, "c.i8_non_max_suppression_s", "param"]], "i8_nonzero_p": [[137, 1, 1, "c.i8_nonzero_p", "dim_strides"], [137, 1, 1, "c.i8_nonzero_p", "input"], [137, 1, 1, "c.i8_nonzero_p", "input_rank"], [137, 1, 1, "c.i8_nonzero_p", "length"], [137, 1, 1, "c.i8_nonzero_p", "non_zero_num"], [137, 1, 1, "c.i8_nonzero_p", "output"], [137, 1, 1, "c.i8_nonzero_p", "shape"]], "i8_nonzero_s": [[137, 1, 1, "c.i8_nonzero_s", "core_mask"], [137, 1, 1, "c.i8_nonzero_s", "dim_strides"], [137, 1, 1, "c.i8_nonzero_s", "input"], [137, 1, 1, "c.i8_nonzero_s", "input_rank"], [137, 1, 1, "c.i8_nonzero_s", "length"], [137, 1, 1, "c.i8_nonzero_s", "non_zero_num"], [137, 1, 1, "c.i8_nonzero_s", "output"], [137, 1, 1, "c.i8_nonzero_s", "shape"]], "i8_not_equal_p": [[138, 1, 1, "c.i8_not_equal_p", "Input0"], [138, 1, 1, "c.i8_not_equal_p", "Input1"], [138, 1, 1, "c.i8_not_equal_p", "length"], [138, 1, 1, "c.i8_not_equal_p", "output"]], "i8_not_equal_s": [[138, 1, 1, "c.i8_not_equal_s", "Input0"], [138, 1, 1, "c.i8_not_equal_s", "Input1"], [138, 1, 1, "c.i8_not_equal_s", "core_mask"], [138, 1, 1, "c.i8_not_equal_s", "length"], [138, 1, 1, "c.i8_not_equal_s", "output"]], "i8_onehot_p": [[139, 1, 1, "c.i8_onehot_p", "axis"], [139, 1, 1, "c.i8_onehot_p", "depth"], [139, 1, 1, "c.i8_onehot_p", "indices"], [139, 1, 1, "c.i8_onehot_p", "indices_shape"], [139, 1, 1, "c.i8_onehot_p", "indices_shape_size"], [139, 1, 1, "c.i8_onehot_p", "on_off"], [139, 1, 1, "c.i8_onehot_p", "output"], [139, 1, 1, "c.i8_onehot_p", "support_neg_index"]], "i8_onehot_s": [[139, 1, 1, "c.i8_onehot_s", "axis"], [139, 1, 1, "c.i8_onehot_s", "core_mask"], [139, 1, 1, "c.i8_onehot_s", "depth"], [139, 1, 1, "c.i8_onehot_s", "indices"], [139, 1, 1, "c.i8_onehot_s", "indices_shape"], [139, 1, 1, "c.i8_onehot_s", "indices_shape_size"], [139, 1, 1, "c.i8_onehot_s", "on_off"], [139, 1, 1, "c.i8_onehot_s", "output"], [139, 1, 1, "c.i8_onehot_s", "support_neg_index"]], "i8_ones_like_p": [[140, 1, 1, "c.i8_ones_like_p", "length"], [140, 1, 1, "c.i8_ones_like_p", "output"]], "i8_ones_like_s": [[140, 1, 1, "c.i8_ones_like_s", "core_mask"], [140, 1, 1, "c.i8_ones_like_s", "length"], [140, 1, 1, "c.i8_ones_like_s", "output"]], "i8_padfusion_p": [[141, 1, 1, "c.i8_padfusion_p", "params"]], "i8_padfusion_s": [[141, 1, 1, "c.i8_padfusion_s", "core_mask"], [141, 1, 1, "c.i8_padfusion_s", "params"]], "i8_pow_fusion_p": [[142, 1, 1, "c.i8_pow_fusion_p", "Input"], [142, 1, 1, "c.i8_pow_fusion_p", "broadcast"], [142, 1, 1, "c.i8_pow_fusion_p", "exponent"], [142, 1, 1, "c.i8_pow_fusion_p", "length_in"], [142, 1, 1, "c.i8_pow_fusion_p", "output"], [142, 1, 1, "c.i8_pow_fusion_p", "scale"], [142, 1, 1, "c.i8_pow_fusion_p", "shift"]], "i8_pow_fusion_s": [[142, 1, 1, "c.i8_pow_fusion_s", "Input"], [142, 1, 1, "c.i8_pow_fusion_s", "broadcast"], [142, 1, 1, "c.i8_pow_fusion_s", "core_mask"], [142, 1, 1, "c.i8_pow_fusion_s", "exponent"], [142, 1, 1, "c.i8_pow_fusion_s", "length_in"], [142, 1, 1, "c.i8_pow_fusion_s", "output"], [142, 1, 1, "c.i8_pow_fusion_s", "scale"], [142, 1, 1, "c.i8_pow_fusion_s", "shift"]], "i8_prelufusion_p": [[144, 1, 1, "c.i8_prelufusion_p", "dst_data"], [144, 1, 1, "c.i8_prelufusion_p", "end"], [144, 1, 1, "c.i8_prelufusion_p", "slope"], [144, 1, 1, "c.i8_prelufusion_p", "src_data"], [144, 1, 1, "c.i8_prelufusion_p", "start"]], "i8_prelufusion_s": [[144, 1, 1, "c.i8_prelufusion_s", "core_mask"], [144, 1, 1, "c.i8_prelufusion_s", "dst_data"], [144, 1, 1, "c.i8_prelufusion_s", "end"], [144, 1, 1, "c.i8_prelufusion_s", "slope"], [144, 1, 1, "c.i8_prelufusion_s", "src_data"], [144, 1, 1, "c.i8_prelufusion_s", "start"]], "i8_raggedrange_p": [[147, 1, 1, "c.i8_raggedrange_p", "deltas"], [147, 1, 1, "c.i8_raggedrange_p", "limits"], [147, 1, 1, "c.i8_raggedrange_p", "range_count"], [147, 1, 1, "c.i8_raggedrange_p", "splits"], [147, 1, 1, "c.i8_raggedrange_p", "starts"], [147, 1, 1, "c.i8_raggedrange_p", "values"]], "i8_raggedrange_s": [[147, 1, 1, "c.i8_raggedrange_s", "core_mask"], [147, 1, 1, "c.i8_raggedrange_s", "deltas"], [147, 1, 1, "c.i8_raggedrange_s", "limits"], [147, 1, 1, "c.i8_raggedrange_s", "range_count"], [147, 1, 1, "c.i8_raggedrange_s", "splits"], [147, 1, 1, "c.i8_raggedrange_s", "starts"], [147, 1, 1, "c.i8_raggedrange_s", "values"]], "i8_range_p": [[150, 1, 1, "c.i8_range_p", "delta"], [150, 1, 1, "c.i8_range_p", "length"], [150, 1, 1, "c.i8_range_p", "output"], [150, 1, 1, "c.i8_range_p", "start"]], "i8_range_s": [[150, 1, 1, "c.i8_range_s", "core_mask"], [150, 1, 1, "c.i8_range_s", "delta"], [150, 1, 1, "c.i8_range_s", "length"], [150, 1, 1, "c.i8_range_s", "output"], [150, 1, 1, "c.i8_range_s", "start"]], "i8_real_div_p": [[152, 1, 1, "c.i8_real_div_p", "input0"], [152, 1, 1, "c.i8_real_div_p", "input1"], [152, 1, 1, "c.i8_real_div_p", "length"], [152, 1, 1, "c.i8_real_div_p", "output"]], "i8_real_div_s": [[152, 1, 1, "c.i8_real_div_s", "core_mask"], [152, 1, 1, "c.i8_real_div_s", "input0"], [152, 1, 1, "c.i8_real_div_s", "input1"], [152, 1, 1, "c.i8_real_div_s", "length"], [152, 1, 1, "c.i8_real_div_s", "output"]], "i8_reciprocal_p": [[153, 1, 1, "c.i8_reciprocal_p", "Input"], [153, 1, 1, "c.i8_reciprocal_p", "length"], [153, 1, 1, "c.i8_reciprocal_p", "output"]], "i8_reciprocal_s": [[153, 1, 1, "c.i8_reciprocal_s", "Input"], [153, 1, 1, "c.i8_reciprocal_s", "core_mask"], [153, 1, 1, "c.i8_reciprocal_s", "length"], [153, 1, 1, "c.i8_reciprocal_s", "output"]], "i8_reduce_p": [[154, 1, 1, "c.i8_reduce_p", "core_mask"], [154, 1, 1, "c.i8_reduce_p", "dst_data"], [154, 1, 1, "c.i8_reduce_p", "param"], [154, 1, 1, "c.i8_reduce_p", "src_data"], [154, 1, 1, "c.i8_reduce_p", "tmp_dst_data"], [154, 1, 1, "c.i8_reduce_p", "tmp_src_data"]], "i8_reduce_s": [[154, 1, 1, "c.i8_reduce_s", "core_mask"], [154, 1, 1, "c.i8_reduce_s", "dst_data"], [154, 1, 1, "c.i8_reduce_s", "param"], [154, 1, 1, "c.i8_reduce_s", "src_data"]], "i8_reduceall_p": [[21, 1, 1, "c.i8_reduceall_p", "axis_size"], [21, 1, 1, "c.i8_reduceall_p", "dst_data"], [21, 1, 1, "c.i8_reduceall_p", "inner_size"], [21, 1, 1, "c.i8_reduceall_p", "outer_size"], [21, 1, 1, "c.i8_reduceall_p", "src_data"]], "i8_reduceall_s": [[21, 1, 1, "c.i8_reduceall_s", "axis_size"], [21, 1, 1, "c.i8_reduceall_s", "core_mask"], [21, 1, 1, "c.i8_reduceall_s", "dst_data"], [21, 1, 1, "c.i8_reduceall_s", "inner_size"], [21, 1, 1, "c.i8_reduceall_s", "outer_size"], [21, 1, 1, "c.i8_reduceall_s", "src_data"]], "i8_reducescatter_p": [[155, 1, 1, "c.i8_reducescatter_p", "data_size"], [155, 1, 1, "c.i8_reducescatter_p", "input_data"], [155, 1, 1, "c.i8_reducescatter_p", "output_data"], [155, 1, 1, "c.i8_reducescatter_p", "reduce_type"]], "i8_reducescatter_s": [[155, 1, 1, "c.i8_reducescatter_s", "core_mask"], [155, 1, 1, "c.i8_reducescatter_s", "data_size"], [155, 1, 1, "c.i8_reducescatter_s", "input_data"], [155, 1, 1, "c.i8_reducescatter_s", "output_data"], [155, 1, 1, "c.i8_reducescatter_s", "reduce_type"]], "i8_relu6_p": [[12, 1, 1, "c.i8_relu6_p", "Input0"], [12, 1, 1, "c.i8_relu6_p", "length"], [12, 1, 1, "c.i8_relu6_p", "output"]], "i8_relu6_s": [[12, 1, 1, "c.i8_relu6_s", "Input0"], [12, 1, 1, "c.i8_relu6_s", "core_mask"], [12, 1, 1, "c.i8_relu6_s", "length"], [12, 1, 1, "c.i8_relu6_s", "output"]], "i8_relu_p": [[12, 1, 1, "c.i8_relu_p", "Input0"], [12, 1, 1, "c.i8_relu_p", "length"], [12, 1, 1, "c.i8_relu_p", "output"]], "i8_relu_s": [[12, 1, 1, "c.i8_relu_s", "Input0"], [12, 1, 1, "c.i8_relu_s", "core_mask"], [12, 1, 1, "c.i8_relu_s", "length"], [12, 1, 1, "c.i8_relu_s", "output"]], "i8_reshape_p": [[156, 1, 1, "c.i8_reshape_p", "input"], [156, 1, 1, "c.i8_reshape_p", "length"], [156, 1, 1, "c.i8_reshape_p", "output"]], "i8_reshape_s": [[156, 1, 1, "c.i8_reshape_s", "core_mask"], [156, 1, 1, "c.i8_reshape_s", "input"], [156, 1, 1, "c.i8_reshape_s", "length"], [156, 1, 1, "c.i8_reshape_s", "output"]], "i8_resize_anycore": [[157, 1, 1, "c.i8_resize_anycore", "core_mask"], [157, 1, 1, "c.i8_resize_anycore", "input"], [157, 1, 1, "c.i8_resize_anycore", "output"], [157, 1, 1, "c.i8_resize_anycore", "param"]], "i8_roipooling_p": [[162, 1, 1, "c.i8_roipooling_p", "in_ptr"], [162, 1, 1, "c.i8_roipooling_p", "input_c"], [162, 1, 1, "c.i8_roipooling_p", "input_h"], [162, 1, 1, "c.i8_roipooling_p", "input_n"], [162, 1, 1, "c.i8_roipooling_p", "input_w"], [162, 1, 1, "c.i8_roipooling_p", "max_c"], [162, 1, 1, "c.i8_roipooling_p", "num_rois"], [162, 1, 1, "c.i8_roipooling_p", "out_ptr"], [162, 1, 1, "c.i8_roipooling_p", "pooled_height"], [162, 1, 1, "c.i8_roipooling_p", "pooled_width"], [162, 1, 1, "c.i8_roipooling_p", "roi"], [162, 1, 1, "c.i8_roipooling_p", "scale"]], "i8_roipooling_s": [[162, 1, 1, "c.i8_roipooling_s", "core_mask"], [162, 1, 1, "c.i8_roipooling_s", "in_ptr"], [162, 1, 1, "c.i8_roipooling_s", "input_c"], [162, 1, 1, "c.i8_roipooling_s", "input_h"], [162, 1, 1, "c.i8_roipooling_s", "input_n"], [162, 1, 1, "c.i8_roipooling_s", "input_w"], [162, 1, 1, "c.i8_roipooling_s", "max_c"], [162, 1, 1, "c.i8_roipooling_s", "num_rois"], [162, 1, 1, "c.i8_roipooling_s", "out_ptr"], [162, 1, 1, "c.i8_roipooling_s", "pooled_height"], [162, 1, 1, "c.i8_roipooling_s", "pooled_width"], [162, 1, 1, "c.i8_roipooling_s", "roi"], [162, 1, 1, "c.i8_roipooling_s", "scale"]], "i8_rsqrt_p": [[164, 1, 1, "c.i8_rsqrt_p", "dst"], [164, 1, 1, "c.i8_rsqrt_p", "length"], [164, 1, 1, "c.i8_rsqrt_p", "src"]], "i8_rsqrt_s": [[164, 1, 1, "c.i8_rsqrt_s", "core_mask"], [164, 1, 1, "c.i8_rsqrt_s", "dst"], [164, 1, 1, "c.i8_rsqrt_s", "length"], [164, 1, 1, "c.i8_rsqrt_s", "src"]], "i8_scalefusion_p": [[166, 1, 1, "c.i8_scalefusion_p", "bias"], [166, 1, 1, "c.i8_scalefusion_p", "dst_data"], [166, 1, 1, "c.i8_scalefusion_p", "length"], [166, 1, 1, "c.i8_scalefusion_p", "scale"], [166, 1, 1, "c.i8_scalefusion_p", "src_data"]], "i8_scalefusion_s": [[166, 1, 1, "c.i8_scalefusion_s", "bias"], [166, 1, 1, "c.i8_scalefusion_s", "core_mask"], [166, 1, 1, "c.i8_scalefusion_s", "dst_data"], [166, 1, 1, "c.i8_scalefusion_s", "length"], [166, 1, 1, "c.i8_scalefusion_s", "scale"], [166, 1, 1, "c.i8_scalefusion_s", "src_data"]], "i8_scatter_elements_p": [[167, 1, 1, "c.i8_scatter_elements_p", "core_mask"], [167, 1, 1, "c.i8_scatter_elements_p", "indices"], [167, 1, 1, "c.i8_scatter_elements_p", "input"], [167, 1, 1, "c.i8_scatter_elements_p", "output"], [167, 1, 1, "c.i8_scatter_elements_p", "param"], [167, 1, 1, "c.i8_scatter_elements_p", "updates"]], "i8_scatter_elements_s": [[167, 1, 1, "c.i8_scatter_elements_s", "core_mask"], [167, 1, 1, "c.i8_scatter_elements_s", "indices"], [167, 1, 1, "c.i8_scatter_elements_s", "input"], [167, 1, 1, "c.i8_scatter_elements_s", "output"], [167, 1, 1, "c.i8_scatter_elements_s", "param"], [167, 1, 1, "c.i8_scatter_elements_s", "updates"]], "i8_scatter_nd_p": [[168, 1, 1, "c.i8_scatter_nd_p", "indices"], [168, 1, 1, "c.i8_scatter_nd_p", "indices_ndim"], [168, 1, 1, "c.i8_scatter_nd_p", "indices_shape"], [168, 1, 1, "c.i8_scatter_nd_p", "output"], [168, 1, 1, "c.i8_scatter_nd_p", "output_ndim"], [168, 1, 1, "c.i8_scatter_nd_p", "output_shape"], [168, 1, 1, "c.i8_scatter_nd_p", "updates"]], "i8_scatter_nd_s": [[168, 1, 1, "c.i8_scatter_nd_s", "core_mask"], [168, 1, 1, "c.i8_scatter_nd_s", "indices"], [168, 1, 1, "c.i8_scatter_nd_s", "indices_ndim"], [168, 1, 1, "c.i8_scatter_nd_s", "indices_shape"], [168, 1, 1, "c.i8_scatter_nd_s", "output"], [168, 1, 1, "c.i8_scatter_nd_s", "output_ndim"], [168, 1, 1, "c.i8_scatter_nd_s", "output_shape"], [168, 1, 1, "c.i8_scatter_nd_s", "updates"]], "i8_scatter_nd_update_p": [[169, 1, 1, "c.i8_scatter_nd_update_p", "indices"], [169, 1, 1, "c.i8_scatter_nd_update_p", "indices_ndim"], [169, 1, 1, "c.i8_scatter_nd_update_p", "indices_shape"], [169, 1, 1, "c.i8_scatter_nd_update_p", "output"], [169, 1, 1, "c.i8_scatter_nd_update_p", "output_ndim"], [169, 1, 1, "c.i8_scatter_nd_update_p", "output_shape"], [169, 1, 1, "c.i8_scatter_nd_update_p", "updates"]], "i8_scatter_nd_update_s": [[169, 1, 1, "c.i8_scatter_nd_update_s", "core_mask"], [169, 1, 1, "c.i8_scatter_nd_update_s", "indices"], [169, 1, 1, "c.i8_scatter_nd_update_s", "indices_ndim"], [169, 1, 1, "c.i8_scatter_nd_update_s", "indices_shape"], [169, 1, 1, "c.i8_scatter_nd_update_s", "output"], [169, 1, 1, "c.i8_scatter_nd_update_s", "output_ndim"], [169, 1, 1, "c.i8_scatter_nd_update_s", "output_shape"], [169, 1, 1, "c.i8_scatter_nd_update_s", "updates"]], "i8_select_p": [[170, 1, 1, "c.i8_select_p", "condition"], [170, 1, 1, "c.i8_select_p", "index_list1"], [170, 1, 1, "c.i8_select_p", "index_list2"], [170, 1, 1, "c.i8_select_p", "index_list3"], [170, 1, 1, "c.i8_select_p", "input0"], [170, 1, 1, "c.i8_select_p", "input1"], [170, 1, 1, "c.i8_select_p", "is_broadcast"], [170, 1, 1, "c.i8_select_p", "output"], [170, 1, 1, "c.i8_select_p", "output_dims"], [170, 1, 1, "c.i8_select_p", "output_dims_num"]], "i8_select_s": [[170, 1, 1, "c.i8_select_s", "condition"], [170, 1, 1, "c.i8_select_s", "core_mask"], [170, 1, 1, "c.i8_select_s", "index_list1"], [170, 1, 1, "c.i8_select_s", "index_list2"], [170, 1, 1, "c.i8_select_s", "index_list3"], [170, 1, 1, "c.i8_select_s", "input0"], [170, 1, 1, "c.i8_select_s", "input1"], [170, 1, 1, "c.i8_select_s", "is_broadcast"], [170, 1, 1, "c.i8_select_s", "output"], [170, 1, 1, "c.i8_select_s", "output_dims"], [170, 1, 1, "c.i8_select_s", "output_dims_num"]], "i8_sigmoid_p": [[12, 1, 1, "c.i8_sigmoid_p", "Input0"], [12, 1, 1, "c.i8_sigmoid_p", "length"], [12, 1, 1, "c.i8_sigmoid_p", "output"]], "i8_sigmoid_s": [[12, 1, 1, "c.i8_sigmoid_s", "Input0"], [12, 1, 1, "c.i8_sigmoid_s", "core_mask"], [12, 1, 1, "c.i8_sigmoid_s", "length"], [12, 1, 1, "c.i8_sigmoid_s", "output"]], "i8_sigmoidcrossentropywithlogits_p": [[174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_p", "input0"], [174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_p", "input1"], [174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_p", "length"], [174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_p", "output"]], "i8_sigmoidcrossentropywithlogits_s": [[174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_s", "core_mask"], [174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_s", "input0"], [174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_s", "input1"], [174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_s", "length"], [174, 1, 1, "c.i8_sigmoidcrossentropywithlogits_s", "output"]], "i8_sin_p": [[175, 1, 1, "c.i8_sin_p", "dst_data"], [175, 1, 1, "c.i8_sin_p", "length"], [175, 1, 1, "c.i8_sin_p", "src_data"]], "i8_sin_s": [[175, 1, 1, "c.i8_sin_s", "core_mask"], [175, 1, 1, "c.i8_sin_s", "dst_data"], [175, 1, 1, "c.i8_sin_s", "length"], [175, 1, 1, "c.i8_sin_s", "src_data"]], "i8_slice_p": [[178, 1, 1, "c.i8_slice_p", "begin"], [178, 1, 1, "c.i8_slice_p", "input"], [178, 1, 1, "c.i8_slice_p", "input_shape"], [178, 1, 1, "c.i8_slice_p", "ndim"], [178, 1, 1, "c.i8_slice_p", "output"], [178, 1, 1, "c.i8_slice_p", "size"]], "i8_slice_s": [[178, 1, 1, "c.i8_slice_s", "begin"], [178, 1, 1, "c.i8_slice_s", "core_mask"], [178, 1, 1, "c.i8_slice_s", "input"], [178, 1, 1, "c.i8_slice_s", "input_shape"], [178, 1, 1, "c.i8_slice_s", "ndim"], [178, 1, 1, "c.i8_slice_s", "output"], [178, 1, 1, "c.i8_slice_s", "size"]], "i8_smoothl1loss_p": [[179, 1, 1, "c.i8_smoothl1loss_p", "beta"], [179, 1, 1, "c.i8_smoothl1loss_p", "length"], [179, 1, 1, "c.i8_smoothl1loss_p", "out"], [179, 1, 1, "c.i8_smoothl1loss_p", "predict"], [179, 1, 1, "c.i8_smoothl1loss_p", "target"]], "i8_smoothl1loss_s": [[179, 1, 1, "c.i8_smoothl1loss_s", "beta"], [179, 1, 1, "c.i8_smoothl1loss_s", "core_mask"], [179, 1, 1, "c.i8_smoothl1loss_s", "length"], [179, 1, 1, "c.i8_smoothl1loss_s", "out"], [179, 1, 1, "c.i8_smoothl1loss_s", "predict"], [179, 1, 1, "c.i8_smoothl1loss_s", "target"]], "i8_softmax_cross_entropy_with_logits_p": [[182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "batch_size"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "grads"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "labels"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "logits"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "need_grads"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "num_of_classes"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "output"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "probs"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_p", "sum_data"]], "i8_softmax_cross_entropy_with_logits_s": [[182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "batch_size"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "core_mask"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "grads"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "labels"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "logits"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "need_grads"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "num_of_classes"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "output"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "probs"], [182, 1, 1, "c.i8_softmax_cross_entropy_with_logits_s", "sum_data"]], "i8_softmax_p": [[181, 1, 1, "c.i8_softmax_p", "axis"], [181, 1, 1, "c.i8_softmax_p", "axis_size"], [181, 1, 1, "c.i8_softmax_p", "inner_size"], [181, 1, 1, "c.i8_softmax_p", "input_ptr"], [181, 1, 1, "c.i8_softmax_p", "n_dim"], [181, 1, 1, "c.i8_softmax_p", "output_ptr"], [181, 1, 1, "c.i8_softmax_p", "outter_size"], [181, 1, 1, "c.i8_softmax_p", "sum_data"]], "i8_softmax_s": [[181, 1, 1, "c.i8_softmax_s", "axis"], [181, 1, 1, "c.i8_softmax_s", "axis_size"], [181, 1, 1, "c.i8_softmax_s", "core_mask"], [181, 1, 1, "c.i8_softmax_s", "inner_size"], [181, 1, 1, "c.i8_softmax_s", "input_ptr"], [181, 1, 1, "c.i8_softmax_s", "n_dim"], [181, 1, 1, "c.i8_softmax_s", "output_ptr"], [181, 1, 1, "c.i8_softmax_s", "outter_size"], [181, 1, 1, "c.i8_softmax_s", "sum_data"]], "i8_softplus_p": [[12, 1, 1, "c.i8_softplus_p", "Input0"], [12, 1, 1, "c.i8_softplus_p", "length"], [12, 1, 1, "c.i8_softplus_p", "output"]], "i8_softplus_s": [[12, 1, 1, "c.i8_softplus_s", "Input0"], [12, 1, 1, "c.i8_softplus_s", "core_mask"], [12, 1, 1, "c.i8_softplus_s", "length"], [12, 1, 1, "c.i8_softplus_s", "output"]], "i8_softshrink_p": [[12, 1, 1, "c.i8_softshrink_p", "Input0"], [12, 1, 1, "c.i8_softshrink_p", "lambd"], [12, 1, 1, "c.i8_softshrink_p", "length"], [12, 1, 1, "c.i8_softshrink_p", "output"]], "i8_softshrink_s": [[12, 1, 1, "c.i8_softshrink_s", "Input0"], [12, 1, 1, "c.i8_softshrink_s", "core_mask"], [12, 1, 1, "c.i8_softshrink_s", "lambd"], [12, 1, 1, "c.i8_softshrink_s", "length"], [12, 1, 1, "c.i8_softshrink_s", "output"]], "i8_softsignopt_p": [[12, 1, 1, "c.i8_softsignopt_p", "Input0"], [12, 1, 1, "c.i8_softsignopt_p", "length"], [12, 1, 1, "c.i8_softsignopt_p", "output"]], "i8_softsignopt_s": [[12, 1, 1, "c.i8_softsignopt_s", "Input0"], [12, 1, 1, "c.i8_softsignopt_s", "core_mask"], [12, 1, 1, "c.i8_softsignopt_s", "length"], [12, 1, 1, "c.i8_softsignopt_s", "output"]], "i8_spacetobatch_p": [[183, 1, 1, "c.i8_spacetobatch_p", "block_size"], [183, 1, 1, "c.i8_spacetobatch_p", "data_size"], [183, 1, 1, "c.i8_spacetobatch_p", "input"], [183, 1, 1, "c.i8_spacetobatch_p", "input_shape"], [183, 1, 1, "c.i8_spacetobatch_p", "output"], [183, 1, 1, "c.i8_spacetobatch_p", "paddings"]], "i8_spacetobatch_s": [[183, 1, 1, "c.i8_spacetobatch_s", "block_size"], [183, 1, 1, "c.i8_spacetobatch_s", "core_mask"], [183, 1, 1, "c.i8_spacetobatch_s", "data_size"], [183, 1, 1, "c.i8_spacetobatch_s", "input"], [183, 1, 1, "c.i8_spacetobatch_s", "input_shape"], [183, 1, 1, "c.i8_spacetobatch_s", "output"], [183, 1, 1, "c.i8_spacetobatch_s", "paddings"]], "i8_spacetobatchnd_p": [[184, 1, 1, "c.i8_spacetobatchnd_p", "block_size"], [184, 1, 1, "c.i8_spacetobatchnd_p", "data_size"], [184, 1, 1, "c.i8_spacetobatchnd_p", "input"], [184, 1, 1, "c.i8_spacetobatchnd_p", "input_shape"], [184, 1, 1, "c.i8_spacetobatchnd_p", "output"], [184, 1, 1, "c.i8_spacetobatchnd_p", "paddings"]], "i8_spacetobatchnd_s": [[184, 1, 1, "c.i8_spacetobatchnd_s", "block_size"], [184, 1, 1, "c.i8_spacetobatchnd_s", "core_mask"], [184, 1, 1, "c.i8_spacetobatchnd_s", "data_size"], [184, 1, 1, "c.i8_spacetobatchnd_s", "input"], [184, 1, 1, "c.i8_spacetobatchnd_s", "input_shape"], [184, 1, 1, "c.i8_spacetobatchnd_s", "output"], [184, 1, 1, "c.i8_spacetobatchnd_s", "paddings"]], "i8_spacetodepth_p": [[185, 1, 1, "c.i8_spacetodepth_p", "block"], [185, 1, 1, "c.i8_spacetodepth_p", "data_size"], [185, 1, 1, "c.i8_spacetodepth_p", "in_shape"], [185, 1, 1, "c.i8_spacetodepth_p", "input"], [185, 1, 1, "c.i8_spacetodepth_p", "output"]], "i8_spacetodepth_s": [[185, 1, 1, "c.i8_spacetodepth_s", "block"], [185, 1, 1, "c.i8_spacetodepth_s", "core_mask"], [185, 1, 1, "c.i8_spacetodepth_s", "data_size"], [185, 1, 1, "c.i8_spacetodepth_s", "in_shape"], [185, 1, 1, "c.i8_spacetodepth_s", "input"], [185, 1, 1, "c.i8_spacetodepth_s", "output"]], "i8_sparsefillemptyrows_p": [[187, 1, 1, "c.i8_sparsefillemptyrows_p", "N"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "default_value"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "dense_rows"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "filled_count"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "indices_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "output_reverse_index_map_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "output_y_indices_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "output_y_values_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "rank"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "scratch_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_p", "values_ptr"]], "i8_sparsefillemptyrows_s": [[187, 1, 1, "c.i8_sparsefillemptyrows_s", "N"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "core_mask"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "default_value"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "dense_rows"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "filled_count"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "indices_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "output_reverse_index_map_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "output_y_indices_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "output_y_values_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "rank"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "scratch_ptr"], [187, 1, 1, "c.i8_sparsefillemptyrows_s", "values_ptr"]], "i8_sparsetodense_p": [[190, 1, 1, "c.i8_sparsetodense_p", "indices_vec"], [190, 1, 1, "c.i8_sparsetodense_p", "is_scalar"], [190, 1, 1, "c.i8_sparsetodense_p", "output"], [190, 1, 1, "c.i8_sparsetodense_p", "output_strides"], [190, 1, 1, "c.i8_sparsetodense_p", "sparse_length"], [190, 1, 1, "c.i8_sparsetodense_p", "sparse_values"]], "i8_sparsetodense_s": [[190, 1, 1, "c.i8_sparsetodense_s", "core_mask"], [190, 1, 1, "c.i8_sparsetodense_s", "indices_vec"], [190, 1, 1, "c.i8_sparsetodense_s", "is_scalar"], [190, 1, 1, "c.i8_sparsetodense_s", "output"], [190, 1, 1, "c.i8_sparsetodense_s", "output_strides"], [190, 1, 1, "c.i8_sparsetodense_s", "sparse_length"], [190, 1, 1, "c.i8_sparsetodense_s", "sparse_values"]], "i8_splice_p": [[191, 1, 1, "c.i8_splice_p", "context_dim"], [191, 1, 1, "c.i8_splice_p", "dst_col"], [191, 1, 1, "c.i8_splice_p", "dst_data"], [191, 1, 1, "c.i8_splice_p", "dst_row"], [191, 1, 1, "c.i8_splice_p", "forward_indexes"], [191, 1, 1, "c.i8_splice_p", "forward_indexes_dims"], [191, 1, 1, "c.i8_splice_p", "src_col"], [191, 1, 1, "c.i8_splice_p", "src_data"], [191, 1, 1, "c.i8_splice_p", "src_row"]], "i8_splice_s": [[191, 1, 1, "c.i8_splice_s", "context_dim"], [191, 1, 1, "c.i8_splice_s", "core_mask"], [191, 1, 1, "c.i8_splice_s", "dst_col"], [191, 1, 1, "c.i8_splice_s", "dst_data"], [191, 1, 1, "c.i8_splice_s", "dst_row"], [191, 1, 1, "c.i8_splice_s", "forward_indexes"], [191, 1, 1, "c.i8_splice_s", "forward_indexes_dims"], [191, 1, 1, "c.i8_splice_s", "src_col"], [191, 1, 1, "c.i8_splice_s", "src_data"], [191, 1, 1, "c.i8_splice_s", "src_row"]], "i8_split_p": [[192, 1, 1, "c.i8_split_p", "axis"], [192, 1, 1, "c.i8_split_p", "input"], [192, 1, 1, "c.i8_split_p", "input_ndim"], [192, 1, 1, "c.i8_split_p", "input_shape"], [192, 1, 1, "c.i8_split_p", "num_split"], [192, 1, 1, "c.i8_split_p", "outputs"], [192, 1, 1, "c.i8_split_p", "split_sizes"]], "i8_split_s": [[192, 1, 1, "c.i8_split_s", "axis"], [192, 1, 1, "c.i8_split_s", "core_mask"], [192, 1, 1, "c.i8_split_s", "input"], [192, 1, 1, "c.i8_split_s", "input_ndim"], [192, 1, 1, "c.i8_split_s", "input_shape"], [192, 1, 1, "c.i8_split_s", "num_split"], [192, 1, 1, "c.i8_split_s", "outputs"], [192, 1, 1, "c.i8_split_s", "split_sizes"]], "i8_split_with_overlap_p": [[193, 1, 1, "c.i8_split_with_overlap_p", "axis"], [193, 1, 1, "c.i8_split_with_overlap_p", "end_indices"], [193, 1, 1, "c.i8_split_with_overlap_p", "input"], [193, 1, 1, "c.i8_split_with_overlap_p", "input_ndim"], [193, 1, 1, "c.i8_split_with_overlap_p", "input_shape"], [193, 1, 1, "c.i8_split_with_overlap_p", "num_split"], [193, 1, 1, "c.i8_split_with_overlap_p", "outputs"], [193, 1, 1, "c.i8_split_with_overlap_p", "start_indices"]], "i8_split_with_overlap_s": [[193, 1, 1, "c.i8_split_with_overlap_s", "axis"], [193, 1, 1, "c.i8_split_with_overlap_s", "core_mask"], [193, 1, 1, "c.i8_split_with_overlap_s", "end_indices"], [193, 1, 1, "c.i8_split_with_overlap_s", "input"], [193, 1, 1, "c.i8_split_with_overlap_s", "input_ndim"], [193, 1, 1, "c.i8_split_with_overlap_s", "input_shape"], [193, 1, 1, "c.i8_split_with_overlap_s", "num_split"], [193, 1, 1, "c.i8_split_with_overlap_s", "outputs"], [193, 1, 1, "c.i8_split_with_overlap_s", "start_indices"]], "i8_sqrt_p": [[194, 1, 1, "c.i8_sqrt_p", "dst_data"], [194, 1, 1, "c.i8_sqrt_p", "length"], [194, 1, 1, "c.i8_sqrt_p", "src_data"]], "i8_sqrt_s": [[194, 1, 1, "c.i8_sqrt_s", "core_mask"], [194, 1, 1, "c.i8_sqrt_s", "dst_data"], [194, 1, 1, "c.i8_sqrt_s", "length"], [194, 1, 1, "c.i8_sqrt_s", "src_data"]], "i8_sqrtgrad_p": [[195, 1, 1, "c.i8_sqrtgrad_p", "input1"], [195, 1, 1, "c.i8_sqrtgrad_p", "input2"], [195, 1, 1, "c.i8_sqrtgrad_p", "output"], [195, 1, 1, "c.i8_sqrtgrad_p", "size"]], "i8_sqrtgrad_s": [[195, 1, 1, "c.i8_sqrtgrad_s", "core_mask"], [195, 1, 1, "c.i8_sqrtgrad_s", "input1"], [195, 1, 1, "c.i8_sqrtgrad_s", "input2"], [195, 1, 1, "c.i8_sqrtgrad_s", "output"], [195, 1, 1, "c.i8_sqrtgrad_s", "size"]], "i8_square_p": [[196, 1, 1, "c.i8_square_p", "dst"], [196, 1, 1, "c.i8_square_p", "length"], [196, 1, 1, "c.i8_square_p", "src"]], "i8_square_s": [[196, 1, 1, "c.i8_square_s", "core_mask"], [196, 1, 1, "c.i8_square_s", "dst"], [196, 1, 1, "c.i8_square_s", "length"], [196, 1, 1, "c.i8_square_s", "src"]], "i8_squaredifference_p": [[197, 1, 1, "c.i8_squaredifference_p", "input0"], [197, 1, 1, "c.i8_squaredifference_p", "input1"], [197, 1, 1, "c.i8_squaredifference_p", "length"], [197, 1, 1, "c.i8_squaredifference_p", "output"]], "i8_squaredifference_s": [[197, 1, 1, "c.i8_squaredifference_s", "core_mask"], [197, 1, 1, "c.i8_squaredifference_s", "input0"], [197, 1, 1, "c.i8_squaredifference_s", "input1"], [197, 1, 1, "c.i8_squaredifference_s", "length"], [197, 1, 1, "c.i8_squaredifference_s", "output"]], "i8_stack_p": [[199, 1, 1, "c.i8_stack_p", "axis"], [199, 1, 1, "c.i8_stack_p", "input_ndim"], [199, 1, 1, "c.i8_stack_p", "input_shape"], [199, 1, 1, "c.i8_stack_p", "inputs"], [199, 1, 1, "c.i8_stack_p", "num_inputs"], [199, 1, 1, "c.i8_stack_p", "output"]], "i8_stack_s": [[199, 1, 1, "c.i8_stack_s", "axis"], [199, 1, 1, "c.i8_stack_s", "core_mask"], [199, 1, 1, "c.i8_stack_s", "input_ndim"], [199, 1, 1, "c.i8_stack_s", "input_shape"], [199, 1, 1, "c.i8_stack_s", "inputs"], [199, 1, 1, "c.i8_stack_s", "num_inputs"], [199, 1, 1, "c.i8_stack_s", "output"]], "i8_subrelu6_p": [[202, 1, 1, "c.i8_subrelu6_p", "input0"], [202, 1, 1, "c.i8_subrelu6_p", "input1"], [202, 1, 1, "c.i8_subrelu6_p", "output"], [202, 1, 1, "c.i8_subrelu6_p", "size"]], "i8_subrelu6_s": [[202, 1, 1, "c.i8_subrelu6_s", "core_mask"], [202, 1, 1, "c.i8_subrelu6_s", "input0"], [202, 1, 1, "c.i8_subrelu6_s", "input1"], [202, 1, 1, "c.i8_subrelu6_s", "output"], [202, 1, 1, "c.i8_subrelu6_s", "size"]], "i8_subrelu_p": [[202, 1, 1, "c.i8_subrelu_p", "input0"], [202, 1, 1, "c.i8_subrelu_p", "input1"], [202, 1, 1, "c.i8_subrelu_p", "output"], [202, 1, 1, "c.i8_subrelu_p", "size"]], "i8_subrelu_s": [[202, 1, 1, "c.i8_subrelu_s", "core_mask"], [202, 1, 1, "c.i8_subrelu_s", "input0"], [202, 1, 1, "c.i8_subrelu_s", "input1"], [202, 1, 1, "c.i8_subrelu_s", "output"], [202, 1, 1, "c.i8_subrelu_s", "size"]], "i8_swish_p": [[12, 1, 1, "c.i8_swish_p", "Input0"], [12, 1, 1, "c.i8_swish_p", "length"], [12, 1, 1, "c.i8_swish_p", "output"]], "i8_swish_s": [[12, 1, 1, "c.i8_swish_s", "Input0"], [12, 1, 1, "c.i8_swish_s", "core_mask"], [12, 1, 1, "c.i8_swish_s", "length"], [12, 1, 1, "c.i8_swish_s", "output"]], "i8_tanh_p": [[12, 1, 1, "c.i8_tanh_p", "Input0"], [12, 1, 1, "c.i8_tanh_p", "length"], [12, 1, 1, "c.i8_tanh_p", "output"]], "i8_tanh_s": [[12, 1, 1, "c.i8_tanh_s", "Input0"], [12, 1, 1, "c.i8_tanh_s", "core_mask"], [12, 1, 1, "c.i8_tanh_s", "length"], [12, 1, 1, "c.i8_tanh_s", "output"]], "i8_tensor_scatter_add_p": [[206, 1, 1, "c.i8_tensor_scatter_add_p", "index_depth"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "indices"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "input"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "input_rank"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "input_shape"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "num_unit"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "output"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "output_unit_offsets"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "strides"], [206, 1, 1, "c.i8_tensor_scatter_add_p", "updates"]], "i8_tensor_scatter_add_s": [[206, 1, 1, "c.i8_tensor_scatter_add_s", "core_mask"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "index_depth"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "indices"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "input"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "input_rank"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "input_shape"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "num_unit"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "output"], [206, 1, 1, "c.i8_tensor_scatter_add_s", "updates"]], "i8_tensorarrayread_p": [[208, 1, 1, "c.i8_tensorarrayread_p", "handle_data"], [208, 1, 1, "c.i8_tensorarrayread_p", "handle_size"], [208, 1, 1, "c.i8_tensorarrayread_p", "index"], [208, 1, 1, "c.i8_tensorarrayread_p", "output_data"], [208, 1, 1, "c.i8_tensorarrayread_p", "output_size"]], "i8_tensorarrayread_s": [[208, 1, 1, "c.i8_tensorarrayread_s", "core_mask"], [208, 1, 1, "c.i8_tensorarrayread_s", "handle_data"], [208, 1, 1, "c.i8_tensorarrayread_s", "handle_size"], [208, 1, 1, "c.i8_tensorarrayread_s", "index"], [208, 1, 1, "c.i8_tensorarrayread_s", "output_data"], [208, 1, 1, "c.i8_tensorarrayread_s", "output_size"]], "i8_tensorlistfromtensor_p": [[210, 1, 1, "c.i8_tensorlistfromtensor_p", "input_tensor_shape"], [210, 1, 1, "c.i8_tensorlistfromtensor_p", "input_tensor_total_elements"], [210, 1, 1, "c.i8_tensorlistfromtensor_p", "input_tensor_values"], [210, 1, 1, "c.i8_tensorlistfromtensor_p", "output_tensors"]], "i8_tensorlistfromtensor_s": [[210, 1, 1, "c.i8_tensorlistfromtensor_s", "core_mask"], [210, 1, 1, "c.i8_tensorlistfromtensor_s", "input_tensor_shape"], [210, 1, 1, "c.i8_tensorlistfromtensor_s", "input_tensor_total_elements"], [210, 1, 1, "c.i8_tensorlistfromtensor_s", "input_tensor_values"], [210, 1, 1, "c.i8_tensorlistfromtensor_s", "output_tensors"]], "i8_tile_p": [[215, 1, 1, "c.i8_tile_p", "input"], [215, 1, 1, "c.i8_tile_p", "input_shape"], [215, 1, 1, "c.i8_tile_p", "output"], [215, 1, 1, "c.i8_tile_p", "stride"], [215, 1, 1, "c.i8_tile_p", "tile_dim"], [215, 1, 1, "c.i8_tile_p", "tile_num"]], "i8_tile_s": [[215, 1, 1, "c.i8_tile_s", "core_mask"], [215, 1, 1, "c.i8_tile_s", "input"], [215, 1, 1, "c.i8_tile_s", "input_shape"], [215, 1, 1, "c.i8_tile_s", "output"], [215, 1, 1, "c.i8_tile_s", "stride"], [215, 1, 1, "c.i8_tile_s", "tile_dim"], [215, 1, 1, "c.i8_tile_s", "tile_num"]], "i8_to_fp_dequant_p": [[146, 1, 1, "c.i8_to_fp_dequant_p", "input"], [146, 1, 1, "c.i8_to_fp_dequant_p", "length"], [146, 1, 1, "c.i8_to_fp_dequant_p", "output"], [146, 1, 1, "c.i8_to_fp_dequant_p", "scale"], [146, 1, 1, "c.i8_to_fp_dequant_p", "zp"]], "i8_to_fp_dequant_s": [[146, 1, 1, "c.i8_to_fp_dequant_s", "core_mask"], [146, 1, 1, "c.i8_to_fp_dequant_s", "input"], [146, 1, 1, "c.i8_to_fp_dequant_s", "length"], [146, 1, 1, "c.i8_to_fp_dequant_s", "output"], [146, 1, 1, "c.i8_to_fp_dequant_s", "scale"], [146, 1, 1, "c.i8_to_fp_dequant_s", "zp"]], "i8_to_hp_dequant_p": [[146, 1, 1, "c.i8_to_hp_dequant_p", "input"], [146, 1, 1, "c.i8_to_hp_dequant_p", "length"], [146, 1, 1, "c.i8_to_hp_dequant_p", "output"], [146, 1, 1, "c.i8_to_hp_dequant_p", "scale"], [146, 1, 1, "c.i8_to_hp_dequant_p", "zp"]], "i8_to_hp_dequant_s": [[146, 1, 1, "c.i8_to_hp_dequant_s", "core_mask"], [146, 1, 1, "c.i8_to_hp_dequant_s", "input"], [146, 1, 1, "c.i8_to_hp_dequant_s", "length"], [146, 1, 1, "c.i8_to_hp_dequant_s", "output"], [146, 1, 1, "c.i8_to_hp_dequant_s", "scale"], [146, 1, 1, "c.i8_to_hp_dequant_s", "zp"]], "i8_transpose_p": [[217, 1, 1, "c.i8_transpose_p", "in_data"], [217, 1, 1, "c.i8_transpose_p", "num_axes"], [217, 1, 1, "c.i8_transpose_p", "out_data"], [217, 1, 1, "c.i8_transpose_p", "out_strides"], [217, 1, 1, "c.i8_transpose_p", "output_shape"], [217, 1, 1, "c.i8_transpose_p", "perm"], [217, 1, 1, "c.i8_transpose_p", "strides"]], "i8_transpose_s": [[217, 1, 1, "c.i8_transpose_s", "core_mask"], [217, 1, 1, "c.i8_transpose_s", "in_data"], [217, 1, 1, "c.i8_transpose_s", "num_axes"], [217, 1, 1, "c.i8_transpose_s", "out_data"], [217, 1, 1, "c.i8_transpose_s", "out_strides"], [217, 1, 1, "c.i8_transpose_s", "output_shape"], [217, 1, 1, "c.i8_transpose_s", "perm"], [217, 1, 1, "c.i8_transpose_s", "strides"]], "i8_tril_p": [[218, 1, 1, "c.i8_tril_p", "dst"], [218, 1, 1, "c.i8_tril_p", "height"], [218, 1, 1, "c.i8_tril_p", "k"], [218, 1, 1, "c.i8_tril_p", "out_elems"], [218, 1, 1, "c.i8_tril_p", "src"], [218, 1, 1, "c.i8_tril_p", "width"]], "i8_tril_s": [[218, 1, 1, "c.i8_tril_s", "core_mask"], [218, 1, 1, "c.i8_tril_s", "dst"], [218, 1, 1, "c.i8_tril_s", "height"], [218, 1, 1, "c.i8_tril_s", "k"], [218, 1, 1, "c.i8_tril_s", "out_elems"], [218, 1, 1, "c.i8_tril_s", "src"], [218, 1, 1, "c.i8_tril_s", "width"]], "i8_triu_p": [[219, 1, 1, "c.i8_triu_p", "dst"], [219, 1, 1, "c.i8_triu_p", "height"], [219, 1, 1, "c.i8_triu_p", "k"], [219, 1, 1, "c.i8_triu_p", "out_elems"], [219, 1, 1, "c.i8_triu_p", "src"], [219, 1, 1, "c.i8_triu_p", "width"]], "i8_triu_s": [[219, 1, 1, "c.i8_triu_s", "core_mask"], [219, 1, 1, "c.i8_triu_s", "dst"], [219, 1, 1, "c.i8_triu_s", "height"], [219, 1, 1, "c.i8_triu_s", "k"], [219, 1, 1, "c.i8_triu_s", "out_elems"], [219, 1, 1, "c.i8_triu_s", "src"], [219, 1, 1, "c.i8_triu_s", "width"]], "i8_unsorted_segment_sum_p": [[222, 1, 1, "c.i8_unsorted_segment_sum_p", "dim0"], [222, 1, 1, "c.i8_unsorted_segment_sum_p", "dim1"], [222, 1, 1, "c.i8_unsorted_segment_sum_p", "id_max"], [222, 1, 1, "c.i8_unsorted_segment_sum_p", "index"], [222, 1, 1, "c.i8_unsorted_segment_sum_p", "input"], [222, 1, 1, "c.i8_unsorted_segment_sum_p", "output"]], "i8_unsorted_segment_sum_s": [[222, 1, 1, "c.i8_unsorted_segment_sum_s", "core_mask"], [222, 1, 1, "c.i8_unsorted_segment_sum_s", "dim0"], [222, 1, 1, "c.i8_unsorted_segment_sum_s", "dim1"], [222, 1, 1, "c.i8_unsorted_segment_sum_s", "id_max"], [222, 1, 1, "c.i8_unsorted_segment_sum_s", "index"], [222, 1, 1, "c.i8_unsorted_segment_sum_s", "input"], [222, 1, 1, "c.i8_unsorted_segment_sum_s", "output"]], "i8_where_p": [[225, 1, 1, "c.i8_where_p", "condition"], [225, 1, 1, "c.i8_where_p", "input0"], [225, 1, 1, "c.i8_where_p", "input1"], [225, 1, 1, "c.i8_where_p", "length"], [225, 1, 1, "c.i8_where_p", "output"]], "i8_where_s": [[225, 1, 1, "c.i8_where_s", "condition"], [225, 1, 1, "c.i8_where_s", "core_mask"], [225, 1, 1, "c.i8_where_s", "input0"], [225, 1, 1, "c.i8_where_s", "input1"], [225, 1, 1, "c.i8_where_s", "length"], [225, 1, 1, "c.i8_where_s", "output"]], "i8_zerolike_p": [[226, 1, 1, "c.i8_zerolike_p", "length"], [226, 1, 1, "c.i8_zerolike_p", "output"]], "i8_zerolike_s": [[226, 1, 1, "c.i8_zerolike_s", "core_mask"], [226, 1, 1, "c.i8_zerolike_s", "length"], [226, 1, 1, "c.i8_zerolike_s", "output"]], "mindradar": [[6, 2, 1, "", "ComplexAbs"], [7, 2, 1, "", "FFT"], [8, 2, 1, "", "IFFT"]], "rank_p": [[151, 1, 1, "c.rank_p", "n"], [151, 1, 1, "c.rank_p", "output"]], "rank_s": [[151, 1, 1, "c.rank_s", "core_mask"], [151, 1, 1, "c.rank_s", "n"], [151, 1, 1, "c.rank_s", "output"]], "shape": [[172, 1, 1, "c.shape", "inputs"], [172, 1, 1, "c.shape", "inputs_size"], [172, 1, 1, "c.shape", "outputs"], [172, 1, 1, "c.shape", "outputs_size"], [172, 1, 1, "c.shape", "param"]], "size_p": [[176, 1, 1, "c.size_p", "n"], [176, 1, 1, "c.size_p", "output"], [176, 1, 1, "c.size_p", "shape"]], "size_s": [[176, 1, 1, "c.size_s", "core_mask"], [176, 1, 1, "c.size_s", "n"], [176, 1, 1, "c.size_s", "output"], [176, 1, 1, "c.size_s", "shape"]], "skipgram_p": [[177, 1, 1, "c.skipgram_p", "blank"], [177, 1, 1, "c.skipgram_p", "grams"], [177, 1, 1, "c.skipgram_p", "grams_word_count"], [177, 1, 1, "c.skipgram_p", "include_all_ngrams"], [177, 1, 1, "c.skipgram_p", "len"], [177, 1, 1, "c.skipgram_p", "max_skip_size"], [177, 1, 1, "c.skipgram_p", "ngram_size"], [177, 1, 1, "c.skipgram_p", "offset"], [177, 1, 1, "c.skipgram_p", "output_tensor"], [177, 1, 1, "c.skipgram_p", "sentence"], [177, 1, 1, "c.skipgram_p", "shape"], [177, 1, 1, "c.skipgram_p", "stack"], [177, 1, 1, "c.skipgram_p", "words"]], "skipgram_s": [[177, 1, 1, "c.skipgram_s", "blank"], [177, 1, 1, "c.skipgram_s", "core_mask"], [177, 1, 1, "c.skipgram_s", "grams"], [177, 1, 1, "c.skipgram_s", "grams_word_count"], [177, 1, 1, "c.skipgram_s", "include_all_ngrams"], [177, 1, 1, "c.skipgram_s", "len"], [177, 1, 1, "c.skipgram_s", "max_skip_size"], [177, 1, 1, "c.skipgram_s", "ngram_size"], [177, 1, 1, "c.skipgram_s", "offset"], [177, 1, 1, "c.skipgram_s", "output_tensor"], [177, 1, 1, "c.skipgram_s", "sentence"], [177, 1, 1, "c.skipgram_s", "shape"], [177, 1, 1, "c.skipgram_s", "stack"], [177, 1, 1, "c.skipgram_s", "words"]], "sparsereshape_p": [[188, 1, 1, "c.sparsereshape_p", "N"], [188, 1, 1, "c.sparsereshape_p", "in_indices_ptr"], [188, 1, 1, "c.sparsereshape_p", "in_inshape_ptr"], [188, 1, 1, "c.sparsereshape_p", "in_outshape_ptr"], [188, 1, 1, "c.sparsereshape_p", "in_stride"], [188, 1, 1, "c.sparsereshape_p", "input_rank"], [188, 1, 1, "c.sparsereshape_p", "out_indices_ptr"], [188, 1, 1, "c.sparsereshape_p", "out_outshape_ptr"], [188, 1, 1, "c.sparsereshape_p", "out_stride"], [188, 1, 1, "c.sparsereshape_p", "output_rank"]], "sparsereshape_s": [[188, 1, 1, "c.sparsereshape_s", "N"], [188, 1, 1, "c.sparsereshape_s", "core_mask"], [188, 1, 1, "c.sparsereshape_s", "in_indices_ptr"], [188, 1, 1, "c.sparsereshape_s", "in_inshape_ptr"], [188, 1, 1, "c.sparsereshape_s", "in_outshape_ptr"], [188, 1, 1, "c.sparsereshape_s", "in_stride"], [188, 1, 1, "c.sparsereshape_s", "input_rank"], [188, 1, 1, "c.sparsereshape_s", "out_indices_ptr"], [188, 1, 1, "c.sparsereshape_s", "out_outshape_ptr"], [188, 1, 1, "c.sparsereshape_s", "out_stride"], [188, 1, 1, "c.sparsereshape_s", "output_rank"]], "stridedslice": [[200, 1, 1, "c.stridedslice", "begins"], [200, 1, 1, "c.stridedslice", "cal_num_per_thread"], [200, 1, 1, "c.stridedslice", "caled_num"], [200, 1, 1, "c.stridedslice", "core_mask"], [200, 1, 1, "c.stridedslice", "data_type_bytes"], [200, 1, 1, "c.stridedslice", "ends"], [200, 1, 1, "c.stridedslice", "fast_run"], [200, 1, 1, "c.stridedslice", "in_shape"], [200, 1, 1, "c.stridedslice", "in_shape_size"], [200, 1, 1, "c.stridedslice", "inner"], [200, 1, 1, "c.stridedslice", "inner_size"], [200, 1, 1, "c.stridedslice", "input"], [200, 1, 1, "c.stridedslice", "out_shape"], [200, 1, 1, "c.stridedslice", "out_shape_size"], [200, 1, 1, "c.stridedslice", "outer"], [200, 1, 1, "c.stridedslice", "output"], [200, 1, 1, "c.stridedslice", "parallel_on_outer"], [200, 1, 1, "c.stridedslice", "parallel_on_split_axis"], [200, 1, 1, "c.stridedslice", "soft_copy_mode"], [200, 1, 1, "c.stridedslice", "split_axis"], [200, 1, 1, "c.stridedslice", "strides"], [200, 1, 1, "c.stridedslice", "temp_space"]], "switch_p": [[204, 1, 1, "c.switch_p", "condition"], [204, 1, 1, "c.switch_p", "input_x"], [204, 1, 1, "c.switch_p", "input_y"], [204, 1, 1, "c.switch_p", "output"]], "switch_s": [[204, 1, 1, "c.switch_s", "condition"], [204, 1, 1, "c.switch_s", "input_x"], [204, 1, 1, "c.switch_s", "input_y"], [204, 1, 1, "c.switch_s", "output"]], "switchlayer_p": [[205, 1, 1, "c.switchlayer_p", "index"], [205, 1, 1, "c.switchlayer_p", "input_tensors"], [205, 1, 1, "c.switchlayer_p", "output"]], "switchlayer_s": [[205, 1, 1, "c.switchlayer_s", "core_mask"], [205, 1, 1, "c.switchlayer_s", "index"], [205, 1, 1, "c.switchlayer_s", "input_tensors"], [205, 1, 1, "c.switchlayer_s", "output"]], "tensorarrayread_p": [[207, 1, 1, "c.tensorarrayread_p", "index"], [207, 1, 1, "c.tensorarrayread_p", "output"], [207, 1, 1, "c.tensorarrayread_p", "tensors"]], "tensorarrayread_s": [[207, 1, 1, "c.tensorarrayread_s", "core_mask"], [207, 1, 1, "c.tensorarrayread_s", "index"], [207, 1, 1, "c.tensorarrayread_s", "output"], [207, 1, 1, "c.tensorarrayread_s", "tensors"]], "tensorarraywrite_p": [[207, 1, 1, "c.tensorarraywrite_p", "index"], [207, 1, 1, "c.tensorarraywrite_p", "input"], [207, 1, 1, "c.tensorarraywrite_p", "tensors"]], "tensorarraywrite_s": [[207, 1, 1, "c.tensorarraywrite_s", "core_mask"], [207, 1, 1, "c.tensorarraywrite_s", "index"], [207, 1, 1, "c.tensorarraywrite_s", "input"], [207, 1, 1, "c.tensorarraywrite_s", "tensors"]], "tensorlistgetitem_p": [[211, 1, 1, "c.tensorlistgetitem_p", "data_type"], [211, 1, 1, "c.tensorlistgetitem_p", "dst"], [211, 1, 1, "c.tensorlistgetitem_p", "length"], [211, 1, 1, "c.tensorlistgetitem_p", "src"]], "tensorlistgetitem_s": [[211, 1, 1, "c.tensorlistgetitem_s", "core_mask"], [211, 1, 1, "c.tensorlistgetitem_s", "data_type"], [211, 1, 1, "c.tensorlistgetitem_s", "dst"], [211, 1, 1, "c.tensorlistgetitem_s", "length"], [211, 1, 1, "c.tensorlistgetitem_s", "src"]], "tensorlistsetitem_p": [[213, 1, 1, "c.tensorlistsetitem_p", "copy_size"], [213, 1, 1, "c.tensorlistsetitem_p", "in_data"], [213, 1, 1, "c.tensorlistsetitem_p", "in_item"], [213, 1, 1, "c.tensorlistsetitem_p", "index"], [213, 1, 1, "c.tensorlistsetitem_p", "out_data"], [213, 1, 1, "c.tensorlistsetitem_p", "tensorlist_size"]], "tensorlistsetitem_s": [[213, 1, 1, "c.tensorlistsetitem_s", "copy_size"], [213, 1, 1, "c.tensorlistsetitem_s", "core_mask"], [213, 1, 1, "c.tensorlistsetitem_s", "in_data"], [213, 1, 1, "c.tensorlistsetitem_s", "in_item"], [213, 1, 1, "c.tensorlistsetitem_s", "index"], [213, 1, 1, "c.tensorlistsetitem_s", "out_data"], [213, 1, 1, "c.tensorlistsetitem_s", "tensorlist_size"]], "tensorliststack_p": [[214, 1, 1, "c.tensorliststack_p", "output_data"], [214, 1, 1, "c.tensorliststack_p", "tensor_data"], [214, 1, 1, "c.tensorliststack_p", "tensor_data_type"], [214, 1, 1, "c.tensorliststack_p", "tensor_element_nums"], [214, 1, 1, "c.tensorliststack_p", "tensor_num"], [214, 1, 1, "c.tensorliststack_p", "unknown_type_offset"]], "tensorliststack_s": [[214, 1, 1, "c.tensorliststack_s", "core_mask"], [214, 1, 1, "c.tensorliststack_s", "output_data"], [214, 1, 1, "c.tensorliststack_s", "tensor_data"], [214, 1, 1, "c.tensorliststack_s", "tensor_data_type"], [214, 1, 1, "c.tensorliststack_s", "tensor_element_nums"], [214, 1, 1, "c.tensorliststack_s", "tensor_num"], [214, 1, 1, "c.tensorliststack_s", "unknown_type_offset"]], "unstack_p": [[224, 1, 1, "c.unstack_p", "axis"], [224, 1, 1, "c.unstack_p", "data_size"], [224, 1, 1, "c.unstack_p", "input"], [224, 1, 1, "c.unstack_p", "ndim"], [224, 1, 1, "c.unstack_p", "output"], [224, 1, 1, "c.unstack_p", "shape"]], "unstack_s": [[224, 1, 1, "c.unstack_s", "axis"], [224, 1, 1, "c.unstack_s", "core_mask"], [224, 1, 1, "c.unstack_s", "data_size"], [224, 1, 1, "c.unstack_s", "input"], [224, 1, 1, "c.unstack_s", "ndim"], [224, 1, 1, "c.unstack_s", "output"], [224, 1, 1, "c.unstack_s", "shape"]]}, "objnames": {"0": ["c", "function", "C \u51fd\u6570"], "1": ["c", "functionParam", "C \u51fd\u6570\u53c2\u6570"], "2": ["py", "class", "Python \u7c7b"]}, "objtypes": {"0": "c:function", "1": "c:functionParam", "2": "py:class"}, "terms": {"0001f": 115, "001f": 14, "004": 4, "01": 0, "01f": 13, "044715": 12, "05f": 146, "09f": 66, "0b0001": [16, 47, 48, 52, 53, 72, 79, 95, 103, 154, 157, 159, 167, 198, 223], "0b1011": [45, 74, 75, 110, 111, 192, 199, 215], "0b1111": [16, 47, 48, 52, 53, 72, 79, 95, 103, 154, 157, 159, 160, 167, 198, 223], "0f": [30, 31, 44, 60, 63, 64, 68, 74, 78, 100, 104, 105, 106, 114, 115, 124, 125, 126, 141, 142, 143, 145, 148, 162, 171, 179, 180, 187, 190], "0i": 140, "0x0b": [140, 156, 168], "0x0f": [124, 125], "0x1": 80, "0x10000": [49, 50], "0x100000": 20, "0x1000000": [136, 154], "0x10000000": [11, 12, 13, 15, 16, 17, 18, 21, 23, 27, 28, 29, 31, 32, 34, 35, 36, 37, 38, 39, 40, 41, 43, 44, 47, 48, 49, 50, 55, 56, 57, 59, 61, 62, 63, 64, 66, 68, 71, 74, 75, 79, 82, 84, 87, 88, 89, 90, 91, 92, 93, 94, 95, 99, 100, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 115, 116, 117, 121, 122, 123, 127, 128, 129, 130, 131, 132, 133, 138, 139, 141, 142, 143, 146, 147, 150, 152, 153, 154, 163, 164, 165, 167, 170, 171, 174, 176, 178, 180, 183, 184, 185, 188, 189, 190, 192, 195, 196, 197, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 213, 214, 215, 216, 221, 222, 225], "0x10000040": 176, "0x10000200": 117, "0x10000400": 117, "0x10000600": 117, "0x10000800": 117, "0x10000a00": 117, "0x10000c00": 117, "0x10000e00": 117, "0x10001000": [11, 13, 17, 18, 21, 24, 25, 48, 56, 57, 61, 71, 74, 91, 99, 104, 105, 108, 110, 111, 116, 117, 122, 127, 129, 130, 132, 133, 138, 139, 142, 143, 147, 152, 153, 164, 165, 167, 170, 177, 180, 188, 190, 195, 196, 197, 201, 202, 203, 204, 205, 207, 208, 213, 225], "0x10001200": 117, "0x10001400": 117, "0x10001600": 117, "0x10001800": 117, "0x10001a00": 117, "0x10001c00": 117, "0x10001f00": 117, "0x10002000": [13, 15, 18, 23, 24, 25, 48, 56, 57, 61, 100, 104, 105, 111, 112, 116, 117, 122, 127, 129, 130, 140, 142, 143, 147, 152, 165, 167, 170, 171, 177, 180, 188, 190, 195, 197, 202, 203, 205, 207, 208, 213, 225], "0x10002200": 117, "0x10003000": [24, 25, 48, 116, 147, 167, 170, 177, 180, 188, 205, 207, 208, 213, 225], "0x10004000": [12, 15, 23, 24, 25, 27, 28, 32, 48, 66, 87, 94, 95, 147, 167, 170, 171, 174, 177, 213], "0x10005000": [24, 25, 48, 167, 170, 177, 213], "0x10006000": [15, 48, 167, 170, 177, 213], "0x10007000": 177, "0x10008000": [75, 87, 94, 95, 174, 177], "0x10008100": 75, "0x10009000": [75, 177], "0x1000c000": [87, 94, 95], "0x1000ll": 160, "0x10010000": [16, 35, 36, 47, 79, 87, 88, 89, 90, 94, 95, 103, 109, 112, 115, 148, 149, 154, 158, 163, 172, 178, 182, 185, 186, 189, 192, 200, 206, 210, 211, 215, 216, 220, 222], "0x10011000": [172, 210], "0x10012000": 210, "0x10014000": [87, 94, 95], "0x10018000": 95, "0x1001c000": 95, "0x10020000": [16, 41, 47, 49, 50, 88, 89, 90, 95, 109, 121, 154, 158, 182, 186, 192, 200, 206, 216, 222], "0x10021000": 154, "0x10022000": 154, "0x10023000": 154, "0x10024000": [95, 154], "0x10030000": [16, 47, 95, 182, 186, 206], "0x10034000": 95, "0x10038000": 95, "0x1003c000": 95, "0x10040000": [16, 29, 47, 49, 50, 59, 121, 182, 183, 184, 186], "0x10050000": [182, 186], "0x10060000": [16, 47, 49, 50, 182, 186], "0x10070000": [16, 47, 49, 50], "0x10080000": 29, "0x100c0000": 29, "0x10100000": [29, 31, 34, 141, 214], "0x10200000": [29, 34, 141, 214], "0x10300000": [34, 141], "0x10400000": [34, 141], "0x10500000": [34, 141], "0x10600000": [34, 141], "0x10700000": [34, 141], "0x10800000": 224, "0x10810000": [10, 19, 20, 22, 30, 33, 42, 45, 46, 51, 54, 58, 67, 69, 70, 73, 80, 81, 83, 85, 86, 96, 97, 98, 101, 113, 114, 126, 134, 135, 137, 144, 151, 155, 156, 161, 162, 166, 168, 169, 173, 175, 179, 181, 187, 191, 193, 194, 199, 217, 218, 219, 224, 226], "0x10811000": 101, "0x10812000": [101, 199], "0x10814000": 19, "0x10820000": [10, 14, 19, 22, 30, 33, 42, 45, 51, 58, 67, 69, 70, 73, 78, 80, 81, 83, 85, 86, 96, 97, 98, 101, 113, 114, 126, 134, 135, 137, 144, 155, 156, 161, 162, 166, 168, 169, 173, 175, 181, 187, 191, 193, 194, 199, 217, 224], "0x10821000": 101, "0x10822000": 101, "0x10830000": [14, 30, 33, 45, 58, 67, 70, 83, 86, 96, 97, 113, 114, 134, 135, 137, 161, 166, 168, 169, 173, 181, 187], "0x10840000": [14, 30, 33, 58, 96, 97, 114, 134, 135, 137, 166, 187], "0x10850000": [14, 54, 96, 134, 135, 137, 179, 187, 218, 219], "0x10860000": [135, 187], "0x108a0000": 179, "0x11000000": [37, 38, 39, 40, 43, 44, 55, 63, 64, 68, 84, 92, 93, 102, 107, 123, 128, 146, 221], "0x12000000": [37, 39, 40, 62, 63, 64, 84, 92, 93, 102, 123, 128, 131], "0x13000000": [39, 40, 62, 102, 123, 128, 131], "0x14000000": [39, 40, 62, 102, 123, 128, 131], "0x15000000": [62, 102, 131], "0x16000000": [62, 102, 131], "0x17000000": [62, 102, 131], "0x18000000": 62, "0x1b000000": [62, 131], "0x1b100000": [62, 131], "0x1b200000": [62, 131], "0x1b300000": [62, 131], "0x1b400000": [62, 131], "0x1b500000": [62, 131], "0x1b600000": [62, 131], "0x1b700000": [62, 131], "0x1b800000": [62, 131], "0x20000": [49, 50], "0x200000": 20, "0x300000": 20, "0x400000": 20, "0x500000": 20, "0x600000": 20, "0x700000": 20, "0x81000000": [30, 126, 141], "0x82000000": [30, 103, 126, 141], "0x83100000": 126, "0x83200000": [30, 126], "0x83300000": 126, "0x83400000": [30, 126], "0x84000000": [141, 157, 160], "0x84003000": 160, "0x84004000": 160, "0x84005000": 160, "0x84006000": 160, "0x84007000": 160, "0x85000000": [141, 157], "0x86000000": [141, 157], "0x87000000": [141, 157], "0x87100000": 157, "0x88000000": [16, 47, 48, 52, 53, 72, 79, 95, 103, 141, 154, 157, 159, 160, 167, 198, 223], "0x88100000": 95, "0x88200000": 95, "0x88300000": 95, "0x88400000": 95, "0x88500000": 95, "0x88600000": 95, "0x88700000": 95, "0x88800000": 95, "0x88900000": 95, "0x88a00000": 95, "0x88b00000": 95, "0x88c00000": 95, "0x88d00000": 95, "0x89000000": [16, 47, 48, 53, 141, 157], "0x8a000000": [53, 157], "0x8b000000": [53, 157], "0x8c000000": [53, 157], "0x8d000000": [53, 157], "0x8e000000": [53, 157], "0x8f000000": 53, "0x90000000": [16, 47, 48, 53, 117], "0x91000000": [16, 47, 48, 53], "0x92000000": [16, 47, 48, 53], "0x93000000": 53, "0x94000000": [48, 53], "0x95000000": 53, "0x98000000": [52, 72, 154, 159, 167, 198, 223], "0xa0000000": [10, 11, 12, 13, 15, 17, 18, 19, 21, 22, 23, 24, 25, 26, 27, 28, 29, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 48, 49, 50, 51, 54, 55, 56, 57, 58, 59, 60, 61, 63, 64, 67, 68, 69, 70, 71, 73, 74, 75, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 96, 97, 98, 99, 100, 101, 102, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 121, 122, 123, 124, 125, 127, 128, 129, 130, 132, 133, 134, 135, 136, 137, 138, 139, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 155, 156, 161, 162, 163, 164, 165, 166, 168, 169, 170, 171, 173, 174, 175, 176, 177, 178, 179, 180, 181, 183, 184, 185, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 224, 225, 226], "0xa0000100": 176, "0xa0001000": [134, 135, 145], "0xa0002000": [134, 135, 145], "0xa0003000": 135, "0xa0004000": 135, "0xa0010000": [19, 86, 112, 177, 187, 213], "0xa0020000": [177, 187, 213], "0xa0030000": [177, 213], "0xa0040000": 177, "0xa0100000": [109, 142, 147, 173, 214], "0xa0200000": 147, "0xa0300000": 147, "0xa0400000": [20, 147], "0xa0872c00": 69, "0xa1000000": [13, 29, 33, 34, 58, 60, 61, 62, 75, 90, 97, 101, 102, 111, 114, 116, 117, 121, 122, 123, 127, 128, 129, 130, 131, 136, 137, 143, 152, 161, 165, 170, 180, 190, 195, 197, 202, 205, 206, 207, 208, 222], "0xa1001000": 75, "0xa2000000": [29, 33, 45, 58, 60, 62, 97, 101, 102, 114, 116, 121, 123, 128, 131, 136, 137, 161, 170, 180, 199, 205, 207, 208], "0xa3000000": [29, 60, 62, 102, 117, 121, 131, 136, 137], "0xa4000000": [29, 45, 60, 62, 102, 131, 199], "0xa5000000": [29, 60, 62, 131], "0xa6000000": [60, 62, 131], "0xa7000000": [60, 62, 131], "0xa8000000": [52, 62, 159, 160, 167], "0xa8010000": 159, "0xa8020000": 159, "0xa8200000": 52, "0xa8410000": 52, "0xa8480000": 154, "0xa8483000": 154, "0xa8484000": 154, "0xa8485000": 154, "0xa8486000": 154, "0xa8490000": 154, "0xab000000": [62, 131], "0xab100000": [62, 131], "0xab200000": [62, 131], "0xab300000": [62, 131], "0xab400000": [62, 131], "0xab500000": [62, 131], "0xab600000": [62, 131], "0xab700000": [62, 131], "0xab800000": [62, 131], "0xb0000000": [11, 13, 15, 17, 18, 21, 23, 24, 25, 27, 28, 31, 32, 34, 35, 36, 37, 38, 39, 40, 41, 43, 44, 45, 49, 50, 54, 55, 56, 57, 59, 61, 63, 64, 67, 68, 70, 71, 74, 75, 83, 84, 87, 88, 89, 90, 92, 93, 94, 96, 102, 104, 105, 107, 109, 110, 111, 112, 116, 117, 122, 123, 124, 125, 127, 128, 129, 130, 132, 133, 138, 139, 140, 142, 143, 145, 146, 152, 153, 156, 163, 164, 165, 166, 168, 169, 170, 171, 174, 177, 178, 179, 180, 183, 184, 185, 188, 189, 190, 192, 193, 195, 196, 197, 199, 200, 201, 202, 203, 204, 205, 207, 208, 210, 211, 213, 214, 215, 216, 218, 219, 221, 222, 225], "0xb0001000": [49, 50, 87, 94], "0xb0002000": [49, 50, 87, 94], "0xb0003000": [87, 94], "0xb0010000": [177, 213], "0xb0020000": [177, 213], "0xb0030000": 177, "0xb0100000": [117, 207, 210], "0xb0200000": [117, 207, 210], "0xb0300000": 117, "0xb0400000": 117, "0xb0500000": 117, "0xb0600000": 117, "0xb0700000": 117, "0xb0800000": 117, "0xb0900000": 117, "0xb0b00000": 117, "0xb1000000": [34, 102, 123, 128, 166, 192, 193, 216], "0xb2000000": 102, "0xb8000000": [159, 167], "0xc0000000": [10, 11, 12, 14, 15, 17, 18, 19, 22, 23, 24, 25, 26, 33, 34, 37, 39, 40, 42, 49, 50, 51, 56, 57, 58, 63, 64, 67, 69, 70, 73, 78, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 91, 92, 93, 94, 96, 97, 98, 99, 100, 101, 104, 105, 108, 113, 114, 115, 117, 124, 125, 134, 135, 137, 138, 144, 155, 161, 162, 166, 168, 169, 170, 171, 173, 174, 175, 179, 181, 182, 186, 187, 188, 191, 194, 200, 203, 217, 224, 225], "0xc0001000": 134, "0xc0010000": 187, "0xc0020000": 187, "0xc0030000": 187, "0xc0100000": [117, 170, 188, 224], "0xc0200000": [117, 170], "0xc1000000": [14, 34, 182, 186], "0xc2000000": [14, 182, 186], "0xc3000000": [14, 182, 186], "0xc4000000": [182, 186], "0xc5000000": [182, 186], "0xc8000000": [159, 160, 167], "0xc8020000": 167, "0xc8040000": 167, "0xd0000000": [15, 24, 25, 34, 39, 40, 96, 101, 113, 125, 181, 225], "0xd1000000": 101, "0xe0000000": [24, 25, 34, 39, 40, 96], "0xff": [10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 21, 22, 23, 24, 25, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 46, 49, 50, 51, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 78, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 96, 97, 98, 99, 100, 101, 102, 104, 105, 106, 107, 108, 109, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 155, 158, 161, 162, 163, 164, 165, 166, 169, 170, 171, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 193, 194, 195, 196, 197, 201, 202, 203, 205, 206, 207, 208, 210, 211, 213, 214, 216, 217, 218, 219, 220, 221, 222, 224, 225, 226], "10": [0, 7, 8, 16, 20, 41, 45, 47, 55, 56, 57, 59, 60, 72, 74, 75, 88, 89, 124, 125, 141, 159, 160, 162, 176, 183, 184, 190, 192, 193, 198, 199, 211, 213, 215, 218, 219, 221, 223, 233], "100": [57, 60, 75, 83, 88, 96, 98, 106, 136, 183, 184, 186, 191, 216], "1000": [11, 12, 17, 27, 28, 46, 67, 70, 73, 79, 82, 83, 99, 103, 104, 105, 106, 108, 132, 133, 134, 135, 138, 150, 153, 160, 164, 165, 166, 174, 179, 180, 182, 186, 195, 196, 197, 202, 205, 207, 208, 225], "10000": 129, "1001": 106, "1024": [10, 13, 19, 39, 40, 43, 44, 46, 48, 51, 61, 63, 64, 66, 68, 71, 74, 78, 84, 87, 92, 93, 100, 101, 107, 109, 110, 111, 112, 122, 127, 130, 137, 140, 142, 143, 146, 148, 149, 152, 155, 156, 161, 163, 173, 175, 194, 220, 226], "10pt": [69, 96], "11": [4, 20, 55, 124, 125, 141, 190, 218, 219, 233], "114": 0, "11f": 66, "12": [20, 24, 25, 45, 55, 62, 124, 125, 131, 141, 170, 178, 199, 201, 215, 218, 219, 233], "120": [176, 190, 201], "1234": [148, 149, 220], "127": [31, 66, 74, 146], "128": [0, 29, 32, 33, 66, 74, 81, 116, 121, 146, 178, 182, 222, 224, 235], "128x128": 0, "12f": 66, "12pt": 69, "13": [18, 20, 124, 125, 218, 219], "139": 13, "14": [20, 32, 62, 124, 125, 131, 206, 218, 219, 233], "14f": 46, "15": [20, 45, 124, 125, 218, 219], "150": 193, "150528": 217, "1536": 4, "16": [20, 30, 31, 32, 38, 54, 58, 59, 86, 88, 90, 94, 101, 116, 124, 125, 126, 145, 178, 215, 218, 219, 222, 235], "16000": 46, "17": [20, 58, 125], "1733": 4, "18": [4, 20], "183": 18, "19": [20, 145], "1917": 60, "1d": 37, "1e": [14, 15, 23, 33, 87, 94, 97, 101, 114, 142, 171], "1f": [60, 66], "1i": 4, "1j": 4, "1x1": [49, 50], "20": [4, 20, 41, 56, 117, 178, 190, 192, 210, 215], "200": 193, "2000": [14, 75, 117], "2016": 12, "2019": 233, "2023": 4, "2048": [4, 15, 16, 23, 42, 47, 171], "21": 20, "224": [31, 37, 80, 81, 85, 217], "23": [20, 141], "24": [170, 189, 215], "25": [63, 64, 144], "255": [0, 74, 75], "256": [22, 80, 97, 102, 114, 115, 123, 128], "28": 141, "29": [20, 141], "299": 0, "2_": [94, 97], "2d": 31, "2f": [15, 23, 63, 64, 171], "2i": 4, "2j": 4, "2x2": [18, 231], "30": [16, 47, 145, 167, 233], "300": [56, 191], "3072": 4, "31": [20, 141], "32": [0, 20, 29, 30, 31, 34, 81, 87, 94, 116, 118, 119, 120, 124, 126, 134, 135, 161, 162, 178, 182, 186, 214, 222, 233, 235], "32768": 0, "33": 177, "3328": 4, "3414": 4, "36": 94, "3f": [15, 23, 171], "3j": [6, 7, 8], "40": 213, "400": [35, 36, 88, 126, 176], "4000": [30, 126], "4096": [4, 15, 23, 43, 44, 63, 64, 68, 71, 84, 92, 93, 107, 109, 112, 146, 163, 171], "47": [4, 18], "480000": 130, "49": 94, "4f": [15, 87, 94], "4i": 4, "4j": 4, "4x4": [218, 219], "50": [0, 4, 41, 56, 60, 88, 96, 124, 125, 136], "50176": 217, "512": [87, 121, 129, 182], "5678": 220, "587": 0, "5e": [15, 23, 171], "5f": [33, 46, 57, 60, 63, 64, 87, 94, 97, 106, 143, 145, 150, 202], "60": [145, 190, 210], "64": [0, 29, 32, 33, 34, 87, 94, 97, 102, 114, 116, 162, 178, 186, 191, 214, 216, 222, 224, 233, 235], "6678": 145, "6678e": 229, "6900": 4, "6f": [15, 114], "6pt": [12, 67, 117], "7004": [91, 139, 145, 229], "7062": 4, "71": 206, "75": 157, "75f": 115, "789": 79, "78ne": [13, 45, 74, 75, 88, 89, 110, 111, 116, 122, 127, 129, 130, 140, 156, 168, 169, 178, 192, 193, 199, 215], "80": [88, 193], "800": [88, 126], "800000": 14, "88": 12, "8f": [14, 15], "96": 115, "960000": 75, "960001": [74, 110, 111, 122, 127, 140, 156], "999": 144, "999f": [14, 15], "99f": 23, "9f": [14, 15, 23, 171], "_1": 214, "_2": 214, "__init__": [0, 4, 231], "__main__": 0, "__name__": 0, "_addr": 158, "_area": 136, "_at": 210, "_class": 182, "_corner": 158, "_count": [55, 147], "_data": [189, 208, 209, 214], "_decay": 171, "_dim": [88, 215], "_dim0": 210, "_end": 162, "_h": [32, 124, 125, 162], "_height": [158, 162], "_i": [34, 43, 44, 55, 63, 64, 68, 71, 84, 92, 93, 100, 101, 102, 106, 107, 109, 112, 123, 128, 131, 134, 144, 163, 164, 165, 170, 180, 195, 196, 197, 202], "_id": 189, "_index": [188, 189], "_indic": [188, 193], "_input": [81, 199], "_j": [34, 101, 102, 113, 181], "_k": [69, 147], "_l": [124, 125], "_len": 117, "_lerp": 158, "_list": 211, "_max": [74, 75, 222], "_maximum": 67, "_min": [74, 75], "_min_c": 75, "_mode": 67, "_n": [40, 214], "_ndim": [76, 77, 88, 161], "_norm": 69, "_num": [75, 215], "_of": 182, "_offset": 158, "_output": [81, 162], "_p": 235, "_prod": 67, "_rank": 188, "_rate": [23, 171], "_regul": 69, "_relu": [100, 103], "_relu6": 100, "_s": 235, "_scalar": 190, "_scale": [73, 158], "_segment": 189, "_shape": [81, 185, 189, 199, 200], "_size": [47, 69, 75, 76, 77, 96, 117, 161, 182, 189, 192], "_start": 162, "_stride": 188, "_sum": 67, "_t": [15, 118, 119, 120, 171], "_tensor": [205, 210, 211], "_threshold": 136, "_u": [124, 125], "_val": [12, 31, 74], "_val_c": 75, "_valu": [139, 187, 190], "_w": [32, 124, 125, 162], "_width": [158, 162], "_x": [90, 158, 204], "_y": [158, 204], "a_": 86, "a_i": 6, "abs": [4, 65, 227, 228, 229], "absgrad": [65, 227, 228, 229], "ac": 130, "accu_": 23, "accu_t": 23, "accumul": [0, 23, 171], "accuraci": 0, "act": 121, "activ": [31, 65, 86, 227, 228, 229, 233], "activation_non": 86, "activation_relu": 86, "activation_relu6": 86, "activation_typ": [20, 86, 121], "activationgrad": [65, 227, 228, 229], "activationtype_hsigmoid": 20, "activationtype_hswish": 20, "activationtype_no_activ": 20, "activationtype_relu": 20, "activationtype_relu6": 20, "activationtype_sigmoid": 20, "activationtype_softplus": 20, "activationtype_swish": 20, "activationtype_tanh": 20, "ad": 130, "adam": [0, 15, 65, 227, 228, 229], "adamweightdecay": [65, 227, 228], "adapthisteq": 4, "add": [167, 233], "add_channel": 0, "adder": [65, 227, 228], "adderfus": 229, "adderparamet": 16, "addfus": [65, 227, 228, 229], "addgrad": [65, 227, 228, 229], "addn": [65, 227, 228, 229], "affin": [65, 227, 228, 229], "after": 0, "ai": [0, 5, 230], "align": [12, 15, 23, 35, 95, 117, 147, 158, 171, 174], "align_corn": [157, 158], "all": [4, 22, 65, 227, 228, 229], "all_class_index": 60, "all_class_scor": 60, "all_freq": 126, "allgath": [65, 227, 228, 229], "along": [113, 181, 224], "alpha": [4, 12, 13, 17, 68, 103, 115, 202], "am": [14, 16, 47, 48, 56, 95, 103, 154, 167, 182, 186], "amd64": 233, "amzipp": 229, "anaconda": 233, "anaconda3": 233, "anchor": 60, "and": [0, 21, 54, 92, 93, 112, 159], "ani": 233, "anytype_crop_anycor": 52, "anytype_expand_dims_anycor": 72, "anytype_fillv2_": 79, "anytype_fillv2_p": 79, "anytype_reverse_sequence_anycor": 159, "anytype_reversev2_anycor": 160, "anytype_squeeze_anycor": 198, "anytype_unsqueeze_anycor": 223, "api": [4, 227, 228, 230, 239], "app": 0, "append": 4, "applymomentum": [65, 227, 228, 229], "approxim": 12, "arang": [4, 6, 7, 8], "arcsin": 229, "arg": [12, 125], "arg_el": [24, 25], "argc": [10, 11, 12, 13, 14, 17, 18, 19, 21, 22, 24, 25, 26, 27, 28, 29, 30, 31, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 46, 51, 54, 55, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 78, 80, 81, 82, 83, 84, 86, 90, 91, 92, 93, 96, 97, 99, 100, 102, 104, 105, 106, 107, 108, 109, 110, 111, 112, 114, 115, 117, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 142, 143, 146, 147, 148, 149, 150, 151, 152, 153, 156, 158, 161, 163, 164, 165, 166, 170, 172, 173, 174, 175, 176, 177, 179, 180, 182, 183, 184, 185, 186, 187, 188, 189, 190, 192, 194, 195, 196, 197, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 212, 213, 214, 216, 218, 219, 220, 221, 222, 224, 225, 226], "argmax": [0, 25, 65, 227, 228], "argmax_indic": 0, "argmaxfus": 229, "argmin": [65, 227, 228], "argminfus": 229, "argv": [10, 11, 12, 13, 14, 17, 18, 19, 21, 22, 24, 25, 26, 27, 28, 29, 30, 31, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 46, 51, 54, 55, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 78, 80, 81, 82, 83, 84, 86, 90, 91, 92, 93, 96, 97, 99, 100, 102, 104, 105, 106, 107, 108, 109, 110, 111, 112, 114, 115, 117, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 142, 143, 146, 147, 148, 149, 150, 151, 152, 153, 156, 158, 161, 163, 164, 165, 166, 170, 172, 173, 174, 175, 176, 177, 179, 180, 182, 183, 184, 185, 186, 187, 188, 189, 190, 192, 194, 195, 196, 197, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 212, 213, 214, 216, 218, 219, 220, 221, 222, 224, 225, 226], "arithmetic_infershap": 172, "arithmeticparamet": 172, "arm": [231, 233], "arm_toolchain": 231, "array": [0, 4, 145, 160], "array_input": [104, 105], "as": [0, 4, 6, 7, 8, 231], "asnumpi": 0, "assert": [65, 227, 228, 229], "assert_": 26, "assign": [65, 227, 228, 229], "assignadd": [65, 227, 228], "astyp": [0, 6, 7, 8], "at": 159, "attent": [65, 227, 228], "audio": 30, "audiospectrogram": [65, 227, 228], "auto": 0, "averag": 31, "avgpool": [65, 227, 228], "avgpoolfus": 229, "avgpoolgrad": 229, "avgpoolinggrad": [65, 227, 228], "axe": 154, "axi": [0, 4, 24, 25, 45, 52, 66, 88, 113, 139, 154, 159, 160, 167, 181, 192, 193, 198, 199, 216, 224], "axis_": [167, 216], "axis_dim": 54, "axis_flag": 160, "axis_flag_": 160, "axis_ndim": 160, "axis_ndim_": 160, "axis_num": 66, "axis_s": [21, 113, 181, 186], "axis_sizes_": 154, "azimuthfftfftshift": 229, "azimuthifft": 229, "b_": [86, 95, 117], "b_f": 117, "b_g": 117, "b_h": 35, "b_i": [6, 117], "b_o": 117, "b_w": 35, "backprop": [49, 50], "backward": [7, 8, 76, 77, 131, 161, 165, 180, 195, 201, 203], "bat": 231, "batch": [0, 31, 32, 33, 34, 35, 36, 47, 48, 49, 50, 58, 59, 76, 77, 85, 87, 88, 94, 95, 97, 102, 114, 117, 124, 125, 134, 135, 136, 161, 182, 183, 184, 186], "batch_": [95, 117], "batch_dim": [88, 159], "batch_dim_": 159, "batch_index": 162, "batch_num": 136, "batch_siz": [0, 29, 95, 134, 158, 182, 186], "batchnorm": [65, 227, 228, 229], "batchnormgrad": [65, 227, 228, 229], "batchtospac": [65, 227, 228, 229], "batchtospacend": [65, 227, 228, 229], "bc": 130, "bd": 130, "be": [37, 38, 159], "been": 233, "befor": [0, 159], "begin": [0, 10, 11, 12, 15, 21, 23, 26, 35, 54, 55, 67, 68, 69, 70, 73, 95, 96, 99, 100, 103, 104, 105, 112, 117, 123, 125, 128, 135, 138, 139, 144, 147, 158, 170, 171, 173, 174, 178, 179, 180, 190, 200, 201, 204, 218, 219, 225], "begin_": 178, "begin_0": 178, "begin_1": 178, "beta": [34, 97, 101, 102, 114, 115, 179, 180], "beta1": [14, 15], "beta1_pow": 14, "beta1_power_v": 14, "beta2": [14, 15], "beta2_pow": 14, "beta2_power_v": 14, "beta_1": [14, 15], "beta_2": [14, 15], "beta_c": [97, 114], "beta_data": 101, "beta_i": 101, "between": 159, "bi": 130, "bias": [16, 33, 34, 37, 47, 48, 86, 115, 117, 121, 166], "bias_": 86, "bias_data": [16, 47, 48], "bias_i": 166, "biasadd": [65, 227, 228, 229], "biasaddgrad": [65, 227, 228, 229], "bidirect": 95, "bidirectional_": [95, 117], "big": 124, "bigg": 124, "bigl": [12, 146], "bigr": [12, 146], "bilinear": 157, "bin": 162, "binari": 233, "binarycrossentropi": [65, 227, 228, 229], "binarycrossentropygrad": [65, 227, 228, 229], "bit": 116, "bitrev": 229, "bits_per_hash": 116, "blacklist": 55, "blank": 177, "blank_char": 177, "block": [36, 59, 185], "block_h": [35, 36, 183, 184], "block_num": 102, "block_siz": [35, 36, 59, 102, 183, 184], "block_w": [35, 36, 183, 184], "bmod": 55, "bool": [11, 17, 23, 26, 30, 60, 67, 69, 70, 74, 75, 76, 77, 92, 93, 104, 105, 117, 126, 136, 138, 142, 161, 170, 171, 204, 216, 225, 229], "bordertyp": [30, 126], "bottom": [35, 36, 183, 184], "bottom_i": 158, "box": [53, 136, 145, 149], "box_i": 136, "box_idx": 53, "box_index": 53, "box_j": 136, "box_num": 136, "boxes_scal": 60, "boxes_zero_point": 60, "brdm_2": 0, "break": 154, "broadcast": [92, 93, 104, 105, 142], "broadcastto": [65, 227, 228, 229], "bsearch": 96, "btr_60": 0, "buffer": [95, 117], "buffer_size_": [16, 47, 48, 49, 50], "build": [0, 231], "by": 160, "c128": [130, 156, 235], "c128_abs_": 10, "c128_abs_p": 10, "c128_add_": 235, "c128_add_p": 235, "c128_addext_": 17, "c128_addext_p": 17, "c128_addn_": 19, "c128_addn_p": 19, "c128_addrelu6_": 17, "c128_addrelu6_p": 17, "c128_addrelu_": 17, "c128_addrelu_p": 17, "c128_allgather_": 22, "c128_allgather_p": 22, "c128_assign_": 27, "c128_assign_p": 27, "c128_assignadd_": 28, "c128_assignadd_p": 28, "c128_batchtospace_": 35, "c128_batchtospace_p": 35, "c128_batchtospacend_": 36, "c128_batchtospacend_p": 36, "c128_biasadd_": 37, "c128_biasadd_p": 37, "c128_broadcastto_": 41, "c128_broadcastto_p": 41, "c128_concat_": 45, "c128_concat_p": 45, "c128_constant_of_shape_": 46, "c128_constant_of_shape_p": 46, "c128_cumsum_": 54, "c128_cumsum_p": 54, "c128_depthtospace_": 59, "c128_depthtospace_p": 59, "c128_div_fusion_": 61, "c128_div_fusion_p": 61, "c128_eltwise_": 67, "c128_eltwise_p": 67, "c128_equal_": 70, "c128_equal_p": 70, "c128_expfusion_": 73, "c128_expfusion_p": 73, "c128_extract_features_": 55, "c128_extract_features_p": 55, "c128_fill_": 78, "c128_fill_p": 78, "c128_formattranspose_": 85, "c128_formattranspose_p": 85, "c128_gather_": 88, "c128_gather_nd_": 89, "c128_gather_nd_p": 89, "c128_gather_p": 88, "c128_gatherd_": 90, "c128_gatherd_p": 90, "c128_isfinite_": 99, "c128_isfinite_p": 99, "c128_matmulfusion_": 121, "c128_matmulfusion_p": 121, "c128_mul_": 130, "c128_mul_p": 130, "c128_neg_": 132, "c128_neg_grad_": 133, "c128_neg_grad_p": 133, "c128_neg_p": 132, "c128_nonzero_": 137, "c128_nonzero_p": 137, "c128_not_equal_": 138, "c128_not_equal_p": 138, "c128_onehot_": 139, "c128_onehot_p": 139, "c128_ones_like_": 140, "c128_ones_like_p": 140, "c128_padfusion_": 141, "c128_padfusion_p": 141, "c128_real_div_": 152, "c128_real_div_p": 152, "c128_reciprocal_": 153, "c128_reciprocal_p": 153, "c128_reduceall_": 21, "c128_reduceall_p": 21, "c128_reshape_": 156, "c128_reshape_p": 156, "c128_rfft_p": 161, "c128_rfft_s": 161, "c128_rsqrt_p": 164, "c128_rsqrt_s": 164, "c128_scatter_elements_": 167, "c128_scatter_elements_p": 167, "c128_scatter_nd_": 168, "c128_scatter_nd_p": 168, "c128_scatter_nd_update_": 169, "c128_scatter_nd_update_p": 169, "c128_select_": 170, "c128_select_p": 170, "c128_slice_": 178, "c128_slice_p": 178, "c128_spacetobatch_": 183, "c128_spacetobatch_p": 183, "c128_spacetobatchnd_": 184, "c128_spacetobatchnd_p": 184, "c128_spacetodepth_": 185, "c128_spacetodepth_p": 185, "c128_sparsefillemptyrows_": 187, "c128_sparsefillemptyrows_p": 187, "c128_sparsesegmentsum_": 189, "c128_sparsesegmentsum_p": 189, "c128_sparsetodense_": 190, "c128_sparsetodense_p": 190, "c128_splice_": 191, "c128_splice_p": 191, "c128_split_": 192, "c128_split_p": 192, "c128_split_with_overlap_": 193, "c128_split_with_overlap_p": 193, "c128_sqrt_p": 194, "c128_sqrt_s": 194, "c128_sqrtgrad_": 195, "c128_sqrtgrad_p": 195, "c128_square_": 196, "c128_square_p": 196, "c128_squaredifference_": 197, "c128_squaredifference_p": 197, "c128_stack_": 199, "c128_stack_p": 199, "c128_subext_": 202, "c128_subext_p": 202, "c128_subrelu6_": 202, "c128_subrelu6_p": 202, "c128_subrelu_": 202, "c128_subrelu_p": 202, "c128_tensor_scatter_add_": 206, "c128_tensor_scatter_add_p": 206, "c128_tensorarrayread_": 208, "c128_tensorarrayread_p": 208, "c128_tensorlistfromtensor_": 210, "c128_tensorlistfromtensor_p": 210, "c128_tile_": 215, "c128_tile_p": 215, "c128_transpose_": 217, "c128_transpose_p": 217, "c128_tril_": 218, "c128_tril_p": 218, "c128_triu_": 219, "c128_triu_p": 219, "c128_unique_": 221, "c128_unique_p": 221, "c128_unsorted_segment_sum_": 222, "c128_unsorted_segment_sum_p": 222, "c128_where_": 225, "c128_where_p": 225, "c128_zerolike_": 226, "c128_zerolike_p": 226, "c64": [46, 130, 156, 235], "c64_abs_": 10, "c64_abs_p": 10, "c64_add_": 235, "c64_add_p": 235, "c64_addext_": 17, "c64_addext_p": 17, "c64_addn_": 19, "c64_addn_p": 19, "c64_addrelu6_": 17, "c64_addrelu6_p": 17, "c64_addrelu_": 17, "c64_addrelu_p": 17, "c64_allgather_": 22, "c64_allgather_p": 22, "c64_assign_": 27, "c64_assign_p": 27, "c64_assignadd_": 28, "c64_assignadd_p": 28, "c64_batchtospace_": 35, "c64_batchtospace_p": 35, "c64_batchtospacend_": 36, "c64_batchtospacend_p": 36, "c64_biasadd_": 37, "c64_biasadd_p": 37, "c64_broadcastto_": 41, "c64_broadcastto_p": 41, "c64_concat_": 45, "c64_concat_p": 45, "c64_constant_of_shape_": 46, "c64_constant_of_shape_p": 46, "c64_cumsum_": 54, "c64_cumsum_p": 54, "c64_depthtospace_": 59, "c64_depthtospace_p": 59, "c64_div_fusion_": 61, "c64_div_fusion_p": 61, "c64_eltwise_": 67, "c64_eltwise_p": 67, "c64_equal_": 70, "c64_equal_p": 70, "c64_expfusion_": 73, "c64_expfusion_p": 73, "c64_extract_features_": 55, "c64_extract_features_p": 55, "c64_fill_": 78, "c64_fill_p": 78, "c64_formattranspose_": 85, "c64_formattranspose_p": 85, "c64_gather_": 88, "c64_gather_nd_": 89, "c64_gather_nd_p": 89, "c64_gather_p": 88, "c64_gatherd_": 90, "c64_gatherd_p": 90, "c64_isfinite_": 99, "c64_isfinite_p": 99, "c64_matmulfusion_": 121, "c64_matmulfusion_p": 121, "c64_mul_": 130, "c64_mul_p": 130, "c64_neg_": 132, "c64_neg_grad_": 133, "c64_neg_grad_p": 133, "c64_neg_p": 132, "c64_nonzero_": 137, "c64_nonzero_p": 137, "c64_not_equal_": 138, "c64_not_equal_p": 138, "c64_onehot_": 139, "c64_onehot_p": 139, "c64_ones_like_": 140, "c64_ones_like_p": 140, "c64_padfusion_": 141, "c64_padfusion_p": 141, "c64_real_div_": 152, "c64_real_div_p": 152, "c64_reciprocal_": 153, "c64_reciprocal_p": 153, "c64_reduceall_": 21, "c64_reduceall_p": 21, "c64_reshape_": 156, "c64_reshape_p": 156, "c64_rfft_p": 161, "c64_rfft_s": 161, "c64_rsqrt_p": 164, "c64_rsqrt_s": 164, "c64_scatter_elements_": 167, "c64_scatter_elements_p": 167, "c64_scatter_nd_": 168, "c64_scatter_nd_p": 168, "c64_scatter_nd_update_": 169, "c64_scatter_nd_update_p": 169, "c64_select_": 170, "c64_select_p": 170, "c64_slice_": 178, "c64_slice_p": 178, "c64_spacetobatch_": 183, "c64_spacetobatch_p": 183, "c64_spacetobatchnd_": 184, "c64_spacetobatchnd_p": 184, "c64_spacetodepth_": 185, "c64_spacetodepth_p": 185, "c64_sparsefillemptyrows_": 187, "c64_sparsefillemptyrows_p": 187, "c64_sparsesegmentsum_": 189, "c64_sparsesegmentsum_p": 189, "c64_sparsetodense_": 190, "c64_sparsetodense_p": 190, "c64_splice_": 191, "c64_splice_p": 191, "c64_split_": 192, "c64_split_p": 192, "c64_split_with_overlap_": 193, "c64_split_with_overlap_p": 193, "c64_sqrt_p": 194, "c64_sqrt_s": 194, "c64_sqrtgrad_": 195, "c64_sqrtgrad_p": 195, "c64_square_": 196, "c64_square_p": 196, "c64_squaredifference_": 197, "c64_squaredifference_p": 197, "c64_stack_": 199, "c64_stack_p": 199, "c64_subext_": 202, "c64_subext_p": 202, "c64_subrelu6_": 202, "c64_subrelu6_p": 202, "c64_subrelu_": 202, "c64_subrelu_p": 202, "c64_tensor_scatter_add_": 206, "c64_tensor_scatter_add_p": 206, "c64_tensorarrayread_": 208, "c64_tensorarrayread_p": 208, "c64_tensorlistfromtensor_": 210, "c64_tensorlistfromtensor_p": 210, "c64_tile_": 215, "c64_tile_p": 215, "c64_transpose_": 217, "c64_transpose_p": 217, "c64_tril_": 218, "c64_tril_p": 218, "c64_triu_": 219, "c64_triu_p": 219, "c64_unique_": 221, "c64_unique_p": 221, "c64_unsorted_segment_sum_": 222, "c64_unsorted_segment_sum_p": 222, "c64_where_": 225, "c64_where_p": 225, "c64_zerolike_": 226, "c64_zerolike_p": 226, "c6678": 71, "c_": [16, 35, 47, 48, 117], "c_0": 117, "c_str": 0, "c_t": [117, 118, 119, 120], "cal_num_per_thread": 200, "calcul": [159, 233], "caled_num": 200, "call": 229, "callback": 0, "cand": 136, "candid": 136, "case": [10, 11, 12, 21, 23, 26, 54, 55, 67, 68, 69, 70, 73, 96, 99, 100, 103, 104, 105, 112, 123, 125, 128, 135, 138, 139, 144, 158, 170, 171, 173, 179, 180, 190, 204, 218, 219, 225], "cast": [4, 65, 227, 228, 229], "castc128toc128_": 42, "castc128toc128_p": 42, "castc64toc64_": 42, "castc64toc64_p": 42, "castdptodp_": 42, "castdptodp_p": 42, "castdptofp_": 42, "castdptofp_p": 42, "castfptofp_": 42, "castfptofp_p": 42, "castfptoint16_": 42, "castfptoint16_p": 42, "castfptoint32_": 42, "castfptoint32_p": 42, "castfptoint8_": 42, "castfptoint8_p": 42, "casti16tofp_": 42, "casti16tofp_p": 42, "casti16tointi16_": 42, "casti16tointi16_p": 42, "casti16tointi32_": 42, "casti16tointi32_p": 42, "casti16tointi8_": 42, "casti16tointi8_p": 42, "casti32tofp_": 42, "casti32tofp_p": 42, "casti32tointi16_": 42, "casti32tointi16_p": 42, "casti32tointi32_": 42, "casti32tointi32_p": 42, "casti32tointi8_": 42, "casti32tointi8_p": 42, "casti8tofp_": 42, "casti8tofp_p": 42, "casti8tointi16_": 42, "casti8tointi16_p": 42, "casti8tointi32_": 42, "casti8tointi32_p": 42, "casti8tointi8_": 42, "casti8tointi8_p": 42, "category_": 204, "cc": 0, "ccor": 47, "cd_run_param": 4, "cddata1": 4, "cddata2": 4, "cdot": [12, 13, 14, 15, 20, 23, 31, 32, 34, 39, 40, 58, 62, 63, 64, 67, 68, 69, 73, 84, 86, 87, 94, 97, 101, 102, 103, 106, 107, 114, 117, 118, 119, 120, 124, 125, 134, 135, 144, 147, 158, 166, 171, 174], "ceil": [65, 227, 228, 229], "cell": [0, 4, 117, 231], "cell_buff": 117, "cell_stat": 117, "cell_state_": 117, "celu": 12, "center": [30, 126, 136], "center_point_box": 136, "cepstral": 126, "cfar": 229, "channel": [0, 31, 32, 33, 34, 35, 36, 59, 75, 85, 87, 94, 97, 114, 115, 124, 125, 158, 183, 184, 185], "channel_num": 75, "char": [10, 11, 12, 13, 14, 17, 18, 19, 21, 22, 24, 25, 26, 27, 28, 29, 30, 31, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 46, 51, 54, 55, 56, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 78, 80, 81, 82, 83, 84, 86, 90, 91, 92, 93, 96, 97, 99, 100, 102, 104, 105, 106, 107, 108, 109, 110, 111, 112, 114, 115, 117, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 142, 143, 146, 147, 148, 149, 150, 151, 152, 153, 156, 158, 161, 163, 164, 165, 166, 170, 172, 173, 174, 175, 176, 177, 179, 180, 182, 183, 184, 185, 186, 187, 188, 189, 190, 192, 194, 195, 196, 197, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 212, 213, 214, 216, 218, 219, 220, 221, 222, 224, 225, 226], "check": [48, 154, 157, 159, 160], "check_output_hidden_st": 95, "check_ptr": [124, 125], "check_seq_len": 95, "check_seq_len_": 95, "chmod": 231, "cin_channel": 0, "circshift": 4, "ckpt": 0, "clamp": 74, "class": [0, 4, 6, 7, 8, 136, 186, 231], "class_label": 0, "class_nam": 0, "class_num": [134, 135, 136], "clip": [0, 12, 31, 65, 66, 124, 146, 227, 228, 229], "clip_by_valu": 0, "cliplimit": 4, "close": 4, "clx64": [85, 191, 217], "cmake": 231, "cmakelist": [0, 231], "cmd": 233, "cnn": 0, "cnn_model": 0, "coeffici": 126, "col": 0, "color": 0, "color_imag": 0, "color_rgb2gray": 0, "common_infershap": 172, "complex": [4, 121], "complex128": [6, 7, 8, 76, 77, 161, 235], "complex64": [4, 6, 7, 8, 46, 76, 77, 161, 235], "complexab": [4, 9, 227, 229, 233], "compon": 233, "compos": 0, "computestrid": 159, "concat": [65, 227, 228, 229], "condit": [170, 204, 225], "condition_i": 225, "connect": 0, "const": [0, 14, 15, 23, 32, 35, 36, 37, 41, 49, 50, 59, 60, 80, 87, 94, 103, 116, 117, 126, 134, 162, 171, 183, 184, 185, 217, 231], "constant": [4, 141], "constant_valu": [4, 141], "constantofshap": [65, 227, 228, 229], "construct": [0, 4, 231], "constscalar": 172, "consttensor": 172, "context": [0, 20, 231], "context_dim": 191, "context_s": 20, "contigu": [24, 25], "continu": 12, "conv1": 0, "conv2": 0, "conv2d": [0, 48, 49, 50, 65, 227, 228], "conv2dbackpropfilterfus": [65, 227, 228, 229], "conv2dbackpropinputfus": [65, 227, 228, 229], "conv2dfus": 229, "conv2dgradfilt": 49, "conv2dtranspos": [65, 227, 228], "conv2dtransposefus": 229, "conv3": 0, "conv_param": [16, 47, 48, 49, 50], "conv_paramet": [49, 50], "convert": 231, "convert_gray": 0, "convertcolor": 0, "converter_lit": 0, "convertmod": 0, "convertto": 0, "convertto2d": 0, "convolut": 58, "convparamet": [16, 47, 49, 50], "convquantparamet": [16, 47], "convtransposeparamet": 48, "coordinate_transform_mode_": 157, "coordinatetransformmod": 157, "copi": 159, "copy_elem_num_": 159, "copy_s": 213, "core_id": [16, 20, 47, 48, 52, 53, 79, 95, 154, 157, 159, 160, 167], "core_mask": [10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 205, 206, 207, 208, 209, 210, 211, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226], "core_num": [16, 47, 48, 52, 53, 72, 79, 95, 154, 157, 159, 160, 167, 198, 223], "corner": 136, "correct": 233, "correl": 47, "cos": [65, 227, 228, 229], "cosin": 51, "count": 221, "countelementafterdim": 159, "countelementbeforedim": 159, "cout": 0, "cplx128": [10, 17, 19, 21, 22, 27, 28, 35, 36, 37, 41, 42, 45, 46, 54, 55, 59, 61, 67, 70, 72, 73, 76, 77, 78, 79, 80, 85, 88, 89, 99, 130, 132, 133, 137, 138, 139, 140, 141, 151, 152, 153, 159, 160, 161, 164, 167, 168, 169, 170, 176, 178, 183, 184, 185, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 202, 206, 210, 215, 217, 218, 219, 221, 223, 224, 225, 226, 229], "cplx64": [10, 17, 19, 21, 22, 27, 28, 35, 36, 37, 41, 42, 45, 46, 54, 55, 59, 61, 67, 70, 72, 73, 76, 77, 78, 79, 80, 85, 88, 89, 99, 130, 132, 133, 137, 138, 139, 140, 141, 151, 152, 153, 159, 160, 161, 164, 167, 168, 169, 170, 176, 178, 183, 184, 185, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 202, 206, 208, 209, 210, 215, 217, 218, 219, 221, 223, 224, 225, 226, 229], "cpu": [0, 6, 7, 8, 229, 231, 233], "cpu_info": 0, "cpudeviceinfo": 0, "creat": 233, "create_dict_iter": 0, "crop": [35, 36, 65, 227, 228, 229], "cropandres": [65, 227, 228, 229], "cropandresizeparamet": 53, "cross": [47, 173, 182, 186], "crossentropyloss": 0, "cs": 4, "cubic": 157, "cubic_coeff": 157, "cubic_coeff_": 157, "cumsum": [65, 227, 228, 229], "cur_coord": 160, "cur_coord_": 160, "current": 233, "custom": 233, "custom_extract_featur": 55, "customextractfeatur": [65, 227, 228, 229], "customnorm": [65, 227, 228, 229], "customnormalize_": 56, "customnormalize_p": 56, "custompredict": [65, 227, 228, 229], "custompredict_": 57, "custompredict_p": 57, "cv": 0, "cv2": 0, "cv_32f": 0, "cvmulcv": 229, "cvmulcvifft": 229, "d_": [192, 199], "d_0": [192, 199], "d_1": [192, 199], "d_k": 29, "da_t": [118, 119, 120], "da_tmp_": [118, 119, 120], "dampen": 171, "data": [0, 4, 55, 177, 205, 207], "data0": 208, "data1": 208, "data2": 208, "data_": 204, "data_buffers_": 154, "data_format": 37, "data_handl": [0, 231], "data_it": 0, "data_path": 0, "data_s": [22, 35, 36, 41, 59, 155, 183, 184, 185, 224, 231], "data_typ": 211, "data_type_": [172, 204], "data_type_byt": 200, "datas": 0, "dataset": 0, "dataset_dir": 0, "dataset_sink_mod": 0, "db": 102, "dbia": [34, 38], "dbias_c": 38, "dc_": [118, 119, 120], "dc_t": [118, 119, 120], "dct": 126, "dct_mat": 126, "dct_type": 126, "ddr": [10, 11, 12, 14, 15, 17, 18, 21, 23, 24, 25, 27, 28, 29, 31, 32, 33, 35, 36, 37, 38, 39, 40, 41, 43, 44, 46, 49, 50, 51, 54, 56, 57, 59, 61, 63, 64, 67, 68, 69, 70, 71, 73, 80, 81, 82, 83, 84, 85, 87, 89, 90, 91, 92, 93, 94, 96, 97, 98, 99, 100, 101, 102, 104, 105, 106, 107, 108, 112, 113, 114, 115, 117, 121, 123, 128, 132, 133, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 152, 153, 154, 155, 156, 162, 163, 164, 165, 166, 168, 169, 170, 171, 173, 174, 175, 177, 179, 180, 181, 182, 183, 184, 185, 186, 188, 189, 190, 191, 195, 196, 197, 200, 201, 202, 203, 205, 207, 208, 210, 211, 213, 214, 215, 217, 218, 219, 220, 221, 222, 225, 226], "decay": 15, "decod": 0, "decoded_box": 60, "deconv2dgradfilt": 229, "deconvgradfilt": [65, 227, 228], "deconvolut": 58, "def": [0, 4, 231], "default": 187, "default_valu": 187, "defin": [0, 20, 67], "delegatemod": 0, "delta": [124, 125, 147, 150], "delta_k": 147, "dens": 0, "dense1": 0, "dense2": 0, "dense_row": 187, "depth": [59, 139], "depth_radius": 115, "depthtospac": [65, 227, 228, 229], "depthwis": [49, 50], "dequant": 146, "detectionpostprocess": [65, 227, 228, 229], "detectionpostprocessparamet": 60, "detections_per_class": 60, "device_list": 0, "device_target": [0, 233], "dfrac": [135, 173], "dg": 102, "dh_": [118, 119, 120], "dh_t": [118, 119, 120], "di": 130, "diff": 180, "different_aspect_ratio": 145, "different_aspect_ratios_s": 145, "differenti": 12, "dilat": [16, 47, 48, 49, 50], "dilation_h_": [16, 47, 48, 49, 50], "dilation_w_": [16, 47, 48, 49, 50], "dim": [4, 7, 8, 37, 90, 123, 128], "dim0": [190, 222], "dim1": [190, 222], "dim2": 190, "dim3": 190, "dim_size_": 216, "dim_strid": 137, "dir": [76, 77, 161], "directori": 0, "displaystyl": [69, 125], "distribut": 4, "divfus": [65, 227, 228, 229], "divgrad": [65, 131, 227, 228, 229], "divgrad1l": 62, "divgrad2l": 62, "dividor": [124, 125], "dividor2": [124, 125], "dl": 40, "dloss": 40, "dma": [22, 78, 117, 140, 169, 215], "dnum": 20, "doppler": 4, "doquantizefp32toint8": 66, "dot": [7, 8, 29, 45, 76, 77, 78, 90, 98, 106, 139, 147, 150, 161, 178, 187, 192, 193, 199], "doubl": [4, 10, 17, 19, 21, 22, 27, 28, 35, 36, 37, 41, 42, 43, 44, 45, 46, 51, 54, 55, 59, 61, 67, 70, 71, 73, 78, 82, 83, 84, 85, 88, 89, 90, 92, 93, 99, 104, 105, 108, 110, 111, 112, 116, 121, 122, 125, 127, 129, 130, 132, 133, 137, 138, 139, 140, 141, 142, 147, 150, 152, 153, 154, 155, 156, 163, 164, 166, 167, 168, 169, 170, 175, 178, 183, 184, 185, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 199, 202, 206, 210, 214, 215, 217, 218, 219, 221, 222, 225, 226], "down_slop": 126, "download": 233, "dp": [10, 51, 78, 85, 110, 111, 116, 122, 127, 129, 130, 155, 175, 187, 191, 194, 217, 226, 235], "dp_abs_": 10, "dp_abs_p": 10, "dp_add_": 235, "dp_add_p": 235, "dp_addext_": 17, "dp_addext_p": 17, "dp_addn_": 19, "dp_addn_p": 19, "dp_addrelu6_": 17, "dp_addrelu6_p": 17, "dp_addrelu_": 17, "dp_addrelu_p": 17, "dp_allgather_": 22, "dp_allgather_p": 22, "dp_and_": 112, "dp_and_p": 112, "dp_assign_": 27, "dp_assign_p": 27, "dp_assignadd_": 28, "dp_assignadd_p": 28, "dp_batchtospace_": 35, "dp_batchtospace_p": 35, "dp_batchtospacend_": 36, "dp_batchtospacend_p": 36, "dp_biasadd_": 37, "dp_biasadd_p": 37, "dp_broadcastto_": 41, "dp_broadcastto_p": 41, "dp_ceil_": 43, "dp_ceil_p": 43, "dp_clip_": 44, "dp_clip_p": 44, "dp_concat_": 45, "dp_concat_p": 45, "dp_constant_of_shape_": 46, "dp_constant_of_shape_p": 46, "dp_cos_": 51, "dp_cos_p": 51, "dp_cumsum_": 54, "dp_cumsum_p": 54, "dp_depthtospace_": 59, "dp_depthtospace_p": 59, "dp_div_fusion_": 61, "dp_div_fusion_p": 61, "dp_eltwise_": 67, "dp_eltwise_p": 67, "dp_equal_": 70, "dp_equal_p": 70, "dp_erf_": 71, "dp_erf_p": 71, "dp_expfusion_": 73, "dp_expfusion_p": 73, "dp_extract_features_": 55, "dp_extract_features_p": 55, "dp_fill_": 78, "dp_fill_p": 78, "dp_floor_": 82, "dp_floor_p": 82, "dp_floordiv_": 83, "dp_floordiv_p": 83, "dp_floormod_": 84, "dp_floormod_p": 84, "dp_formattranspose_": 85, "dp_formattranspose_p": 85, "dp_gather_": 88, "dp_gather_nd_": 89, "dp_gather_nd_p": 89, "dp_gather_p": 88, "dp_gatherd_": 90, "dp_gatherd_p": 90, "dp_greater_": 92, "dp_greaterequal_": 93, "dp_isfinite_": 99, "dp_isfinite_p": 99, "dp_less_": 104, "dp_less_p": 104, "dp_lessequal_": 105, "dp_lessequal_p": 105, "dp_log1p_": 108, "dp_log1p_p": 108, "dp_logical_not_": 110, "dp_logical_not_p": 110, "dp_logical_or_": 111, "dp_logical_or_p": 111, "dp_lsh_projection_": 116, "dp_lsh_projection_p": 116, "dp_matmulfusion_": 121, "dp_matmulfusion_p": 121, "dp_maximum_": 122, "dp_maximum_p": 122, "dp_maxpool_fusion_": 124, "dp_maxpool_fusion_p": 124, "dp_maxpool_grad_": 125, "dp_maxpool_grad_p": 125, "dp_minimum_": 127, "dp_minimum_p": 127, "dp_mod_": 129, "dp_mod_p": 129, "dp_mul_": 130, "dp_mul_p": 130, "dp_neg_": 132, "dp_neg_grad_": 133, "dp_neg_grad_p": 133, "dp_neg_p": 132, "dp_nonzero_": 137, "dp_nonzero_p": 137, "dp_not_equal_": 138, "dp_not_equal_p": 138, "dp_onehot_": 139, "dp_onehot_p": 139, "dp_ones_like_": 140, "dp_ones_like_p": 140, "dp_padfusion_": 141, "dp_padfusion_p": 141, "dp_pow_fusion_": 142, "dp_pow_fusion_p": 142, "dp_raggedrange_": 147, "dp_raggedrange_p": 147, "dp_range_": 150, "dp_range_p": 150, "dp_real_div_": 152, "dp_real_div_p": 152, "dp_reciprocal_": 153, "dp_reciprocal_p": 153, "dp_reduce_": 154, "dp_reduce_p": 154, "dp_reduceall_": 21, "dp_reduceall_p": 21, "dp_reducescatter_": 155, "dp_reducescatter_p": 155, "dp_reshape_": 156, "dp_reshape_p": 156, "dp_round_": 163, "dp_round_p": 163, "dp_rsqrt_p": 164, "dp_rsqrt_s": 164, "dp_scalefusion_": 166, "dp_scalefusion_p": 166, "dp_scatter_elements_": 167, "dp_scatter_elements_p": 167, "dp_scatter_nd_": 168, "dp_scatter_nd_p": 168, "dp_scatter_nd_update_": 169, "dp_scatter_nd_update_p": 169, "dp_select_": 170, "dp_select_p": 170, "dp_sin_": 175, "dp_sin_p": 175, "dp_slice_": 178, "dp_slice_p": 178, "dp_spacetobatch_": 183, "dp_spacetobatch_p": 183, "dp_spacetobatchnd_": 184, "dp_spacetobatchnd_p": 184, "dp_spacetodepth_": 185, "dp_spacetodepth_p": 185, "dp_sparsefillemptyrows_": 187, "dp_sparsefillemptyrows_p": 187, "dp_sparsesegmentsum_": 189, "dp_sparsesegmentsum_p": 189, "dp_sparsetodense_": 190, "dp_sparsetodense_p": 190, "dp_splice_": 191, "dp_splice_p": 191, "dp_split_": 192, "dp_split_p": 192, "dp_split_with_overlap_": 193, "dp_split_with_overlap_p": 193, "dp_sqrt_p": 194, "dp_sqrt_s": 194, "dp_sqrtgrad_": 195, "dp_sqrtgrad_p": 195, "dp_square_": 196, "dp_square_p": 196, "dp_squaredifference_": 197, "dp_squaredifference_p": 197, "dp_stack_": 199, "dp_stack_p": 199, "dp_subext_": 202, "dp_subext_p": 202, "dp_subrelu6_": 202, "dp_subrelu6_p": 202, "dp_subrelu_": 202, "dp_subrelu_p": 202, "dp_tensor_scatter_add_": 206, "dp_tensor_scatter_add_p": 206, "dp_tensorarrayread_": 208, "dp_tensorarrayread_p": 208, "dp_tensorlistfromtensor_": 210, "dp_tensorlistfromtensor_p": 210, "dp_tile_": 215, "dp_tile_p": 215, "dp_transpose_": 217, "dp_transpose_p": 217, "dp_tril_": 218, "dp_tril_p": 218, "dp_triu_": 219, "dp_triu_p": 219, "dp_unique_": 221, "dp_unique_p": 221, "dp_unsorted_segment_sum_": 222, "dp_unsorted_segment_sum_p": 222, "dp_where_": 225, "dp_where_p": 225, "dp_xxx_xxx": 235, "dp_zerolike_": 226, "dp_zerolike_p": 226, "dropout": [64, 65, 172, 227, 228, 229], "dropout_infershap": 172, "dropoutgrad": [65, 227, 228, 229], "ds": 0, "dscale": 34, "dsp": [0, 5, 72, 198, 223, 227, 228, 229, 230, 236, 239], "dsp_info": 0, "dst": [13, 27, 42, 44, 53, 72, 101, 144, 159, 160, 164, 196, 198, 211, 218, 219, 223], "dst_": [33, 86], "dst_col": 191, "dst_data": [10, 21, 51, 73, 82, 83, 85, 101, 144, 154, 166, 175, 191, 194], "dst_format": 85, "dst_i": [10, 13, 19, 27, 42, 51, 73, 78, 82, 83, 166, 173, 175, 194, 226], "dst_ptr": 0, "dst_row": 191, "dt": 71, "dtype": [0, 4, 231], "du": [118, 119, 120], "dw": [49, 58, 118, 119, 120], "dw_data": 58, "dx": [34, 40, 50, 64, 102, 165, 195, 201], "dx0": [123, 128, 131], "dx1": [18, 62, 123, 128, 131, 180, 203], "dx1_dim": 18, "dx2": [18, 62, 131, 203], "dx2_dim": 18, "dx_": [118, 119, 120], "dx_1": 62, "dx_2": 62, "dx_i": [34, 64], "dx_n": 40, "dx_shape": 201, "dx_t": [118, 119, 120], "dxx": 18, "dy": [18, 34, 38, 49, 50, 58, 62, 64, 102, 109, 123, 125, 128, 131, 165, 180, 195, 201, 203], "dy_b": 58, "dy_data": 58, "dy_dim": [18, 38, 203], "dy_i": [34, 64], "dy_j": 34, "dy_ptr": 125, "dy_siz": [62, 131], "dy_t": [118, 119, 120], "dynam": 66, "dynamic_param": [118, 119, 120], "dynamicqu": [65, 227, 228, 229], "dynamicquant_infershap": 172, "each": 224, "echo": 4, "echo1": 4, "echo2": 4, "echo_d1_mf": 4, "echo_d3_mf": 4, "echo_r": 4, "echo_s1": 4, "echo_s2": 4, "echo_s3": 4, "echo_s4": 4, "echo_s5": 4, "echo_shap": 4, "elem_cnt": 103, "elem_cnt_": [78, 80], "element": [0, 113, 159, 181, 221], "element_count": 0, "element_num": [66, 92, 93], "element_shap": 212, "elementnum": 0, "els": [0, 96, 142, 157], "eltwis": [65, 227, 228, 229], "eltwise_maximum": 67, "eltwise_mode_": 67, "eltwise_prod": 67, "eltwise_sum": 67, "elu": [12, 13, 65, 227, 228, 229], "elugrad": 13, "embeddinglookup": [65, 227, 228], "embeddinglookupfus": 229, "empti": [0, 187], "enabl": [92, 93], "encod": 139, "end": [0, 10, 11, 12, 14, 15, 21, 23, 26, 35, 46, 54, 55, 67, 68, 69, 70, 73, 95, 96, 99, 100, 103, 104, 105, 106, 112, 117, 123, 125, 128, 135, 138, 139, 144, 147, 158, 170, 171, 173, 174, 179, 180, 190, 193, 200, 204, 218, 219, 225], "end_idx": [32, 193], "end_indic": 193, "endl": 0, "entropi": [173, 182, 186], "eps": 14, "epsilon": [14, 15, 33, 34, 40, 87, 94, 97, 101, 102, 114, 137], "equal": [65, 227, 228, 229], "erf": [12, 65, 227, 228, 229], "error": [12, 71], "exampl": 55, "exclus": 54, "exe": 233, "exist": 96, "exp": [0, 4, 6, 7, 8, 73, 113, 142, 181], "exp_complex": [30, 126], "exp_x": 0, "expand_dim": 0, "expanddim": [65, 227, 228, 229], "expfus": [65, 227, 228, 229], "expon": 142, "exponent_0": 142, "exponent_i": 142, "exponenti": [4, 68], "export": [0, 4, 231], "export_cnn": 0, "export_connect": 0, "export_gray": 0, "ext": 35, "extens": 0, "extrapolation_valu": 53, "f0": 4, "f32_typecast": 0, "f_": [118, 119, 120], "f_diff": 126, "f_max": 126, "f_min": 126, "f_nc": 4, "f_pts": 126, "f_t": [117, 118, 119, 120], "fa": 4, "fa_axi": 4, "fab": 142, "factor": [76, 77, 161], "fail": 0, "fakequ": 75, "fakequantwithminmaxvar": [65, 227, 228], "fakequantwithminmaxvarsperchannel": [65, 227, 228], "fals": [0, 4, 12, 23, 26, 30, 39, 40, 54, 70, 74, 75, 76, 77, 92, 93, 104, 105, 110, 111, 112, 117, 126, 136, 138, 142, 161, 170, 171, 204, 216, 225], "fast": 60, "fast_run": 200, "fb": 126, "featur": 116, "feature_num": 116, "fft": [4, 8, 9, 30, 76, 77, 126, 161, 172, 227, 229], "fft1": 4, "fft_forward": [76, 77, 161], "fft_infershap": 172, "fft_invers": [76, 77, 161], "fft_nobitrev": 229, "fft_size": [76, 77, 161], "fft_size1": [76, 77, 161], "fft_size2": [76, 77, 161], "fft_time": 229, "fft_window": [30, 126], "fft_window_lat": [30, 126], "fftimag": [65, 227, 228], "fftreal": [65, 227, 228], "fftshift": [4, 229], "fftw_complex": 161, "figur": 4, "file_format": [0, 4, 231], "file_nam": [0, 4, 231], "fill": [65, 227, 228, 229], "filled_count": 187, "fillparamet": 78, "fillv2": [65, 227, 228], "filter": [16, 49], "filter_zp_ptr_": 47, "finish": 233, "flag": 12, "flatten": [0, 65, 81, 172, 227, 228, 229], "flatten_infershap": 172, "flattengrad": [65, 227, 228, 229], "flattenparamet": 80, "flip": 4, "flipud": 4, "float": [0, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 97, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 121, 122, 123, 124, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 152, 153, 154, 155, 156, 157, 158, 159, 160, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 173, 174, 175, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 206, 208, 210, 211, 213, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226], "float16": [136, 182, 186, 205, 235], "float32": [0, 4, 6, 46, 76, 77, 106, 161, 200, 205, 207, 231, 235], "float64": [6, 76, 77, 161, 235], "float_ep": 137, "floatarraytovector": 0, "floor": [65, 84, 227, 228, 229], "floordiv": [65, 227, 228, 229], "floorf": [82, 83], "floormod": [65, 227, 228, 229], "fmap_h": 145, "fmap_w": 145, "fmk": 0, "fmod": 129, "fnv1a": 116, "for": [0, 20, 24, 25, 31, 78, 98, 100, 113, 141, 154, 159, 160, 167, 170, 181, 189, 224, 233], "foral": [69, 96], "forg": 233, "forget": 117, "format": [37, 38, 158], "format_": [172, 204], "format_nchw": [172, 204], "formated_input_shap": 141, "formated_output_shap": 141, "formated_pad": 141, "formattranspos": [65, 227, 228], "forward": [7, 8, 34, 40, 64, 76, 77, 161], "forward_index": 191, "forward_indexes_dim": 191, "fourpointinterpolatori": 229, "fp": [10, 51, 58, 74, 75, 78, 85, 86, 101, 110, 111, 113, 116, 118, 119, 120, 122, 127, 129, 130, 134, 135, 144, 155, 162, 175, 181, 187, 191, 194, 217, 226, 235], "fp16": [11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 27, 28, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 52, 53, 54, 55, 59, 60, 61, 62, 63, 64, 67, 68, 69, 70, 71, 72, 73, 74, 75, 79, 80, 81, 82, 83, 84, 87, 88, 89, 91, 92, 93, 94, 95, 97, 99, 100, 102, 103, 104, 105, 107, 108, 109, 110, 111, 112, 114, 115, 116, 117, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 136, 137, 138, 139, 140, 141, 142, 143, 145, 146, 148, 149, 150, 152, 153, 154, 156, 157, 158, 159, 160, 163, 164, 165, 166, 167, 168, 169, 170, 171, 173, 174, 178, 179, 180, 182, 183, 184, 185, 186, 189, 190, 192, 193, 195, 196, 197, 198, 199, 200, 201, 202, 203, 206, 208, 209, 210, 215, 216, 218, 219, 220, 221, 223, 224, 225, 229], "fp32": [11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 52, 53, 54, 55, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 79, 80, 81, 82, 83, 84, 87, 88, 89, 91, 92, 93, 94, 95, 97, 99, 100, 102, 103, 104, 105, 107, 108, 109, 110, 111, 112, 114, 115, 116, 117, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 136, 137, 138, 139, 140, 141, 142, 143, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 156, 157, 158, 159, 160, 163, 164, 165, 166, 167, 168, 169, 170, 171, 173, 174, 176, 178, 179, 180, 182, 183, 184, 185, 186, 188, 189, 190, 192, 193, 195, 196, 197, 198, 199, 200, 201, 202, 203, 206, 208, 209, 210, 215, 216, 218, 219, 220, 221, 223, 224, 225, 229], "fp32fp16": 66, "fp64": [17, 19, 21, 22, 27, 28, 35, 36, 37, 41, 42, 44, 45, 46, 54, 55, 59, 61, 67, 70, 71, 72, 73, 79, 80, 82, 83, 88, 89, 92, 93, 99, 104, 105, 108, 110, 111, 112, 116, 122, 124, 125, 127, 129, 130, 132, 133, 137, 138, 139, 140, 141, 142, 147, 150, 151, 152, 153, 154, 156, 159, 160, 163, 164, 166, 167, 168, 169, 170, 176, 178, 183, 184, 185, 189, 190, 192, 193, 195, 196, 197, 198, 199, 200, 202, 206, 210, 215, 216, 218, 219, 221, 223, 224, 225, 229], "fp_": [151, 176, 216], "fp_abs_": 10, "fp_abs_p": 10, "fp_absgrad_": 11, "fp_absgrad_p": 11, "fp_adam_": 14, "fp_adam_p": 14, "fp_adamweightdecay_": 15, "fp_adamweightdecay_p": 15, "fp_add_": 235, "fp_add_p": 235, "fp_adder_": 16, "fp_adder_p": 16, "fp_addext_": 17, "fp_addext_p": 17, "fp_addgrad_": 18, "fp_addgrad_p": 18, "fp_addn_": 19, "fp_addn_p": 19, "fp_addrelu6_": 17, "fp_addrelu6_p": 17, "fp_addrelu_": 17, "fp_addrelu_p": 17, "fp_affine_": 20, "fp_affine_p": 20, "fp_allgather_": 22, "fp_allgather_p": 22, "fp_and_": 112, "fp_and_p": 112, "fp_applymomentum_": 23, "fp_applymomentum_p": 23, "fp_argmax_": 24, "fp_argmax_p": 24, "fp_argmin_": 25, "fp_argmin_p": 25, "fp_assign_": 27, "fp_assign_p": 27, "fp_assignadd_": 28, "fp_assignadd_p": 28, "fp_attention_": 29, "fp_attention_p": 29, "fp_audio_spectrogram_": 30, "fp_audio_spectrogram_p": 30, "fp_avgpool_fusion_": 31, "fp_avgpool_fusion_p": 31, "fp_avgpoolinggrad_": 32, "fp_avgpoolinggrad_p": 32, "fp_batchnorm_": 33, "fp_batchnorm_p": 33, "fp_batchnormgrad_": 34, "fp_batchnormgrad_p": 34, "fp_batchtospace_": 35, "fp_batchtospace_p": 35, "fp_batchtospacend_": 36, "fp_batchtospacend_p": 36, "fp_biasadd_": 37, "fp_biasadd_p": 37, "fp_biasaddgrad_": 38, "fp_biasaddgrad_p": 38, "fp_binarycrossentropy_": 39, "fp_binarycrossentropy_p": 39, "fp_binarycrossentropygrad_": 40, "fp_binarycrossentropygrad_p": 40, "fp_broadcastto_": 41, "fp_broadcastto_p": 41, "fp_ceil_": 43, "fp_ceil_p": 43, "fp_celu_": 12, "fp_celu_p": 12, "fp_clip_": [12, 44], "fp_clip_p": [12, 44], "fp_concat_": 45, "fp_concat_p": 45, "fp_constant_of_shape_": 46, "fp_constant_of_shape_p": 46, "fp_conv2d_": 47, "fp_conv2d_p": 47, "fp_conv2dbackpropfilterfusion_": 49, "fp_conv2dbackpropfilterfusion_p": 49, "fp_conv2dbackpropinputfusion_": 50, "fp_conv2dbackpropinputfusion_p": 50, "fp_convtranspose_": 48, "fp_convtranspose_p": 48, "fp_cos_": 51, "fp_cos_p": 51, "fp_crop_and_resize_anycor": 53, "fp_cumsum_": 54, "fp_cumsum_p": 54, "fp_deconvgradfilter_": 58, "fp_deconvgradfilter_p": 58, "fp_depthtospace_": 59, "fp_depthtospace_p": 59, "fp_detection_post_process_": 60, "fp_detection_post_process_p": 60, "fp_div_fusion_": 61, "fp_div_fusion_p": 61, "fp_dropout_": 63, "fp_dropout_p": 63, "fp_dropoutgrad_": 64, "fp_dropoutgrad_p": 64, "fp_eltwise_": 67, "fp_eltwise_p": 67, "fp_elu_": [12, 68], "fp_elu_grad_": 13, "fp_elu_grad_p": 13, "fp_elu_p": [12, 68], "fp_embeddinglookup_": 69, "fp_embeddinglookup_p": 69, "fp_equal_": 70, "fp_equal_p": 70, "fp_erf_": 71, "fp_erf_p": 71, "fp_expfusion_": 73, "fp_expfusion_p": 73, "fp_extract_features_": 55, "fp_extract_features_p": 55, "fp_fake_quant_with_min_max_vars_": 74, "fp_fake_quant_with_min_max_vars_p": 74, "fp_fake_quant_with_min_max_vars_per_channel_": 75, "fp_fake_quant_with_min_max_vars_per_channel_p": 75, "fp_fill_": 78, "fp_fill_p": 78, "fp_flattengrad_": 81, "fp_flattengrad_p": 81, "fp_floor_": 82, "fp_floor_p": 82, "fp_floordiv_": 83, "fp_floordiv_p": 83, "fp_floormod_": 84, "fp_floormod_p": 84, "fp_formattranspose_": 85, "fp_formattranspose_p": 85, "fp_fullconnection_": 86, "fp_fullconnection_p": 86, "fp_fusedbatchnorm_": 87, "fp_fusedbatchnorm_p": 87, "fp_gather_": 88, "fp_gather_nd_": 89, "fp_gather_nd_p": 89, "fp_gather_p": 88, "fp_gatherd_": 90, "fp_gatherd_p": 90, "fp_gelu_": 12, "fp_gelu_grad_": 13, "fp_gelu_grad_p": 13, "fp_gelu_p": 12, "fp_glu_": 91, "fp_glu_p": 91, "fp_graddiv1l_": 62, "fp_graddiv1l_p": 62, "fp_graddiv2l_": 62, "fp_graddiv2l_p": 62, "fp_graddiv_": 62, "fp_graddiv_p": 62, "fp_gradmul1l_": 131, "fp_gradmul1l_p": 131, "fp_gradmul2l_": 131, "fp_gradmul2l_p": 131, "fp_gradmul_": 131, "fp_gradmul_p": 131, "fp_greater_": 92, "fp_greater_p": 92, "fp_greaterequal_": 93, "fp_greaterequal_p": 93, "fp_groupnormfusion_": 94, "fp_groupnormfusion_p": 94, "fp_gru_": 95, "fp_gru_p": 95, "fp_h_sigmoid_grad_": 13, "fp_h_sigmoid_grad_p": 13, "fp_h_swish_grad_": 13, "fp_h_swish_grad_p": 13, "fp_hard_shrink_grad_": 13, "fp_hard_shrink_grad_p": 13, "fp_hardshrink_": 12, "fp_hardshrink_p": 12, "fp_hardtanh_": 12, "fp_hardtanh_p": 12, "fp_hsigmoid_": 12, "fp_hsigmoid_p": 12, "fp_hswish_": 12, "fp_hswish_p": 12, "fp_instancenorm_": 97, "fp_instancenorm_p": 97, "fp_isfinite_": 99, "fp_isfinite_p": 99, "fp_l2norm_": 100, "fp_l2norm_p": 100, "fp_l_relu_grad_": 13, "fp_l_relu_grad_p": 13, "fp_layernormfusion_": 101, "fp_layernormfusion_p": 101, "fp_layernormgrad_": 102, "fp_layernormgrad_p": 102, "fp_leaky_relu_": 103, "fp_leaky_relu_p": 103, "fp_less_": 104, "fp_less_p": 104, "fp_lessequal_": 105, "fp_lessequal_p": 105, "fp_linspace_": 106, "fp_linspace_p": 106, "fp_log1p_": 108, "fp_log1p_p": 108, "fp_log_": 107, "fp_log_grad_": 109, "fp_log_grad_p": 109, "fp_log_p": 107, "fp_logical_not_": 110, "fp_logical_not_p": 110, "fp_logical_or_": 111, "fp_logical_or_p": 111, "fp_logsoftmax_": 113, "fp_logsoftmax_p": 113, "fp_lpnorm_": 114, "fp_lpnorm_p": 114, "fp_lrelu_": 12, "fp_lrelu_p": 12, "fp_lrn_p": 115, "fp_lrn_s": 115, "fp_lsh_projection_": 116, "fp_lsh_projection_p": 116, "fp_lstm_p": 117, "fp_lstm_s": 117, "fp_lstmgrad_": 118, "fp_lstmgrad_p": 118, "fp_lstmgraddata_": 119, "fp_lstmgraddata_p": 119, "fp_lstmgradweight_": 120, "fp_lstmgradweight_p": 120, "fp_matmulfusion_": 121, "fp_matmulfusion_p": 121, "fp_maximum_": 122, "fp_maximum_p": 122, "fp_maximumgrad_": 123, "fp_maximumgrad_p": 123, "fp_maxpool_fusion_": 124, "fp_maxpool_fusion_p": 124, "fp_maxpool_grad_": 125, "fp_maxpool_grad_p": 125, "fp_mfcc_p": 126, "fp_mfcc_s": 126, "fp_minimum_": 127, "fp_minimum_p": 127, "fp_minimumgrad_": 128, "fp_minimumgrad_p": 128, "fp_mod_": 129, "fp_mod_p": 129, "fp_mul_": 130, "fp_mul_p": 130, "fp_neg_": 132, "fp_neg_grad_": 133, "fp_neg_grad_p": 133, "fp_neg_p": 132, "fp_nllloss_": 134, "fp_nllloss_p": 134, "fp_nlllossgrad_": 135, "fp_nlllossgrad_p": 135, "fp_non_max_suppression_": 136, "fp_non_max_suppression_p": 136, "fp_nonzero_": 137, "fp_nonzero_p": 137, "fp_not_equal_": 138, "fp_not_equal_p": 138, "fp_onehot_": 139, "fp_onehot_p": 139, "fp_ones_like_": 140, "fp_ones_like_p": 140, "fp_padfusion_": 141, "fp_padfusion_p": 141, "fp_pow_fusion_": 142, "fp_pow_fusion_p": 142, "fp_power_grad_": 143, "fp_power_grad_p": 143, "fp_prelufusion_": 144, "fp_prelufusion_p": 144, "fp_priorbox_": 145, "fp_priorbox_p": 145, "fp_quantdata_": 66, "fp_quantdata_p": 66, "fp_raggedrange_": 147, "fp_raggedrange_p": 147, "fp_random_normal_": 148, "fp_random_normal_p": 148, "fp_random_standard_normal_": 149, "fp_random_standard_normal_p": 149, "fp_range_": 150, "fp_range_p": 150, "fp_real_div_": 152, "fp_real_div_p": 152, "fp_reciprocal_": 153, "fp_reciprocal_p": 153, "fp_reduce_": 154, "fp_reduce_p": 154, "fp_reduceall_": 21, "fp_reduceall_p": 21, "fp_reducescatter_": 155, "fp_reducescatter_p": 155, "fp_relu6_": 12, "fp_relu6_grad_": 13, "fp_relu6_grad_p": 13, "fp_relu6_p": 12, "fp_relu_": 12, "fp_relu_grad_": 13, "fp_relu_grad_p": 13, "fp_relu_p": 12, "fp_reshape_": 156, "fp_reshape_p": 156, "fp_resize_anycor": 157, "fp_resizebilineargrad_": 158, "fp_resizebilineargrad_p": 158, "fp_resizenearestneighborgrad_": 158, "fp_resizenearestneighborgrad_p": 158, "fp_roipooling_": 162, "fp_roipooling_p": 162, "fp_round_": 163, "fp_round_p": 163, "fp_rsqrt_p": 164, "fp_rsqrt_s": 164, "fp_rsqrtgrad_": 165, "fp_rsqrtgrad_p": 165, "fp_scalefusion_": 166, "fp_scalefusion_p": 166, "fp_scatter_elements_": 167, "fp_scatter_elements_p": 167, "fp_scatter_nd_": 168, "fp_scatter_nd_p": 168, "fp_scatter_nd_update_": 169, "fp_scatter_nd_update_p": 169, "fp_select_": 170, "fp_select_p": 170, "fp_sgd_p": 171, "fp_sgd_s": 171, "fp_sigmoid_": 12, "fp_sigmoid_grad_": 13, "fp_sigmoid_grad_p": 13, "fp_sigmoid_p": 12, "fp_sigmoidcrossentropywithlogits_": 174, "fp_sigmoidcrossentropywithlogits_p": 174, "fp_sigmoidcrossentropywithlogitsgrad_": 173, "fp_sigmoidcrossentropywithlogitsgrad_p": 173, "fp_sin_": 175, "fp_sin_p": 175, "fp_slice_": 178, "fp_slice_p": 178, "fp_smoothl1loss_": 179, "fp_smoothl1loss_p": 179, "fp_smoothl1lossgrad_": 180, "fp_smoothl1lossgrad_p": 180, "fp_softmax_": 181, "fp_softmax_cross_entropy_with_logits_": 182, "fp_softmax_cross_entropy_with_logits_p": 182, "fp_softmax_p": 181, "fp_softplus_": 12, "fp_softplus_grad_": 13, "fp_softplus_grad_p": 13, "fp_softplus_p": 12, "fp_softshrink_": 12, "fp_softshrink_grad_": 13, "fp_softshrink_grad_p": 13, "fp_softshrink_p": 12, "fp_softsignopt_": 12, "fp_softsignopt_p": 12, "fp_spacetobatch_": 183, "fp_spacetobatch_p": 183, "fp_spacetobatchnd_": 184, "fp_spacetobatchnd_p": 184, "fp_spacetodepth_": 185, "fp_spacetodepth_p": 185, "fp_sparse_softmax_cross_entropy_with_logits_": 186, "fp_sparse_softmax_cross_entropy_with_logits_p": 186, "fp_sparsefillemptyrows_": 187, "fp_sparsefillemptyrows_p": 187, "fp_sparsesegmentsum_": 189, "fp_sparsesegmentsum_p": 189, "fp_sparsetodense_": 190, "fp_sparsetodense_p": 190, "fp_splice_": 191, "fp_splice_p": 191, "fp_split_": 192, "fp_split_p": 192, "fp_split_with_overlap_": 193, "fp_split_with_overlap_p": 193, "fp_sqrt_p": 194, "fp_sqrt_s": 194, "fp_sqrtgrad_": 195, "fp_sqrtgrad_p": 195, "fp_square_": 196, "fp_square_p": 196, "fp_squaredifference_": 197, "fp_squaredifference_p": 197, "fp_stack_": 199, "fp_stack_p": 199, "fp_stridedslicegrad_": 201, "fp_stridedslicegrad_p": 201, "fp_subext_": 202, "fp_subext_p": 202, "fp_subgrad_": 203, "fp_subgrad_p": 203, "fp_subrelu6_": 202, "fp_subrelu6_p": 202, "fp_subrelu_": 202, "fp_subrelu_p": 202, "fp_swish_": 12, "fp_swish_p": 12, "fp_tanh_": 12, "fp_tanh_grad_": 13, "fp_tanh_grad_p": 13, "fp_tanh_p": 12, "fp_tensor_scatter_add_": 206, "fp_tensor_scatter_add_p": 206, "fp_tensorarrayread_": 208, "fp_tensorarrayread_p": 208, "fp_tensorlistfromtensor_": 210, "fp_tensorlistfromtensor_p": 210, "fp_tile_": 215, "fp_tile_p": 215, "fp_to_i8_quant_": 146, "fp_to_i8_quant_p": 146, "fp_topk_fusion_": 216, "fp_topk_fusion_p": 216, "fp_transpose_": 217, "fp_transpose_p": 217, "fp_tril_": 218, "fp_tril_p": 218, "fp_triu_": 219, "fp_triu_p": 219, "fp_uniform_real_": 220, "fp_uniform_real_p": 220, "fp_unique_": 221, "fp_unique_p": 221, "fp_unsorted_segment_sum_": 222, "fp_unsorted_segment_sum_p": 222, "fp_where_": 225, "fp_where_p": 225, "fp_xxx_xxx": 235, "fp_zerolike_": 226, "fp_zerolike_p": 226, "fprintf": 0, "fr": 4, "fr_axi": 4, "fr_gap": 4, "frac": [7, 8, 12, 14, 15, 18, 20, 29, 31, 32, 33, 34, 35, 40, 58, 61, 62, 66, 69, 71, 74, 75, 76, 77, 83, 84, 87, 94, 97, 100, 101, 102, 106, 109, 113, 114, 135, 136, 146, 147, 152, 153, 161, 162, 164, 165, 173, 174, 179, 180, 181, 182, 186, 189, 195, 200, 203], "frequenc": 126, "from": [0, 4, 64, 231, 233], "ft78ne": [0, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 231], "ft78nedeviceinfo": 0, "ftp": 233, "full": 4, "full_input": 20, "full_input_shap": 20, "full_run": 20, "fullconnect": [65, 227, 228, 229], "function": [71, 172], "fuse": 87, "fusedbatchnorm": [65, 227, 228, 229], "fusion": [31, 49, 50, 94], "g_i": 135, "g_t": [14, 15, 23, 117, 118, 119, 120, 171], "gamma": [34, 97, 101, 102, 114], "gamma_c": [97, 114], "gamma_data": 101, "gamma_i": 101, "gate": [12, 91, 117], "gather": [4, 22, 65, 90, 227, 228, 229], "gatherd": [65, 227, 228, 229], "gathernd": [65, 227, 228, 229], "gaussian": 12, "gbatch": 125, "gcc": 233, "ge": [12, 68, 93, 123], "gelu": [12, 13], "gelugrad": 13, "gemm": 58, "generat": 63, "geq": [11, 103, 219], "get": [0, 136], "get_core_id": [16, 47, 48, 52, 53, 79, 95, 154, 157, 159, 160, 167], "getcorenum": [16, 20, 47, 48, 52, 53, 72, 79, 95, 154, 157, 159, 160, 167, 198, 223], "getinput": 231, "getinputdata": 231, "getlogiccoreid": [16, 20, 47, 48, 52, 53, 79, 95, 154, 157, 159, 160, 167], "getoutput": 231, "getoutputdata": [0, 231], "gimpel": 12, "gin_h": 125, "gin_w": 125, "given": [113, 181], "glu": [65, 227, 228, 229], "gnueabihf": 233, "googl": 12, "grad": [81, 182], "grad_flat": 81, "grad_out": 81, "gradient": [13, 14, 15, 23, 34, 40, 64, 171, 182, 186], "gradmul1l": 131, "gradmul2l": 131, "gram": 177, "grams_word_count": 177, "graph_mod": 0, "graphcel": 0, "gray": 0, "gray_cnn": 1, "graycnn": 0, "greater": [65, 159, 227, 228, 229], "greater_dim": 159, "greaterequ": [65, 227, 228, 229], "group": [16, 47, 48, 49, 50, 58, 94, 116], "group_": [16, 47, 48, 49, 50], "groupnormfus": [65, 227, 228, 229], "gru": [65, 227, 228, 229], "gru_param": 95, "gruparamet": 95, "gsm": [121, 147], "h_": [35, 47, 48, 95, 117, 118, 119, 120, 124, 185], "h_0": 117, "h_i": [124, 125], "h_o": [124, 125], "h_scale": 60, "h_t": [95, 117], "hadamard": 95, "half": [10, 11, 12, 13, 15, 16, 17, 18, 21, 23, 24, 25, 27, 28, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 43, 44, 45, 47, 48, 49, 50, 51, 53, 54, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 78, 81, 82, 83, 84, 85, 86, 87, 88, 89, 91, 92, 93, 94, 95, 97, 99, 100, 101, 102, 103, 104, 105, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 122, 123, 127, 128, 129, 130, 131, 132, 133, 134, 135, 138, 139, 140, 142, 143, 144, 145, 146, 148, 149, 152, 153, 154, 155, 156, 157, 158, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 173, 174, 175, 178, 179, 180, 181, 183, 184, 185, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 199, 201, 202, 203, 206, 210, 215, 216, 217, 218, 219, 220, 221, 224, 225, 226], "handl": [208, 209], "handle_data": [208, 209], "handle_s": [208, 209], "hard": [12, 13, 20], "hardshrink": 12, "hardshrinkgrad": 13, "hardtanh": 12, "has": 233, "has_bia": 0, "has_bias_": 117, "hash": [55, 116], "hash_group_num": 116, "hash_se": 116, "hashtablelookup": [65, 227, 228, 229], "hat": [14, 15, 34, 87, 94], "head_dim": 29, "head_num": 29, "height": [0, 35, 36, 59, 158, 183, 184, 218, 219], "height_scal": 158, "hello": [55, 56], "hellodsp": [4, 230, 232], "hendryck": 12, "heterogen": [0, 231], "hf": 117, "hg": 117, "hi": 117, "hidden": 117, "hidden_buff": 117, "hidden_s": [95, 117], "hidden_size_": [95, 117], "hidden_st": [95, 117], "hidden_state_": 117, "hidden_state_batch": 95, "hit": 96, "hits_tensor": 96, "hn": 95, "ho": 117, "hop_length": [30, 126], "hot": [139, 186], "hp": [10, 51, 58, 74, 75, 78, 85, 86, 101, 110, 111, 113, 116, 118, 119, 120, 122, 127, 129, 130, 134, 135, 144, 155, 162, 175, 181, 187, 191, 194, 217, 226, 235], "hp_": 216, "hp_abs_": 10, "hp_abs_p": 10, "hp_absgrad_": 11, "hp_absgrad_p": 11, "hp_adamweightdecay_": 15, "hp_adamweightdecay_p": 15, "hp_add_": 235, "hp_add_p": 235, "hp_adder_": 16, "hp_adder_p": 16, "hp_addext_": 17, "hp_addext_p": 17, "hp_addgrad_": 18, "hp_addgrad_p": 18, "hp_addrelu6_": 17, "hp_addrelu6_p": 17, "hp_addrelu_": 17, "hp_addrelu_p": 17, "hp_affine_": 20, "hp_affine_p": 20, "hp_and_": 112, "hp_and_p": 112, "hp_applymomentum_": 23, "hp_applymomentum_p": 23, "hp_argmax_": 24, "hp_argmax_p": 24, "hp_argmin_": 25, "hp_argmin_p": 25, "hp_assign_": 27, "hp_assign_p": 27, "hp_assignadd_": 28, "hp_assignadd_p": 28, "hp_avgpool_fusion_": 31, "hp_avgpool_fusion_p": 31, "hp_avgpoolinggrad_": 32, "hp_avgpoolinggrad_p": 32, "hp_batchnorm_": 33, "hp_batchnorm_p": 33, "hp_batchnormgrad_": 34, "hp_batchnormgrad_p": 34, "hp_batchtospace_": 35, "hp_batchtospace_p": 35, "hp_batchtospacend_": 36, "hp_batchtospacend_p": 36, "hp_biasadd_": 37, "hp_biasadd_p": 37, "hp_biasaddgrad_": 38, "hp_biasaddgrad_p": 38, "hp_binarycrossentropy_": 39, "hp_binarycrossentropy_p": 39, "hp_binarycrossentropygrad_": 40, "hp_binarycrossentropygrad_p": 40, "hp_broadcastto_": 41, "hp_broadcastto_p": 41, "hp_ceil_": 43, "hp_ceil_p": 43, "hp_celu_": 12, "hp_celu_p": 12, "hp_clip_": [12, 44], "hp_clip_p": [12, 44], "hp_concat_": 45, "hp_concat_p": 45, "hp_conv2d_": 47, "hp_conv2d_p": 47, "hp_conv2dbackpropfilterfusion_": 49, "hp_conv2dbackpropfilterfusion_p": 49, "hp_conv2dbackpropinputfusion_": 50, "hp_conv2dbackpropinputfusion_p": 50, "hp_convtranspose_": 48, "hp_convtranspose_p": 48, "hp_cos_": 51, "hp_cos_p": 51, "hp_crop_and_resize_anycor": 53, "hp_cumsum_": 54, "hp_cumsum_p": 54, "hp_deconvgradfilter_": 58, "hp_deconvgradfilter_p": 58, "hp_depthtospace_": 59, "hp_depthtospace_p": 59, "hp_detection_post_process_": 60, "hp_detection_post_process_p": 60, "hp_div_fusion_": 61, "hp_div_fusion_p": 61, "hp_dropout_": 63, "hp_dropout_p": 63, "hp_dropoutgrad_": 64, "hp_dropoutgrad_p": 64, "hp_eltwise_": 67, "hp_eltwise_p": 67, "hp_elu_": [12, 68], "hp_elu_p": [12, 68], "hp_embeddinglookup_": 69, "hp_embeddinglookup_p": 69, "hp_equal_": 70, "hp_equal_p": 70, "hp_erf_": 71, "hp_erf_p": 71, "hp_expfusion_": 73, "hp_expfusion_p": 73, "hp_fake_quant_with_min_max_vars_": 74, "hp_fake_quant_with_min_max_vars_p": 74, "hp_fake_quant_with_min_max_vars_per_channel_": 75, "hp_fake_quant_with_min_max_vars_per_channel_p": 75, "hp_fill_": 78, "hp_fill_p": 78, "hp_flattengrad_": 81, "hp_flattengrad_p": 81, "hp_floor_": 82, "hp_floor_p": 82, "hp_floordiv_": 83, "hp_floordiv_p": 83, "hp_floormod_": 84, "hp_floormod_p": 84, "hp_formattranspose_": 85, "hp_formattranspose_p": 85, "hp_fullconnection_": 86, "hp_fullconnection_p": 86, "hp_fusedbatchnorm_": 87, "hp_fusedbatchnorm_p": 87, "hp_gather_": 88, "hp_gather_nd_": 89, "hp_gather_nd_p": 89, "hp_gather_p": 88, "hp_gelu_": 12, "hp_gelu_p": 12, "hp_glu_": 91, "hp_glu_p": 91, "hp_graddiv1l_": 62, "hp_graddiv1l_p": 62, "hp_graddiv2l_": 62, "hp_graddiv2l_p": 62, "hp_graddiv_": 62, "hp_graddiv_p": 62, "hp_gradmul1l_": 131, "hp_gradmul1l_p": 131, "hp_gradmul2l_": 131, "hp_gradmul2l_p": 131, "hp_gradmul_": 131, "hp_gradmul_p": 131, "hp_greater_": 92, "hp_greater_p": 92, "hp_greaterequal_": 93, "hp_greaterequal_p": 93, "hp_groupnormfusion_": 94, "hp_groupnormfusion_p": 94, "hp_gru_": 95, "hp_gru_p": 95, "hp_hardshrink_": 12, "hp_hardshrink_p": 12, "hp_hardtanh_": 12, "hp_hardtanh_p": 12, "hp_hsigmoid_": 12, "hp_hsigmoid_p": 12, "hp_hswish_": 12, "hp_hswish_p": 12, "hp_instancenorm_": 97, "hp_instancenorm_p": 97, "hp_isfinite_": 99, "hp_isfinite_p": 99, "hp_l2norm_": 100, "hp_l2norm_p": 100, "hp_layernormfusion_": 101, "hp_layernormfusion_p": 101, "hp_layernormgrad_": 102, "hp_layernormgrad_p": 102, "hp_leaky_relu_": 103, "hp_leaky_relu_p": 103, "hp_less_": 104, "hp_less_p": 104, "hp_lessequal_": 105, "hp_lessequal_p": 105, "hp_log1p_": 108, "hp_log1p_p": 108, "hp_log_": 107, "hp_log_grad_": 109, "hp_log_grad_p": 109, "hp_log_p": 107, "hp_logical_not_": 110, "hp_logical_not_p": 110, "hp_logical_or_": 111, "hp_logical_or_p": 111, "hp_logsoftmax_": 113, "hp_logsoftmax_p": 113, "hp_lpnorm_": 114, "hp_lpnorm_p": 114, "hp_lrelu_": 12, "hp_lrelu_p": 12, "hp_lrn_p": 115, "hp_lrn_s": 115, "hp_lsh_projection_": 116, "hp_lsh_projection_p": 116, "hp_lstm_p": 117, "hp_lstm_s": 117, "hp_lstmgrad_": 118, "hp_lstmgrad_p": 118, "hp_lstmgraddata_": 119, "hp_lstmgraddata_p": 119, "hp_lstmgradweight_": 120, "hp_lstmgradweight_p": 120, "hp_maximum_": 122, "hp_maximum_p": 122, "hp_maximumgrad_": 123, "hp_maximumgrad_p": 123, "hp_maxpool_fusion_": 124, "hp_maxpool_fusion_p": 124, "hp_maxpool_grad_": 125, "hp_maxpool_grad_p": 125, "hp_mfcc_p": 126, "hp_mfcc_s": 126, "hp_minimum_": 127, "hp_minimum_p": 127, "hp_minimumgrad_": 128, "hp_minimumgrad_p": 128, "hp_mod_": 129, "hp_mod_p": 129, "hp_mul_": 130, "hp_mul_p": 130, "hp_neg_": 132, "hp_neg_grad_": 133, "hp_neg_grad_p": 133, "hp_neg_p": 132, "hp_nllloss_": 134, "hp_nllloss_p": 134, "hp_nlllossgrad_": 135, "hp_nlllossgrad_p": 135, "hp_non_max_suppression_": 136, "hp_non_max_suppression_p": 136, "hp_not_equal_": 138, "hp_not_equal_p": 138, "hp_onehot_": 139, "hp_onehot_p": 139, "hp_ones_like_": 140, "hp_ones_like_p": 140, "hp_padfusion_": 141, "hp_padfusion_p": 141, "hp_pow_fusion_": 142, "hp_pow_fusion_p": 142, "hp_power_grad_": 143, "hp_power_grad_p": 143, "hp_prelufusion_": 144, "hp_prelufusion_p": 144, "hp_priorbox_": 145, "hp_priorbox_p": 145, "hp_quantdata_": 66, "hp_quantdata_p": 66, "hp_random_normal_": 148, "hp_random_normal_p": 148, "hp_random_standard_normal_": 149, "hp_random_standard_normal_p": 149, "hp_real_div_": 152, "hp_real_div_p": 152, "hp_reciprocal_": 153, "hp_reciprocal_p": 153, "hp_reduce_": 154, "hp_reduce_p": 154, "hp_reduceall_": 21, "hp_reduceall_p": 21, "hp_reducescatter_": 155, "hp_reducescatter_p": 155, "hp_relu6_": 12, "hp_relu6_p": 12, "hp_relu_": 12, "hp_relu_grad_": 13, "hp_relu_grad_p": 13, "hp_relu_p": 12, "hp_reshape_": 156, "hp_reshape_p": 156, "hp_resize_anycor": 157, "hp_resizebilineargrad_": 158, "hp_resizebilineargrad_p": 158, "hp_resizenearestneighborgrad_": 158, "hp_resizenearestneighborgrad_p": 158, "hp_roipooling_": 162, "hp_roipooling_p": 162, "hp_round_": 163, "hp_round_p": 163, "hp_rsqrt_p": 164, "hp_rsqrt_s": 164, "hp_rsqrtgrad_": 165, "hp_rsqrtgrad_p": 165, "hp_scalefusion_": 166, "hp_scalefusion_p": 166, "hp_scatter_elements_": 167, "hp_scatter_elements_p": 167, "hp_scatter_nd_": 168, "hp_scatter_nd_p": 168, "hp_scatter_nd_update_": 169, "hp_scatter_nd_update_p": 169, "hp_select_": 170, "hp_select_p": 170, "hp_sgd_p": 171, "hp_sgd_s": 171, "hp_sigmoid_": 12, "hp_sigmoid_p": 12, "hp_sigmoidcrossentropywithlogits_": 174, "hp_sigmoidcrossentropywithlogits_p": 174, "hp_sigmoidcrossentropywithlogitsgrad_": 173, "hp_sigmoidcrossentropywithlogitsgrad_p": 173, "hp_sin_": 175, "hp_sin_p": 175, "hp_slice_": 178, "hp_slice_p": 178, "hp_smoothl1loss_": 179, "hp_smoothl1loss_p": 179, "hp_smoothl1lossgrad_": 180, "hp_smoothl1lossgrad_p": 180, "hp_softmax_": 181, "hp_softmax_cross_entropy_with_logits_": 182, "hp_softmax_cross_entropy_with_logits_p": 182, "hp_softmax_p": 181, "hp_softplus_": 12, "hp_softplus_p": 12, "hp_softshrink_": 12, "hp_softshrink_p": 12, "hp_softsignopt_": 12, "hp_softsignopt_p": 12, "hp_spacetobatch_": 183, "hp_spacetobatch_p": 183, "hp_spacetobatchnd_": 184, "hp_spacetobatchnd_p": 184, "hp_spacetodepth_": 185, "hp_spacetodepth_p": 185, "hp_sparse_softmax_cross_entropy_with_logits_": 186, "hp_sparse_softmax_cross_entropy_with_logits_p": 186, "hp_sparsefillemptyrows_": 187, "hp_sparsefillemptyrows_p": 187, "hp_sparsesegmentsum_": 189, "hp_sparsesegmentsum_p": 189, "hp_sparsetodense_": 190, "hp_sparsetodense_p": 190, "hp_splice_": 191, "hp_splice_p": 191, "hp_split_": 192, "hp_split_p": 192, "hp_split_with_overlap_": 193, "hp_split_with_overlap_p": 193, "hp_sqrt_p": 194, "hp_sqrt_s": 194, "hp_sqrtgrad_": 195, "hp_sqrtgrad_p": 195, "hp_square_": 196, "hp_square_p": 196, "hp_squaredifference_": 197, "hp_squaredifference_p": 197, "hp_stack_": 199, "hp_stack_p": 199, "hp_stridedslicegrad_": 201, "hp_stridedslicegrad_p": 201, "hp_subext_": 202, "hp_subext_p": 202, "hp_subgrad_": 203, "hp_subgrad_p": 203, "hp_subrelu6_": 202, "hp_subrelu6_p": 202, "hp_subrelu_": 202, "hp_subrelu_p": 202, "hp_swish_": 12, "hp_swish_p": 12, "hp_tanh_": 12, "hp_tanh_p": 12, "hp_tensor_scatter_add_": 206, "hp_tensor_scatter_add_p": 206, "hp_tensorarrayread_": 208, "hp_tensorarrayread_p": 208, "hp_tensorlistfromtensor_": 210, "hp_tensorlistfromtensor_p": 210, "hp_tile_": 215, "hp_tile_p": 215, "hp_to_i8_quant_": 146, "hp_to_i8_quant_p": 146, "hp_topk_fusion_": 216, "hp_topk_fusion_p": 216, "hp_transpose_": 217, "hp_transpose_p": 217, "hp_tril_": 218, "hp_tril_p": 218, "hp_triu_": 219, "hp_triu_p": 219, "hp_uniform_real_": 220, "hp_uniform_real_p": 220, "hp_unique_": 221, "hp_unique_p": 221, "hp_where_": 225, "hp_where_p": 225, "hp_zerolike_": 226, "hp_zerolike_p": 226, "hpp": 0, "hr": 95, "hsigmoid": 12, "hsigmoidgrad": 13, "hswish": 12, "hswishgrad": 13, "https": [0, 233], "hwc2chw": 0, "hyperbol": 12, "hz": 95, "i16": [85, 110, 111, 116, 122, 127, 129, 130, 155, 191, 217, 235], "i16_": 216, "i16_abs_": 10, "i16_abs_p": 10, "i16_add_": 235, "i16_add_p": 235, "i16_addext_": 17, "i16_addext_p": 17, "i16_addn_": 19, "i16_addn_p": 19, "i16_addrelu6_": 17, "i16_addrelu6_p": 17, "i16_addrelu_": 17, "i16_addrelu_p": 17, "i16_allgather_": 22, "i16_allgather_p": 22, "i16_and_": 112, "i16_and_p": 112, "i16_assign_": 27, "i16_assign_p": 27, "i16_assignadd_": 28, "i16_assignadd_p": 28, "i16_batchtospace_": 35, "i16_batchtospace_p": 35, "i16_batchtospacend_": 36, "i16_batchtospacend_p": 36, "i16_biasadd_": 37, "i16_biasadd_p": 37, "i16_broadcastto_": 41, "i16_broadcastto_p": 41, "i16_clip_": 44, "i16_clip_p": 44, "i16_concat_": 45, "i16_concat_p": 45, "i16_constant_of_shape_": 46, "i16_constant_of_shape_p": 46, "i16_cos_": 51, "i16_cos_p": 51, "i16_cumsum_": 54, "i16_cumsum_p": 54, "i16_depthtospace_": 59, "i16_depthtospace_p": 59, "i16_div_fusion_": 61, "i16_div_fusion_p": 61, "i16_eltwise_": 67, "i16_eltwise_p": 67, "i16_equal_": 70, "i16_equal_p": 70, "i16_expfusion_": 73, "i16_expfusion_p": 73, "i16_extract_features_": 55, "i16_extract_features_p": 55, "i16_fill_": 78, "i16_fill_p": 78, "i16_formattranspose_": 85, "i16_formattranspose_p": 85, "i16_gather_": 88, "i16_gather_nd_": 89, "i16_gather_nd_p": 89, "i16_gather_p": 88, "i16_gatherd_": 90, "i16_gatherd_p": 90, "i16_greater_": 92, "i16_greater_p": 92, "i16_greaterequal_": 93, "i16_greaterequal_p": 93, "i16_invertpermutation_": 98, "i16_invertpermutation_p": 98, "i16_isfinite_": 99, "i16_isfinite_p": 99, "i16_less_": 104, "i16_less_p": 104, "i16_lessequal_": 105, "i16_lessequal_p": 105, "i16_log1p_": 108, "i16_log1p_p": 108, "i16_log_": 107, "i16_log_p": 107, "i16_logical_not_": 110, "i16_logical_not_p": 110, "i16_logical_or_": 111, "i16_logical_or_p": 111, "i16_lsh_projection_": 116, "i16_lsh_projection_p": 116, "i16_matmulfusion_": 121, "i16_matmulfusion_p": 121, "i16_maximum_": 122, "i16_maximum_p": 122, "i16_minimum_": 127, "i16_minimum_p": 127, "i16_mod_": 129, "i16_mod_p": 129, "i16_mul_": 130, "i16_mul_p": 130, "i16_neg_": 132, "i16_neg_grad_": 133, "i16_neg_grad_p": 133, "i16_neg_p": 132, "i16_nonzero_": 137, "i16_nonzero_p": 137, "i16_not_equal_": 138, "i16_not_equal_p": 138, "i16_onehot_": 139, "i16_onehot_p": 139, "i16_ones_like_": 140, "i16_ones_like_p": 140, "i16_padfusion_": 141, "i16_padfusion_p": 141, "i16_pow_fusion_": 142, "i16_pow_fusion_p": 142, "i16_raggedrange_": 147, "i16_raggedrange_p": 147, "i16_range_": 150, "i16_range_p": 150, "i16_real_div_": 152, "i16_real_div_p": 152, "i16_reciprocal_": 153, "i16_reciprocal_p": 153, "i16_reduce_": 154, "i16_reduce_p": 154, "i16_reduceall_": 21, "i16_reduceall_p": 21, "i16_reducescatter_": 155, "i16_reducescatter_p": 155, "i16_reshape_": 156, "i16_reshape_p": 156, "i16_rsqrt_": 164, "i16_rsqrt_p": 164, "i16_scalefusion_": 166, "i16_scalefusion_p": 166, "i16_scatter_elements_": 167, "i16_scatter_elements_p": 167, "i16_scatter_nd_": 168, "i16_scatter_nd_p": 168, "i16_scatter_nd_update_": 169, "i16_scatter_nd_update_p": 169, "i16_select_": 170, "i16_select_p": 170, "i16_sin_": 175, "i16_sin_p": 175, "i16_slice_": 178, "i16_slice_p": 178, "i16_spacetobatch_": 183, "i16_spacetobatch_p": 183, "i16_spacetobatchnd_": 184, "i16_spacetobatchnd_p": 184, "i16_spacetodepth_": 185, "i16_spacetodepth_p": 185, "i16_sparsefillemptyrows_": 187, "i16_sparsefillemptyrows_p": 187, "i16_sparsesegmentsum_": 189, "i16_sparsesegmentsum_p": 189, "i16_sparsetodense_": 190, "i16_sparsetodense_p": 190, "i16_splice_": 191, "i16_splice_p": 191, "i16_split_": 192, "i16_split_p": 192, "i16_split_with_overlap_": 193, "i16_split_with_overlap_p": 193, "i16_sqrt_": 194, "i16_sqrt_p": 194, "i16_sqrtgrad_": 195, "i16_sqrtgrad_p": 195, "i16_square_": 196, "i16_square_p": 196, "i16_squaredifference_": 197, "i16_squaredifference_p": 197, "i16_stack_": 199, "i16_stack_p": 199, "i16_subrelu6_": 202, "i16_subrelu6_p": 202, "i16_subrelu_": 202, "i16_subrelu_p": 202, "i16_tensor_scatter_add_": 206, "i16_tensor_scatter_add_p": 206, "i16_tensorarrayread_": 208, "i16_tensorarrayread_p": 208, "i16_tensorlistfromtensor_": 210, "i16_tensorlistfromtensor_p": 210, "i16_tile_": 215, "i16_tile_p": 215, "i16_topk_fusion_": 216, "i16_topk_fusion_p": 216, "i16_transpose_": 217, "i16_transpose_p": 217, "i16_tril_": 218, "i16_tril_p": 218, "i16_triu_": 219, "i16_triu_p": 219, "i16_unique_": 221, "i16_unique_p": 221, "i16_unsorted_segment_sum_": 222, "i16_unsorted_segment_sum_p": 222, "i16_where_": 225, "i16_where_p": 225, "i16_zerolike_": 226, "i16_zerolike_p": 226, "i32": [85, 110, 111, 116, 122, 127, 129, 130, 155, 191, 217, 235], "i32_": 216, "i32_abs_": 10, "i32_abs_p": 10, "i32_add_": 235, "i32_add_p": 235, "i32_addext_": 17, "i32_addext_p": 17, "i32_addn_": 19, "i32_addn_p": 19, "i32_addrelu6_": 17, "i32_addrelu6_p": 17, "i32_addrelu_": 17, "i32_addrelu_p": 17, "i32_allgather_": 22, "i32_allgather_p": 22, "i32_and_": 112, "i32_and_p": 112, "i32_assign_": 27, "i32_assign_p": 27, "i32_assignadd_": 28, "i32_assignadd_p": 28, "i32_batchtospace_": 35, "i32_batchtospace_p": 35, "i32_batchtospacend_": 36, "i32_batchtospacend_p": 36, "i32_biasadd_": 37, "i32_biasadd_p": 37, "i32_broadcastto_": 41, "i32_broadcastto_p": 41, "i32_clip_": 44, "i32_clip_p": 44, "i32_concat_": 45, "i32_concat_p": 45, "i32_constant_of_shape_": 46, "i32_constant_of_shape_p": 46, "i32_cos_": 51, "i32_cos_p": 51, "i32_cumsum_": 54, "i32_cumsum_p": 54, "i32_depthtospace_": 59, "i32_depthtospace_p": 59, "i32_div_fusion_": 61, "i32_div_fusion_p": 61, "i32_eltwise_": 67, "i32_eltwise_p": 67, "i32_equal_": 70, "i32_equal_p": 70, "i32_expfusion_": 73, "i32_expfusion_p": 73, "i32_extract_features_": 55, "i32_extract_features_p": 55, "i32_fill_": 78, "i32_fill_p": 78, "i32_formattranspose_": 85, "i32_formattranspose_p": 85, "i32_gather_": 88, "i32_gather_nd_": 89, "i32_gather_nd_p": 89, "i32_gather_p": 88, "i32_gatherd_": 90, "i32_gatherd_p": 90, "i32_greater_": 92, "i32_greater_p": 92, "i32_greaterequal_": 93, "i32_greaterequal_p": 93, "i32_hashtablelookup_": 96, "i32_hashtablelookup_p": 96, "i32_invertpermutation_": 98, "i32_invertpermutation_p": 98, "i32_isfinite_": 99, "i32_isfinite_p": 99, "i32_less_": 104, "i32_less_p": 104, "i32_lessequal_": 105, "i32_lessequal_p": 105, "i32_log1p_": 108, "i32_log1p_p": 108, "i32_log_": 107, "i32_log_p": 107, "i32_logical_not_": 110, "i32_logical_not_p": 110, "i32_logical_or_": 111, "i32_logical_or_p": 111, "i32_lsh_projection_": 116, "i32_lsh_projection_p": 116, "i32_matmulfusion_": 121, "i32_matmulfusion_p": 121, "i32_maximum_": 122, "i32_maximum_p": 122, "i32_minimum_": 127, "i32_minimum_p": 127, "i32_mod_": 129, "i32_mod_p": 129, "i32_mul_": 130, "i32_mul_p": 130, "i32_neg_": 132, "i32_neg_grad_": 133, "i32_neg_grad_p": 133, "i32_neg_p": 132, "i32_nonzero_": 137, "i32_nonzero_p": 137, "i32_not_equal_": 138, "i32_not_equal_p": 138, "i32_onehot_": 139, "i32_onehot_p": 139, "i32_ones_like_": 140, "i32_ones_like_p": 140, "i32_padfusion_": 141, "i32_padfusion_p": 141, "i32_pow_fusion_": 142, "i32_pow_fusion_p": 142, "i32_raggedrange_": 147, "i32_raggedrange_p": 147, "i32_range_": 150, "i32_range_p": 150, "i32_real_div_": 152, "i32_real_div_p": 152, "i32_reciprocal_": 153, "i32_reciprocal_p": 153, "i32_reduce_": 154, "i32_reduce_p": 154, "i32_reduceall_": 21, "i32_reduceall_p": 21, "i32_reducescatter_": 155, "i32_reducescatter_p": 155, "i32_reshape_": 156, "i32_reshape_p": 156, "i32_rsqrt_": 164, "i32_rsqrt_p": 164, "i32_scalefusion_": 166, "i32_scalefusion_p": 166, "i32_scatter_elements_": 167, "i32_scatter_elements_p": 167, "i32_scatter_nd_": 168, "i32_scatter_nd_p": 168, "i32_scatter_nd_update_": 169, "i32_scatter_nd_update_p": 169, "i32_select_": 170, "i32_select_p": 170, "i32_sin_": 175, "i32_sin_p": 175, "i32_slice_": 178, "i32_slice_p": 178, "i32_spacetobatch_": 183, "i32_spacetobatch_p": 183, "i32_spacetobatchnd_": 184, "i32_spacetobatchnd_p": 184, "i32_spacetodepth_": 185, "i32_spacetodepth_p": 185, "i32_sparsefillemptyrows_": 187, "i32_sparsefillemptyrows_p": 187, "i32_sparsesegmentsum_": 189, "i32_sparsesegmentsum_p": 189, "i32_sparsetodense_": 190, "i32_sparsetodense_p": 190, "i32_splice_": 191, "i32_splice_p": 191, "i32_split_": 192, "i32_split_p": 192, "i32_split_with_overlap_": 193, "i32_split_with_overlap_p": 193, "i32_sqrt_": 194, "i32_sqrt_p": 194, "i32_sqrtgrad_": 195, "i32_sqrtgrad_p": 195, "i32_square_": 196, "i32_square_p": 196, "i32_squaredifference_": 197, "i32_squaredifference_p": 197, "i32_stack_": 199, "i32_stack_p": 199, "i32_subrelu6_": 202, "i32_subrelu6_p": 202, "i32_subrelu_": 202, "i32_subrelu_p": 202, "i32_tensor_scatter_add_": 206, "i32_tensor_scatter_add_p": 206, "i32_tensorarrayread_": 208, "i32_tensorarrayread_p": 208, "i32_tensorlistfromtensor_": 210, "i32_tensorlistfromtensor_p": 210, "i32_tile_": 215, "i32_tile_p": 215, "i32_topk_fusion_": 216, "i32_topk_fusion_p": 216, "i32_transpose_": 217, "i32_transpose_p": 217, "i32_tril_": 218, "i32_tril_p": 218, "i32_triu_": 219, "i32_triu_p": 219, "i32_unique_": 221, "i32_unique_p": 221, "i32_unsorted_segment_sum_": 222, "i32_unsorted_segment_sum_p": 222, "i32_where_": 225, "i32_where_p": 225, "i32_zerolike_": 226, "i32_zerolike_p": 226, "i686": 233, "i8": [110, 111, 116, 122, 127, 129, 130, 182, 235], "i8_": [151, 176], "i8_abs_": 10, "i8_abs_p": 10, "i8_add_": 235, "i8_add_p": 235, "i8_adder_": 16, "i8_adder_p": 16, "i8_addext_": 17, "i8_addext_p": 17, "i8_addn_": 19, "i8_addn_p": 19, "i8_addrelu6_": 17, "i8_addrelu6_p": 17, "i8_addrelu_": 17, "i8_addrelu_p": 17, "i8_affine_": 20, "i8_affine_p": 20, "i8_allgather_": 22, "i8_allgather_p": 22, "i8_and_": 112, "i8_and_p": 112, "i8_assign_": 27, "i8_assign_p": 27, "i8_assignadd_": 28, "i8_assignadd_p": 28, "i8_avgpool_fusion_": 31, "i8_avgpool_fusion_p": 31, "i8_batchnorm_": 33, "i8_batchnorm_p": 33, "i8_batchtospace_": 35, "i8_batchtospace_p": 35, "i8_batchtospacend_": 36, "i8_batchtospacend_p": 36, "i8_biasadd_": 37, "i8_biasadd_p": 37, "i8_binarycrossentropy_": 39, "i8_binarycrossentropy_p": 39, "i8_broadcastto_": 41, "i8_broadcastto_p": 41, "i8_celu_": 12, "i8_celu_p": 12, "i8_clip_": [12, 44], "i8_clip_p": [12, 44], "i8_concat_": 45, "i8_concat_p": 45, "i8_constant_of_shape_": 46, "i8_constant_of_shape_p": 46, "i8_conv2d_": 47, "i8_conv2d_p": 47, "i8_convtranspose_": 48, "i8_convtranspose_p": 48, "i8_cos_": 51, "i8_cos_p": 51, "i8_crop_and_resize_anycor": 53, "i8_cumsum_": 54, "i8_cumsum_p": 54, "i8_depthtospace_": 59, "i8_depthtospace_p": 59, "i8_detection_post_process_": 60, "i8_detection_post_process_p": 60, "i8_div_fusion_": 61, "i8_div_fusion_p": 61, "i8_eltwise_": 67, "i8_eltwise_p": 67, "i8_elu_": [12, 68], "i8_elu_p": [12, 68], "i8_equal_": 70, "i8_equal_p": 70, "i8_expfusion_": 73, "i8_expfusion_p": 73, "i8_extract_features_": 55, "i8_extract_features_p": 55, "i8_fill_": 78, "i8_fill_p": 78, "i8_formattranspose_": 85, "i8_formattranspose_p": 85, "i8_fullconnection_": 86, "i8_fullconnection_p": 86, "i8_gather_": 88, "i8_gather_nd_": 89, "i8_gather_nd_p": 89, "i8_gather_p": 88, "i8_gatherd_": 90, "i8_gatherd_p": 90, "i8_gelu_": 12, "i8_gelu_p": 12, "i8_glu_": 91, "i8_glu_p": 91, "i8_greater_": 92, "i8_greater_p": 92, "i8_greaterequal_": 93, "i8_greaterequal_p": 93, "i8_gru_": 95, "i8_gru_p": 95, "i8_hardshrink_": 12, "i8_hardshrink_p": 12, "i8_hardtanh_": 12, "i8_hardtanh_p": 12, "i8_hsigmoid_": 12, "i8_hsigmoid_p": 12, "i8_hswish_": 12, "i8_hswish_p": 12, "i8_invertpermutation_": 98, "i8_invertpermutation_p": 98, "i8_isfinite_": 99, "i8_isfinite_p": 99, "i8_layernormfusion_": 101, "i8_layernormfusion_p": 101, "i8_leaky_relu_": 103, "i8_leaky_relu_p": 103, "i8_less_": 104, "i8_less_p": 104, "i8_lessequal_": 105, "i8_lessequal_p": 105, "i8_log1p_": 108, "i8_log1p_p": 108, "i8_logical_not_": 110, "i8_logical_not_p": 110, "i8_logical_or_": 111, "i8_logical_or_p": 111, "i8_logsoftmax_": 113, "i8_logsoftmax_p": 113, "i8_lrelu_": 12, "i8_lrelu_p": 12, "i8_lsh_projection_": 116, "i8_lsh_projection_p": 116, "i8_maximum_": 122, "i8_maximum_p": 122, "i8_minimum_": 127, "i8_minimum_p": 127, "i8_mod_": 129, "i8_mod_p": 129, "i8_mul_": 130, "i8_mul_p": 130, "i8_neg_": 132, "i8_neg_grad_": 133, "i8_neg_grad_p": 133, "i8_neg_p": 132, "i8_nllloss_": 134, "i8_nllloss_p": 134, "i8_non_max_suppression_": 136, "i8_non_max_suppression_p": 136, "i8_nonzero_": 137, "i8_nonzero_p": 137, "i8_not_equal_": 138, "i8_not_equal_p": 138, "i8_onehot_": 139, "i8_onehot_p": 139, "i8_ones_like_": 140, "i8_ones_like_p": 140, "i8_padfusion_": 141, "i8_padfusion_p": 141, "i8_pow_fusion_": 142, "i8_pow_fusion_p": 142, "i8_prelufusion_": 144, "i8_prelufusion_p": 144, "i8_raggedrange_": 147, "i8_raggedrange_p": 147, "i8_range_": 150, "i8_range_p": 150, "i8_real_div_": 152, "i8_real_div_p": 152, "i8_reciprocal_": 153, "i8_reciprocal_p": 153, "i8_reduce_": 154, "i8_reduce_p": 154, "i8_reduceall_": 21, "i8_reduceall_p": 21, "i8_reducescatter_": 155, "i8_reducescatter_p": 155, "i8_relu6_": 12, "i8_relu6_p": 12, "i8_relu_": 12, "i8_relu_p": 12, "i8_reshape_": 156, "i8_reshape_p": 156, "i8_resize_anycor": 157, "i8_roipooling_": 162, "i8_roipooling_p": 162, "i8_rsqrt_": 164, "i8_rsqrt_p": 164, "i8_scalefusion_": 166, "i8_scalefusion_p": 166, "i8_scatter_elements_": 167, "i8_scatter_elements_p": 167, "i8_scatter_nd_": 168, "i8_scatter_nd_p": 168, "i8_scatter_nd_update_": 169, "i8_scatter_nd_update_p": 169, "i8_select_": 170, "i8_select_p": 170, "i8_sigmoid_": 12, "i8_sigmoid_p": 12, "i8_sigmoidcrossentropywithlogits_": 174, "i8_sigmoidcrossentropywithlogits_p": 174, "i8_sin_": 175, "i8_sin_p": 175, "i8_slice_": 178, "i8_slice_p": 178, "i8_smoothl1loss_": 179, "i8_smoothl1loss_p": 179, "i8_softmax_": 181, "i8_softmax_cross_entropy_with_logits_": 182, "i8_softmax_cross_entropy_with_logits_p": 182, "i8_softmax_p": 181, "i8_softplus_": 12, "i8_softplus_p": 12, "i8_softshrink_": 12, "i8_softshrink_p": 12, "i8_softsignopt_": 12, "i8_softsignopt_p": 12, "i8_spacetobatch_": 183, "i8_spacetobatch_p": 183, "i8_spacetobatchnd_": 184, "i8_spacetobatchnd_p": 184, "i8_spacetodepth_": 185, "i8_spacetodepth_p": 185, "i8_sparsefillemptyrows_": 187, "i8_sparsefillemptyrows_p": 187, "i8_sparsetodense_": 190, "i8_sparsetodense_p": 190, "i8_splice_": 191, "i8_splice_p": 191, "i8_split_": 192, "i8_split_p": 192, "i8_split_with_overlap_": 193, "i8_split_with_overlap_p": 193, "i8_sqrt_": 194, "i8_sqrt_p": 194, "i8_sqrtgrad_": 195, "i8_sqrtgrad_p": 195, "i8_square_": 196, "i8_square_p": 196, "i8_squaredifference_": 197, "i8_squaredifference_p": 197, "i8_stack_": 199, "i8_stack_p": 199, "i8_subrelu6_": 202, "i8_subrelu6_p": 202, "i8_subrelu_": 202, "i8_subrelu_p": 202, "i8_swish_": 12, "i8_swish_p": 12, "i8_tanh_": 12, "i8_tanh_p": 12, "i8_tensor_scatter_add_": 206, "i8_tensor_scatter_add_p": 206, "i8_tensorarrayread_": 208, "i8_tensorarrayread_p": 208, "i8_tensorlistfromtensor_": 210, "i8_tensorlistfromtensor_p": 210, "i8_tile_": 215, "i8_tile_p": 215, "i8_to_fp_dequant_": 146, "i8_to_fp_dequant_p": 146, "i8_to_hp_dequant_": 146, "i8_to_hp_dequant_p": 146, "i8_transpose_": 217, "i8_transpose_p": 217, "i8_tril_": 218, "i8_tril_p": 218, "i8_triu_": 219, "i8_triu_p": 219, "i8_unique_": 221, "i8_unique_p": 221, "i8_unsorted_segment_sum_": 222, "i8_unsorted_segment_sum_p": 222, "i8_where_": 225, "i8_where_p": 225, "i8_zerolike_": 226, "i8_zerolike_p": 226, "i_": [88, 90, 139, 178], "i_0": [88, 90, 178], "i_1": [139, 178], "i_2": 139, "i_k": 178, "i_n": [90, 139], "i_t": 117, "id": [101, 189, 222], "id_max": 222, "ide": [0, 4, 231], "ident": 121, "ids": 69, "ids_siz": 69, "ids_size_": 69, "idx": 201, "idx1": 170, "idx2": 170, "idx3": 170, "ieee": 194, "if": [0, 11, 12, 16, 20, 21, 23, 26, 47, 48, 52, 53, 54, 55, 67, 68, 69, 70, 79, 95, 96, 99, 100, 103, 104, 105, 112, 117, 123, 125, 128, 138, 139, 142, 144, 154, 157, 158, 159, 160, 167, 170, 171, 179, 180, 187, 190, 204, 218, 219, 225], "ifft": [4, 9, 76, 77, 161, 227, 229], "ifft1": 4, "ifft_nobitrev": 229, "ifft_tim": 229, "ig": 117, "ii": 117, "im": 10, "im2row": 58, "imag": [0, 4, 130], "image_four_channel": 0, "image_height": 53, "image_origin": 0, "imagefolderdataset": 0, "imagepath": 0, "imagesc": 4, "img_siz": 0, "import": [0, 4, 6, 7, 8, 231, 233], "importdata": 4, "imread": 0, "imread_color": 0, "imshow": 4, "in": [16, 21, 34, 47, 48, 55, 56, 57, 69, 73, 95, 96, 110, 124, 125, 158, 162, 188, 189, 222], "in0": [17, 111, 122, 127, 129, 130, 199], "in0_i": 17, "in1": [17, 111, 122, 127, 129, 130, 199], "in1_i": 17, "in2": 199, "in_addr": 158, "in_c": 58, "in_channel": [16, 47, 48, 49, 50], "in_data": [91, 189, 213, 217], "in_data_shap": 189, "in_data_shape_s": 189, "in_elements_num0": [92, 93, 104, 105], "in_h": [31, 49, 50, 58, 124, 125], "in_height": 158, "in_i": 158, "in_indic": 189, "in_indices_ptr": 188, "in_indices_s": 189, "in_inshap": 188, "in_inshape_ptr": 188, "in_item": 213, "in_outshap": 188, "in_outshape_ptr": 188, "in_ptr": 162, "in_scal": 73, "in_segment_id": 189, "in_shap": [24, 25, 52, 59, 185, 200, 201], "in_shape_s": 200, "in_strid": [24, 25, 141, 188], "in_tensor0": 213, "in_tensor1": 213, "in_tensor2": 213, "in_w": [31, 49, 50, 58, 124, 125], "in_width": 158, "in_x": 158, "includ": [0, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 49, 50, 51, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 78, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 96, 97, 98, 99, 100, 101, 102, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 155, 156, 158, 161, 162, 163, 164, 165, 166, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 224, 225, 226], "include_all_ngram": 177, "inclus": 54, "increment_input": 20, "increment_input_shap": 20, "increment_output": 20, "increment_output_shap": 20, "ind_ndim": [168, 169], "ind_shap": [168, 169], "index": [0, 24, 25, 60, 90, 136, 190, 205, 207, 208, 209, 211, 213, 222, 224], "index_depth": 206, "index_i": 222, "index_list1": 170, "index_list2": 170, "index_list3": 170, "index_shap": 90, "indic": [24, 25, 62, 88, 89, 131, 139, 167, 168, 169, 187, 206], "indices_": 139, "indices_data": 167, "indices_depth": 168, "indices_i": 168, "indices_ndim": [88, 89, 168, 169], "indices_ptr": 187, "indices_shap": [88, 89, 139, 167, 168, 169], "indices_shape_s": 139, "indices_stride_": 167, "indices_total_num_": 167, "indices_vec": 190, "inf": [61, 66, 107, 142], "infer": 172, "infti": 66, "initi": 160, "inner": 200, "inner_count": 159, "inner_count_": 159, "inner_dim": 54, "inner_loop_num_": 216, "inner_s": [21, 97, 113, 114, 154, 181, 186, 200], "inner_sizes_": 154, "inner_stride_": 159, "inp_box": 53, "inp_box_idx": 53, "input": [0, 4, 6, 10, 11, 12, 17, 18, 21, 22, 24, 25, 26, 28, 30, 31, 32, 33, 34, 35, 36, 37, 41, 42, 43, 45, 50, 51, 52, 53, 54, 57, 59, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 76, 77, 79, 80, 81, 82, 83, 85, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 103, 104, 105, 107, 108, 110, 113, 114, 115, 117, 124, 125, 126, 132, 133, 137, 138, 141, 142, 146, 153, 154, 155, 156, 157, 161, 162, 163, 164, 166, 167, 172, 175, 178, 181, 183, 184, 185, 186, 188, 189, 191, 192, 193, 194, 196, 199, 200, 201, 204, 205, 206, 207, 210, 211, 215, 216, 217, 218, 219, 221, 222, 224, 231], "input0": [10, 11, 12, 19, 20, 45, 51, 61, 67, 69, 70, 73, 82, 83, 84, 101, 104, 105, 109, 111, 112, 122, 123, 127, 128, 129, 130, 131, 138, 152, 166, 170, 172, 173, 174, 197, 202, 205, 225], "input0_0": [104, 105], "input0_dim": [123, 128], "input0_i": [11, 61, 67, 70, 83, 104, 105, 111, 122, 127, 129, 130, 138, 152, 225], "input0_len": 20, "input0_shap": [20, 172], "input1": [11, 19, 20, 45, 61, 67, 69, 70, 83, 84, 92, 93, 104, 105, 109, 111, 112, 122, 123, 127, 128, 129, 130, 131, 138, 143, 152, 165, 170, 172, 173, 174, 195, 197, 202, 205, 225], "input1_0": [104, 105], "input1_dim": [123, 128], "input1_i": [11, 61, 67, 70, 83, 104, 105, 111, 122, 127, 129, 130, 138, 143, 152, 225], "input1_len": 20, "input1_shap": [20, 172], "input2": [20, 45, 92, 93, 143, 165, 195, 205], "input2_i": 143, "input2_len": 20, "input2_shap": 20, "input_": [32, 45, 75, 193], "input_0": 45, "input_1": 45, "input_axis_size_": 167, "input_batch_": [16, 47, 48, 49, 50], "input_bia": [37, 38, 95, 117], "input_bias_": 117, "input_box": 60, "input_c": 162, "input_channel_": [16, 47, 48, 49, 50], "input_col_align": 95, "input_col_align_": [95, 117], "input_column": 0, "input_data": [16, 30, 47, 48, 69, 126, 155, 159, 160, 167], "input_data_pad": [30, 126], "input_dims_": 167, "input_h": [32, 162], "input_h_": [16, 47, 48, 49, 50], "input_i": [12, 26, 28, 39, 40, 74, 98, 99, 100, 108, 110, 126, 132, 133, 142, 153, 156, 204], "input_index": 189, "input_len": [30, 126, 221], "input_n": 162, "input_ndim": [45, 88, 89, 192, 193, 199], "input_ptr": [113, 124, 125, 181], "input_rank": [22, 137, 188, 206], "input_row_align_": [95, 117], "input_s": [39, 40, 43, 57, 95, 96, 117], "input_scor": 60, "input_shap": [16, 35, 36, 41, 45, 47, 48, 52, 53, 88, 89, 90, 91, 126, 141, 154, 157, 167, 178, 183, 184, 192, 193, 199, 206, 215], "input_shape_": [53, 157, 160], "input_shape_s": [24, 25, 41, 90], "input_size_": [95, 117], "input_strides_": 160, "input_tensor": [0, 96, 205], "input_tensor_list": 211, "input_tensor_shap": 210, "input_tensor_total_el": 210, "input_tensor_valu": 210, "input_tensors_ptr": 172, "input_total_num_": 167, "input_w": [16, 32, 47, 48, 162], "input_w_": [16, 47, 48, 49, 50], "input_win": [30, 126], "input_x": [16, 37, 38, 39, 40, 43, 47, 48, 90, 204], "inputs_s": 172, "instal": [4, 233], "instanc": [97, 114], "instancenorm": [65, 227, 228, 229], "int": [0, 7, 8, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226], "int16": [10, 17, 19, 21, 22, 27, 28, 35, 36, 37, 41, 42, 44, 45, 46, 51, 54, 55, 59, 61, 67, 70, 72, 73, 76, 77, 78, 79, 80, 85, 88, 89, 92, 93, 98, 99, 104, 105, 108, 110, 111, 112, 116, 121, 122, 127, 129, 130, 132, 133, 137, 138, 139, 140, 141, 142, 147, 150, 151, 152, 153, 154, 155, 156, 159, 160, 161, 164, 166, 167, 168, 169, 170, 175, 176, 178, 183, 184, 185, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 202, 206, 208, 209, 210, 215, 216, 217, 218, 219, 221, 223, 224, 225, 226, 229, 235], "int16_t": [10, 17, 19, 21, 22, 27, 28, 35, 36, 37, 41, 42, 44, 45, 46, 51, 54, 55, 59, 61, 67, 70, 73, 78, 85, 88, 89, 90, 92, 93, 98, 99, 104, 105, 107, 108, 110, 111, 112, 116, 121, 122, 127, 129, 130, 132, 133, 137, 138, 139, 140, 142, 147, 150, 152, 153, 154, 155, 156, 164, 166, 167, 168, 169, 170, 175, 178, 183, 184, 185, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 199, 202, 206, 210, 215, 216, 217, 218, 219, 221, 222, 225, 226], "int32": [10, 17, 19, 21, 22, 24, 25, 27, 28, 35, 36, 37, 41, 42, 44, 45, 46, 51, 54, 55, 59, 61, 67, 70, 72, 73, 76, 77, 78, 79, 80, 85, 88, 89, 92, 93, 96, 98, 99, 104, 105, 108, 110, 111, 112, 116, 121, 122, 127, 129, 130, 132, 133, 137, 138, 139, 140, 141, 142, 147, 150, 151, 152, 153, 154, 155, 156, 159, 160, 161, 164, 166, 167, 168, 169, 170, 175, 176, 178, 183, 184, 185, 187, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 202, 206, 208, 209, 210, 215, 216, 217, 218, 219, 221, 223, 224, 225, 226, 229, 235], "int32_t": [10, 17, 19, 22, 24, 25, 27, 28, 35, 36, 37, 41, 42, 44, 45, 46, 47, 53, 54, 55, 59, 60, 61, 67, 70, 78, 88, 90, 92, 93, 98, 99, 104, 105, 108, 110, 111, 112, 116, 121, 122, 127, 129, 130, 132, 133, 134, 137, 138, 140, 142, 147, 150, 152, 153, 156, 164, 168, 170, 178, 183, 184, 185, 187, 189, 190, 192, 193, 195, 196, 197, 199, 202, 206, 210, 215, 216, 218, 219, 221, 222, 225, 226], "int64": [76, 77, 161], "int64_t": [218, 219], "int8": [4, 10, 12, 16, 17, 19, 20, 21, 22, 27, 28, 31, 33, 35, 36, 37, 39, 41, 42, 44, 45, 46, 47, 48, 51, 52, 53, 54, 55, 59, 60, 61, 66, 67, 68, 70, 72, 73, 76, 77, 78, 79, 80, 85, 86, 88, 89, 91, 92, 93, 95, 98, 99, 101, 103, 104, 105, 108, 110, 111, 112, 113, 116, 122, 127, 129, 130, 132, 133, 134, 136, 137, 138, 139, 140, 141, 142, 144, 146, 147, 150, 151, 152, 153, 154, 155, 156, 157, 159, 160, 161, 162, 164, 166, 167, 168, 169, 170, 174, 175, 176, 178, 179, 181, 182, 183, 184, 185, 187, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 202, 206, 210, 215, 216, 217, 218, 219, 221, 223, 224, 225, 226, 229, 235], "int8_t": [10, 12, 16, 17, 19, 21, 22, 27, 28, 31, 33, 35, 36, 37, 39, 41, 42, 44, 45, 46, 47, 48, 51, 53, 54, 55, 59, 60, 61, 66, 67, 68, 70, 73, 78, 85, 86, 88, 89, 90, 91, 92, 93, 95, 98, 99, 101, 103, 104, 105, 108, 110, 111, 112, 113, 116, 122, 127, 129, 130, 132, 133, 134, 136, 137, 138, 139, 140, 142, 144, 146, 147, 150, 152, 153, 154, 155, 156, 157, 162, 164, 166, 167, 168, 169, 170, 174, 175, 178, 179, 181, 182, 183, 184, 185, 187, 190, 191, 192, 193, 194, 195, 196, 197, 199, 202, 206, 210, 215, 217, 218, 219, 221, 222, 225, 226], "int_": 71, "int_param": 139, "integ": 163, "inter_linear": 0, "interest": 162, "interpol": 0, "intersec": 136, "invar": 34, "invers": [34, 76, 77, 158, 161], "invertpermut": [65, 227, 228, 229], "io": [4, 117], "iou": 136, "iou_threshold": 136, "iou_threshold_bit": 136, "ip": 231, "ir": [95, 231], "irff": [76, 77, 161], "irfft": [76, 77, 161], "is": [0, 24, 25, 37, 38, 69, 100, 104, 105, 187, 190, 225, 233], "is_broadcast": 170, "is_full_run": 20, "is_grad": 186, "is_regul": 69, "is_regulated_": 69, "is_relu": 100, "is_relu6": 100, "is_scalar": 190, "is_train": 34, "isfinit": [65, 227, 228, 229], "item": 213, "iz": 95, "j_": 88, "j_0": 88, "jpeg": 0, "jpg": [0, 4], "k_": 216, "k_h": [31, 58], "k_w": [31, 58], "ka": 4, "kconstant": [30, 126, 141], "keep_dim": 154, "keepdim": 0, "kernel": 47, "kernel_h": [49, 50, 58], "kernel_h_": [16, 47, 48, 49, 50], "kernel_s": [0, 47], "kernel_w": [49, 50, 58], "kernel_w_": [16, 47, 48, 49, 50], "key": 96, "keys_siz": 96, "keys_tensor": 96, "khann": [30, 126], "kn": [76, 77, 161], "knumbertypefloat32": [172, 204], "kpnna": 0, "kr": 4, "kreflect": 141, "ksymmetr": 141, "ktypeunknown": [211, 214], "l1": [16, 114, 179, 180], "l2": [10, 11, 14, 15, 17, 18, 19, 22, 23, 27, 28, 29, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 43, 44, 46, 49, 50, 51, 54, 56, 57, 58, 59, 63, 64, 67, 68, 70, 71, 73, 80, 81, 82, 83, 84, 85, 87, 90, 91, 92, 93, 94, 96, 97, 98, 99, 100, 101, 102, 106, 107, 108, 112, 113, 114, 115, 121, 123, 128, 132, 133, 134, 135, 137, 139, 141, 144, 145, 146, 147, 150, 155, 162, 163, 164, 165, 166, 170, 171, 173, 174, 175, 179, 180, 181, 183, 184, 185, 188, 189, 190, 191, 195, 196, 197, 200, 201, 202, 203, 205, 207, 208, 210, 213, 214, 217, 218, 219, 221, 222, 225, 226], "l2_norm_val": 100, "l2norm": [65, 227, 228], "l2normalizefus": 229, "l_k": 147, "l_n": 39, "l_relu_grad": 13, "label": [55, 57, 134, 135, 173, 174, 182, 186], "labelinfo": 57, "lambd": [12, 13], "lambda": [0, 12], "lamda": 4, "land": 112, "large_multipl": [62, 131], "large_shap": [62, 131], "large_strid": [62, 131], "layer": [69, 102], "layer_num": 69, "layer_num_": 69, "layer_s": 69, "layer_size_": 69, "layernorm": [101, 172], "layernorm_infershap": 172, "layernormfus": [65, 227, 228, 229], "layernormgrad": [65, 227, 228, 229], "layout": [24, 25], "lceil": [43, 147, 158, 162, 200], "ldot": [189, 210, 214], "le": [105, 116, 124, 125, 128, 144, 173, 178, 193], "leaki": [12, 13, 103], "leakyrelu": [65, 227, 228, 229], "leakyrelugrad": 13, "learn": [23, 171], "learning_r": [0, 14, 23, 171], "left": [12, 29, 34, 35, 36, 61, 66, 74, 97, 114, 115, 147, 162, 183, 184, 200], "left_matrix": 117, "left_shift_": 47, "left_x": 158, "leftarrow": [69, 96], "len": [0, 76, 77, 91, 161, 177], "length": [10, 12, 13, 15, 17, 19, 23, 27, 28, 37, 38, 42, 44, 51, 55, 61, 63, 64, 67, 68, 69, 70, 71, 73, 74, 75, 79, 82, 83, 99, 100, 101, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 122, 127, 129, 130, 132, 133, 137, 138, 140, 143, 146, 148, 149, 150, 152, 153, 154, 156, 163, 164, 166, 171, 173, 174, 175, 179, 180, 194, 196, 197, 211, 220, 225, 226], "length_in": 142, "leq": [180, 218], "lerp": 158, "less": [65, 105, 159, 227, 228, 229], "less_dim": 159, "lessequ": [65, 227, 228, 229], "lfloor": [84, 158], "lib": 0, "librari": [177, 227, 228, 230, 239], "likelihood": [134, 135], "limit": 147, "limit_k": 147, "linaro": 233, "line_buffers_": [53, 157], "linear": [12, 68, 91], "linspac": [65, 227, 228], "linux": 233, "lite": [0, 4, 229, 237], "ll": 56, "ln": [12, 20, 107, 108, 174], "load": 0, "load_checkpoint": 0, "load_param_into_net": 0, "loadmat": 4, "local": [115, 116], "log": [39, 65, 96, 108, 109, 113, 134, 135, 182, 186, 227, 228, 229], "log1p": [65, 227, 228, 229], "log_grad": 109, "log_mel": 126, "log_prob": 134, "loggrad": [65, 227, 228, 229], "logic": [21, 110, 111, 112], "logic_core_id": [16, 20, 47, 48, 52, 53, 79, 95, 154, 157, 159, 160, 167], "logical_and": 112, "logicaland": [65, 227, 228, 229], "logicalnot": [65, 227, 228, 229], "logicalor": [65, 227, 228, 229], "logit": [135, 173, 174, 182, 186], "logits_grad": 135, "logsoftmax": [65, 134, 172, 227, 228, 229], "logsoftmax_infershap": 172, "long": [20, 56, 58, 62, 118, 119, 120, 124, 125, 131, 141, 170, 177, 205], "lor": 111, "loss": [0, 39, 134, 135, 174, 179, 180, 182, 186], "loss_fn": 0, "loss_grad": 135, "loss_i": [182, 186], "lossmonitor": 0, "lp": 114, "lpnorm": 114, "lpnormal": [65, 227, 228], "lr": [14, 15], "lrelu": 12, "lrn": [65, 227, 228, 229], "lsh": 116, "lshproject": [65, 227, 228, 229], "lstm": [65, 118, 119, 120, 227, 228, 229], "lstm_param": 117, "lstmgrad": [65, 227, 228, 229], "lstmgraddata": [65, 227, 228, 229], "lstmgradweight": [65, 227, 228, 229], "lstmp": 117, "lstmparamet": 117, "m_": [14, 15, 171], "m_pts": 126, "m_t": [14, 15, 171], "main": [0, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226], "major": [29, 80], "make_shar": 0, "map": 0, "mask": [63, 64], "mat": [0, 4], "math": [100, 102, 115, 201], "mathbb": 186, "mathbf": 67, "mathcal": [124, 125, 148, 149, 220], "mathrel": [125, 158, 189], "mathrm": [12, 66, 136, 158], "matio": 4, "matmul": [4, 231], "matmulfus": [65, 227, 228, 229], "matplotlib": 4, "max": [12, 17, 20, 31, 44, 61, 66, 67, 69, 74, 75, 100, 115, 122, 124, 125, 146, 158, 189, 202], "max_": [124, 125, 162], "max_arr": 75, "max_c": 162, "max_classes_per_detect": 60, "max_detect": 60, "max_norm": 69, "max_norm_": 69, "max_output_per_class": 136, "max_siz": 145, "max_sizes_data": 145, "max_sizes_s": 145, "max_skip_s": 177, "max_v": 74, "max_val": [12, 31, 74, 75], "maxf": [124, 125], "maxi_": 47, "maximum": [65, 123, 136, 227, 228, 229], "maximumgrad": [65, 227, 228, 229], "maxmin": 172, "maxmingrad_infershap": 172, "maxpool": 125, "maxpool2d": 0, "maxpoolfus": [65, 227, 228, 229], "maxpoolgrad": [65, 227, 228, 229], "mean": [0, 33, 34, 39, 40, 87, 94, 102, 134, 135, 148], "mean_c": 87, "mean_j": 33, "mel": 126, "mel_htk": 126, "mel_scal": 126, "mel_spectrogram_output": 126, "meltyp": 126, "memcpi": [0, 16, 47, 48, 52, 53, 95, 157, 159, 172, 204], "memori": 145, "memset": [48, 158, 190, 201], "merg": 229, "method": 157, "method_": 157, "metric": 0, "mfcc": [65, 227, 228], "mfcc_i": 126, "mfcc_param": 126, "mfcc_workspac": 126, "mfccparam": 126, "mfccworkspaceparam": 126, "midir": 0, "min": [12, 17, 20, 31, 44, 66, 74, 75, 100, 115, 124, 125, 127, 146, 158, 202], "min_arr": 75, "min_siz": 145, "min_sizes_data": 145, "min_sizes_s": 145, "min_v": 74, "min_val": [12, 31, 74, 75], "mindir": [0, 4, 231], "mindradar": [4, 6, 7, 8], "mindspor": [0, 6, 7, 8, 95, 177, 229, 236, 237], "mindspore_py38": 233, "minf": [124, 125], "mingw32_arm": 233, "mini_": 47, "miniconda3": 233, "minimum": [65, 128, 227, 228, 229], "minimumgrad": [65, 227, 228, 229], "minsdpor": 237, "mobilenetv3": 12, "mod": [65, 227, 228, 229], "mode": [0, 154, 157], "mode_": 154, "model": [0, 4, 231], "model_buf": 231, "model_context": 231, "model_data": 231, "model_typ": 231, "modelfil": 0, "modetyp": 231, "modulus": 84, "moment": [23, 171], "momentum": 23, "mr": [4, 6, 7, 8], "ms": [0, 4, 6, 7, 8, 231], "mstensor": [0, 231], "mt6678": 139, "mt7004": [10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226], "mu": [101, 102], "mu_": [94, 97], "mul": [65, 131, 227, 228], "mulfus": 229, "mulgrad": [65, 227, 228, 229], "muller": 149, "multipl": 233, "multipli": 4, "multiplier_": 47, "mutabledeviceinfo": [0, 231], "my_model": 0, "n1": 210, "n2": 210, "n3": 210, "n_": 35, "n_1": [7, 8, 76, 77, 161], "n_d": [7, 8, 76, 77, 161], "n_dim": [113, 181], "n_fft": [30, 126], "n_i": [7, 8, 47, 76, 77, 161], "n_mel": 126, "n_mfcc": 126, "n_t": 95, "na": 4, "name": 0, "nan": [61, 99, 107, 109, 142, 163, 194], "nc4hw4": 85, "nc8hw8": 85, "nchw": [37, 85, 158], "ndim": [62, 91, 131, 154, 159, 160, 167, 178, 215, 224], "ndim_": [78, 80, 159, 160], "nearest": [157, 163], "need_grad": 182, "neg": [65, 110, 227, 228, 229], "negat": [134, 135, 142], "neggrad": [65, 227, 228, 229], "neq": [21, 70, 73, 110, 111, 112, 135, 138], "nesterov": [14, 23, 171], "net": 6, "netron": 0, "new_shap": 4, "next": 233, "ngram_siz": 177, "nhwc": [16, 31, 37, 38, 47, 48, 58, 85, 124, 125, 158, 185], "nllloss": [65, 135, 227, 228, 229], "nlllossgrad": [65, 227, 228, 229], "nlp": 177, "nms": [60, 136], "nms_candid": 60, "nms_iou_threshold": 60, "nms_score_threshold": 60, "nmsparam": 136, "nmsparamfp16": 136, "nmsparamfp32": 136, "nmsparamint8": 136, "nn": [0, 4, 231], "no": [104, 105], "non": 136, "non_zero_num": 137, "none": [39, 40, 121, 134, 135, 167, 233], "nonmaxsupppress": 229, "nonmaxsuppress": [65, 227, 228], "nonzero": [65, 227, 228, 229], "norm": [7, 8, 76, 77, 114, 126, 161], "norm_inner_s": 101, "norm_m": 126, "norm_mode_ortho": 126, "norm_outer_s": 101, "norm_slaney": 126, "normal": [30, 33, 34, 87, 94, 102, 115, 126], "normmod": 126, "normtyp": 126, "not": [56, 110], "notequ": [65, 227, 228, 229], "np": [0, 4, 6, 7, 8, 231], "nr": 4, "nudg": [74, 75], "nudge_max": 74, "nudge_min": 74, "null": [86, 116, 121, 170, 187], "num": [0, 20, 98, 106, 182, 186, 199], "num_ax": [154, 217], "num_axes_": 154, "num_box": [60, 136], "num_class": [0, 60], "num_classes_with_bg": 60, "num_classes_with_bg_": 60, "num_complex": 130, "num_dim": [18, 123, 128, 203], "num_direct": 95, "num_el": 212, "num_elem_": 160, "num_epoch": 0, "num_group": 94, "num_input": [45, 199], "num_of_class": 182, "num_roi": 162, "num_slic": 168, "num_split": [91, 192, 193], "num_str": 55, "num_unit": 206, "number": 159, "number_of_class": 186, "numpi": [0, 4, 6, 7, 8, 231], "numpy_exp": 0, "nweights_": [49, 50], "o_t": [117, 118, 119, 120], "odot": [95, 117, 118, 119, 120], "of": [37, 38, 92, 93, 159, 162, 233], "off": 139, "off_valu": 139, "offset": [52, 87, 94, 145, 177], "offset_": 52, "offset_c": [87, 94], "offset_s": 154, "omega_1": [7, 8, 76, 77, 161], "omega_d": [7, 8, 76, 77, 161], "omega_i": [7, 8, 76, 77, 161], "on": [139, 233], "on_off": 139, "on_valu": 139, "one": [0, 139, 159, 186, 231], "onehot": [65, 227, 228, 229], "onesid": [30, 76, 77, 126, 161], "oneslik": [65, 227, 228, 229], "op_parameter_": [172, 216], "opencv": 0, "opencv2": 0, "oper": 0, "operatornam": [24, 25, 29, 55, 71, 121, 124, 146, 163], "opparamet": [172, 216], "ops": [0, 4, 231], "optim": [0, 12, 92, 93, 104, 105], "or": 111, "org": 233, "ori": 188, "orig_seq_length": 159, "origin": 81, "origin_ndim": 81, "origin_shap": 81, "ortho": [7, 8, 76, 77, 161], "other": [117, 216], "otherwis": [12, 23, 55, 99, 100, 103, 104, 105, 112, 123, 125, 128, 139, 144, 158, 171, 179, 218, 219], "otim": 91, "out": [0, 6, 7, 8, 17, 35, 47, 73, 91, 110, 111, 122, 127, 129, 130, 158, 179, 185, 188, 189, 231], "out0": [192, 193, 224], "out1": [192, 193, 224], "out_addr": 158, "out_c": 58, "out_channel": [16, 47, 48, 49, 50], "out_data": [0, 91, 189, 213, 217], "out_data_shap": 189, "out_dim": 54, "out_elem": [218, 219], "out_h": [49, 50, 58], "out_height": 158, "out_i": [6, 16, 17, 47, 48, 158], "out_indic": 187, "out_indices_ptr": 188, "out_j": 47, "out_mean": 101, "out_ndim": [168, 169], "out_outshap": 188, "out_outshape_ptr": 188, "out_ptr": 162, "out_scal": 73, "out_shap": [52, 62, 131, 168, 169, 200], "out_shape_s": 200, "out_siz": 115, "out_strid": [24, 25, 62, 131, 141, 188, 217], "out_tensor0": 213, "out_tensor1": 213, "out_tensor2": 213, "out_valu": [24, 25, 187], "out_vari": 101, "out_w": [49, 50, 58], "out_width": 158, "out_x": 158, "outer": 200, "outer_count_": 159, "outer_loop_num_": 216, "outer_s": [21, 154], "outer_sizes_": 154, "outer_stride_": 159, "output": [0, 10, 11, 12, 13, 19, 20, 21, 22, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 45, 46, 51, 52, 53, 54, 59, 61, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 122, 124, 125, 126, 127, 129, 130, 132, 133, 136, 137, 138, 139, 140, 141, 142, 143, 145, 146, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 172, 173, 174, 175, 176, 178, 181, 182, 183, 184, 185, 186, 190, 191, 192, 193, 194, 195, 196, 197, 199, 200, 201, 202, 204, 205, 206, 207, 208, 209, 210, 211, 214, 215, 216, 217, 218, 219, 220, 221, 222, 224, 225, 226, 231], "output0": [210, 221], "output0_len": 221, "output1": 210, "output2": 210, "output_": [32, 75, 117, 139], "output_batch_": [16, 47, 48, 49, 50], "output_box": 60, "output_channel_": [16, 47, 48, 49, 50], "output_class": 60, "output_column": 0, "output_data": [16, 47, 48, 69, 145, 155, 159, 160, 167, 208, 209, 214], "output_dim": [20, 170], "output_dims_num": 170, "output_h": [32, 124, 125], "output_h_": [16, 47, 48, 49, 50], "output_hidden_st": 95, "output_i": [11, 12, 17, 26, 28, 61, 67, 70, 74, 99, 100, 104, 105, 108, 110, 111, 122, 127, 129, 130, 132, 133, 138, 140, 142, 143, 148, 149, 150, 152, 153, 156, 220, 224, 225], "output_index": 216, "output_j": 192, "output_label": [55, 57], "output_mfcc": [30, 126], "output_ndim": [168, 169], "output_np": 0, "output_num": [0, 57, 60], "output_num_": 154, "output_onsid": [30, 126], "output_prob": 0, "output_ptr": [113, 124, 125, 181], "output_rank": [22, 188], "output_reverse_index_map_ptr": 187, "output_s": [145, 158, 190, 208, 209], "output_scor": 60, "output_shap": [16, 20, 30, 41, 47, 48, 52, 53, 126, 141, 157, 168, 169, 172, 190, 217], "output_shape_": [53, 157], "output_shape_s": 41, "output_size_": 117, "output_step_": [95, 117], "output_strid": 190, "output_stride_": 167, "output_tensor": [96, 177, 210, 211], "output_tensors_ptr": 172, "output_unit_offset": 206, "output_valu": [24, 25], "output_w": [32, 124, 125], "output_w_": [16, 47, 48, 49, 50], "output_weight": [55, 57], "output_y_indices_ptr": 187, "output_y_values_ptr": 187, "output_zp_": 47, "outputfil": 0, "outputs_s": 172, "outter_s": [113, 181, 186], "overlap": 193, "p_": [134, 182, 186], "packed_input_": 117, "packed_output": 117, "packed_ptr": 117, "packed_st": 117, "packparam": [154, 167], "pad": [4, 16, 30, 47, 48, 49, 50, 124, 125, 126, 141, 183, 184], "pad_bottom": 31, "pad_d_": [49, 50], "pad_l": [32, 124, 125], "pad_l_": [16, 47, 48, 49, 50], "pad_left": 31, "pad_mod": [30, 126], "pad_r_": [49, 50], "pad_right": 31, "pad_top": 31, "pad_u": [32, 124, 125], "pad_u_": [16, 47, 48, 49, 50], "padarray": 4, "padding_mod": 141, "padfus": [65, 227, 228, 229], "para": 4, "parallel_on_out": 200, "parallel_on_split_axi": 200, "param": [16, 20, 30, 47, 48, 49, 50, 53, 56, 58, 60, 78, 80, 95, 118, 119, 120, 124, 125, 136, 141, 154, 157, 159, 160, 167, 172, 216], "param_dict": 0, "param_inner_s": 101, "param_not_load": 0, "param_num": 102, "param_outer_s": 101, "param_s": 102, "paramet": [0, 4, 31, 117, 145, 216], "parti": 233, "partial": [18, 58, 62, 135, 182, 186, 203], "partial_loss": 186, "partialfus": 229, "pass": [0, 64, 131, 165, 180, 195, 201, 203], "path": 233, "pcolor": 4, "per": [66, 75], "per_channel_": 47, "perm": 217, "phasemul": 229, "phi": 12, "pi": [4, 6, 7, 8, 12, 71, 76, 77, 161], "pil": 0, "pip": [4, 233], "plane": 85, "platform": 233, "plt": 4, "pmod": 129, "png": 0, "point": [66, 74, 136, 146], "pointer": 145, "pool": [31, 162], "pool1": 0, "pool2": 0, "pool3": 0, "pooled_height": 162, "pooled_width": 162, "pos": 201, "power": [30, 126, 143], "powergrad": [65, 227, 228, 229], "powfus": [65, 227, 228, 229], "pre": 63, "predict": [0, 179, 180, 231], "predicted_class": 0, "prelu": 144, "prelu_fus": 144, "prelufus": [65, 227, 228, 229], "preparecropandresizebilinear": 53, "prepareresizebicub": 157, "prepareresizebilinear": 157, "previous_output": 20, "previous_output_shap": 20, "prf": 4, "primit": 216, "print": [0, 4, 6, 7, 8, 231], "printf": 221, "prior": 145, "priorbox": [65, 227, 228, 229], "prob": 182, "probabl": 134, "process": 177, "prod_": [8, 21, 76, 77, 161, 189], "product": 29, "proj_col_align_": 117, "project_size_": 117, "promt": 233, "ptr": 0, "push_back": 0, "py": [0, 231], "py3": 233, "py_transform": 0, "pynative_mod": 0, "pyplot": 4, "python": 4, "python3": 233, "pytorch": 95, "q_": [66, 146], "q_i": [66, 146], "q_max": 146, "q_min": 146, "qk": 29, "qquad": [124, 125], "quad": [13, 61, 69, 73, 75, 76, 77, 78, 87, 94, 96, 98, 101, 106, 113, 124, 130, 140, 146, 147, 150, 161, 173, 181, 187, 189, 193, 224], "quant": [66, 74, 75], "quant_dtype_cast": 146, "quant_max": [74, 75], "quant_min": [74, 75], "quant_param": [16, 47], "quant_valu": 66, "quantdtypecast": [65, 227, 228, 229], "quantiz": [66, 146], "r0": 4, "r_t": 95, "radian": 51, "raggedrang": [65, 227, 228, 229], "rand": 20, "rand_max": 20, "randn": 0, "random": 0, "randomnorm": [65, 227, 228, 229], "randomstandardnorm": [65, 227, 228, 229], "rang": [4, 65, 147, 227, 228, 229], "range_count": 147, "rangfft": 229, "rank": [22, 65, 176, 187, 227, 228, 229], "rank_": 151, "rank_p": 151, "ratio": 145, "ratios_data": 145, "ratios_s": 145, "rceil": [43, 147, 158, 162, 200], "rd": 4, "rdsar": 3, "rdsarv3": 4, "re": [10, 56], "read": 0, "readfil": 231, "readimag": 0, "real": [76, 77, 130, 161], "real_valu": 66, "realdiv": [65, 227, 228, 229], "reciproc": [65, 164, 227, 228, 229], "rectifi": 12, "reduc": [65, 155, 227, 228], "reduce_asum": 154, "reduce_axi": 154, "reduce_l2norm": 154, "reduce_max": 154, "reduce_mean": 154, "reduce_min": 154, "reduce_prod": 154, "reduce_sum": 154, "reduce_sumsquar": 154, "reduce_typ": 155, "reducefus": 229, "reducemax": 155, "reducemean": 155, "reducemin": 155, "reduceparamet": 154, "reducescatt": [65, 227, 228, 229], "reducesum": 155, "reduct": [0, 39, 40, 135, 167], "reduction_typ": [134, 135, 167], "reduction_type_": 167, "refer": [227, 228, 230, 239], "reflect": 141, "region": 162, "regular": 60, "reinterpret_cast": 0, "releas": 233, "reload_cnn": 0, "relu": [0, 12, 13, 17, 20, 61, 86, 100, 103, 121, 202], "relu6": [12, 13, 17, 20, 31, 44, 86, 100, 121, 202], "relu6grad": 13, "relugrad": 13, "requir": 233, "rescale_to_0_1": 0, "rescale_transform": 0, "reserv": 0, "reshap": [0, 7, 8, 65, 80, 81, 188, 227, 228, 229], "resiz": [0, 65, 158, 159, 160, 227, 228, 229], "resize_op": 0, "resizebilineargrad": 158, "resized_imag": 0, "resized_image_tensor": 0, "resizegrad": [65, 227, 228, 229], "resizemethod": 157, "resizenearestneighborgrad": 158, "resizeparamet": 157, "reslut": 0, "respons": 115, "result": [0, 4, 56, 233], "result_len": 56, "return": [0, 4, 10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 49, 50, 51, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 78, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 155, 156, 158, 161, 162, 163, 164, 165, 166, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 224, 225, 226, 231], "return_valu": [24, 25], "revers": 229, "reverse_map": 187, "reversesequ": [65, 227, 228, 229], "reversesequenceparamet": 159, "reversev2": [65, 227, 228, 229], "reversev2paramet": 160, "rfft": [65, 76, 77, 227, 228], "rfloor": [84, 158], "rgb": 0, "right": [12, 29, 34, 35, 36, 61, 66, 74, 97, 114, 115, 147, 162, 183, 184, 200], "right_shift_": 47, "right_x": 158, "rightarrow": [136, 187], "rnn": [95, 117], "roi": 162, "roipool": [65, 227, 228, 229], "roll": 4, "root": 164, "round": [4, 65, 66, 74, 146, 158, 227, 228, 229], "roundf": 163, "row": [0, 29, 80, 187], "rsqrt": [65, 165, 227, 228, 229], "rsqrtgrad": [65, 227, 228, 229], "run_check": 233, "s_": 73, "s_h": 31, "s_i": [21, 55], "s_w": 31, "sample_r": 126, "sar": [0, 3], "sar_imaging_with_rd_cs_wk": 4, "satur": 4, "saturation_tensor": 4, "save_checkpoint": 0, "savefig": 4, "scalar": [92, 93], "scalar_input": [104, 105], "scale": [29, 33, 34, 63, 64, 66, 73, 74, 87, 94, 142, 143, 146, 148, 162, 166], "scale_c": [75, 87, 94], "scale_i": 166, "scalefus": [65, 227, 228, 229], "scatterel": [65, 227, 228], "scatterelementsparamet": 167, "scatternd": [65, 227, 228, 229], "scatterndupd": [65, 227, 228, 229], "scipi": 4, "score": [60, 136], "score_threshold": 136, "score_threshold_bit": 136, "scores_scal": 60, "scores_zero_point": 60, "scratch": [161, 187], "scratch_ptr": [76, 77, 161, 187], "seed": [148, 149, 220], "seed2": 220, "seed_": 116, "segment": [189, 222], "segment_id": 189, "segment_num": 66, "select": [60, 65, 227, 228, 229], "self": [0, 4, 12, 231], "selu": 229, "sensit": 116, "sentenc": 177, "seq": 117, "seq_dim": 159, "seq_dim_": 159, "seq_len": [29, 95], "seq_len_": [95, 117], "seq_length": 159, "serial": 0, "set_context": [0, 233], "setbuiltindeleg": 0, "sgd": [65, 227, 228, 229], "shape": [0, 4, 24, 25, 31, 45, 47, 48, 65, 72, 76, 77, 79, 80, 137, 159, 160, 161, 176, 177, 192, 198, 199, 223, 224, 227, 228, 229, 231], "shape0": 45, "shape1": 45, "shape2": 45, "shape_": [78, 80, 159, 172, 204], "shape_changed_": 204, "shape_s": [37, 38, 212], "shape_size_": [172, 204], "shared_ptr": 231, "shift": [4, 142, 143], "should": [37, 38, 159], "show": 4, "shrink": 13, "shrinkag": 12, "sigma": [12, 20, 34, 91, 94, 95, 97, 101, 102, 117, 173], "sigmoid": [12, 13, 20, 39, 91, 95, 117, 173, 174], "sigmoid_cross_entropi": 173, "sigmoidcroosentropywithlogit": 229, "sigmoidcroosentropywithlogitsgrad": 229, "sigmoidcrossentropywithlogit": [65, 227, 228], "sigmoidcrossentropywithlogitsgrad": [65, 227, 228], "sigmoidgrad": 13, "signal": [76, 77, 161, 177, 231, 233, 236], "signal_ndim": [7, 8, 76, 77, 161], "signal_s": [76, 77, 161], "sim": [148, 149, 220], "simd": 117, "simple_out": 136, "sin": [65, 227, 228, 229], "single_class_index": 60, "sio": 4, "size": [0, 4, 11, 17, 47, 48, 65, 84, 92, 93, 102, 159, 165, 178, 186, 195, 202, 205, 207, 208, 227, 228, 229], "size_": 176, "size_k": 178, "size_p": 176, "size_t": [0, 231], "sizeof": [0, 16, 35, 36, 41, 47, 48, 52, 53, 59, 72, 78, 80, 95, 157, 158, 159, 172, 183, 184, 185, 190, 198, 201, 204, 208, 223, 224], "skip": 177, "skipgram": [65, 227, 228, 229], "skipgram_": 177, "skipgram_p": 177, "slice": [52, 65, 200, 210, 216, 224, 227, 228], "slice_i": [24, 25], "slicefus": 229, "slici": 0, "slope": [126, 144], "sm": 29, "small_multipl": [62, 131], "small_shap": [62, 131], "small_strid": [62, 131], "smc": [121, 147, 154], "smooth": 180, "smoothl1loss": [65, 227, 228, 229], "smoothl1lossgrad": [65, 227, 228, 229], "soft": 12, "soft_copy_mod": 200, "softmax": [29, 65, 113, 172, 182, 186, 227, 228, 229], "softmax_infershap": 172, "softmax_out": 29, "softmaxcrossentropywithlogit": [65, 186, 227, 228], "softmaxgrad": 229, "softplus": [12, 13, 20, 229], "softplusgrad": 13, "softshrink": [12, 13], "softshrinkgrad": 13, "softsign": 12, "softsignopt": 12, "sort": 216, "sorted_": 216, "space": [34, 55, 56, 59], "spacetobatch": [65, 227, 228, 229], "spacetobatchnd": [65, 227, 228, 229], "spacetodepth": [65, 227, 228, 229], "spars": 190, "sparse_length": 190, "sparse_valu": 190, "sparsefillemptyrow": [65, 227, 228], "sparsereshap": [65, 227, 228], "sparsereshape_": 188, "sparsereshape_p": 188, "sparsesegmentsum": [65, 227, 228], "sparsesoftmaxcrossentropywithlogit": [65, 227, 228, 229], "sparsetodens": [65, 227, 228, 229], "spec_f": [30, 126], "spec_param": [30, 126], "spec_workspac": [30, 126], "spectrogram": 30, "spectrogram_output": 126, "spectrogramparam": [30, 126], "splice": [65, 227, 228, 229], "split": [0, 65, 147, 193, 227, 228, 229], "split_axi": 200, "split_data": 91, "split_data_buf": 91, "split_dim": 91, "split_siz": [91, 192], "splitwithoverlap": [65, 227, 228, 229], "sq_fa_axi": 4, "sq_fr_axi": 4, "sqrt": [6, 7, 8, 10, 12, 14, 15, 29, 33, 34, 65, 71, 76, 77, 87, 94, 97, 100, 101, 102, 161, 164, 165, 195, 227, 228, 229], "sqrt_sum": 100, "sqrtf": 100, "sqrtgrad": [65, 227, 228, 229], "squar": [4, 65, 164, 227, 228, 229], "squareddiffer": 229, "squarediffer": [65, 227, 228], "squeez": [0, 65, 227, 228, 229], "srand": [20, 141], "src": [27, 44, 53, 72, 74, 75, 101, 144, 159, 160, 164, 196, 198, 211, 218, 219, 223], "src0": 13, "src0_i": [13, 19], "src1": 13, "src1_i": [13, 19], "src_": 33, "src_col": 191, "src_data": [10, 21, 51, 73, 82, 85, 101, 144, 154, 166, 175, 191, 194], "src_data0": 83, "src_data1": 83, "src_format": 85, "src_i": [10, 27, 42, 51, 73, 82, 166, 175, 194], "src_ptr": 0, "src_row": 191, "ssh": 231, "stack": [65, 177, 227, 228, 229], "start": [14, 15, 23, 46, 106, 124, 144, 147, 150, 171, 193], "start_idx": [32, 193], "start_indic": 193, "start_k": 147, "state_bia": [95, 117], "state_bias_": 117, "state_col_align": 95, "state_col_align_": [95, 117], "state_g": 117, "state_row_align_": [95, 117], "std": [0, 231], "stdbool": [23, 26, 74, 75, 121, 142, 147, 150, 171], "stderr": 0, "stdint": [182, 186], "stdio": [10, 11, 12, 13, 14, 15, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 49, 50, 51, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 73, 74, 75, 78, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 96, 97, 98, 99, 100, 101, 102, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 146, 147, 148, 149, 150, 151, 152, 153, 155, 156, 158, 161, 162, 163, 164, 165, 166, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 224, 225, 226], "stdlib": 20, "step": [4, 106], "step_h": 145, "step_w": 145, "str": [7, 8, 56, 76, 77, 161], "str_len": 56, "strict_load": 0, "stride": [0, 16, 24, 25, 47, 48, 49, 50, 91, 124, 125, 159, 200, 201, 206, 215, 217], "stride0": 190, "stride1": 190, "stride2": 190, "stride3": 190, "stride_h": [31, 32, 124, 125], "stride_h_": [16, 47, 48, 49, 50], "stride_w": [31, 32, 124, 125], "stride_w_": [16, 47, 48, 49, 50], "stridedslic": [65, 201, 227, 228, 229], "stridedslicegrad": [65, 227, 228, 229], "strides_": 159, "string": [0, 55, 56, 214], "string_length": 55, "string_point": 55, "stringpack": 177, "struct": [16, 30, 47, 48, 53, 60, 78, 95, 117, 126, 136, 154, 157, 159, 160, 167, 177, 216], "sub": 203, "subext": 202, "subfus": [65, 227, 228, 229], "subgrad": [65, 227, 228, 229], "subrelu": 202, "subrelu6": 202, "success": 233, "sum": [0, 39, 40, 134, 135], "sum_": [7, 8, 16, 31, 34, 38, 47, 54, 58, 69, 76, 77, 86, 97, 100, 101, 102, 114, 115, 161, 182, 186, 188, 222], "sum_data": [113, 181, 182, 186], "sum_j": [113, 181], "sum_of_squar": 100, "sum_row": 0, "sum_t": [118, 119, 120], "super": [0, 4, 231], "support_neg_index": 139, "suppress": 136, "swish": [12, 13, 20], "switch": [65, 227, 228, 229], "switch_": 204, "switch_p": 204, "switchlay": [65, 227, 228, 229], "switchlayer_": 205, "switchlayer_p": 205, "symmetr": [74, 75, 141], "sys_bar": [16, 47, 48, 52, 53, 79, 95, 154, 157, 159, 160, 167], "system": 233, "t62": 0, "t_": 42, "ta_axi": 4, "ta_gap": 4, "tangent": 12, "tanh": [12, 13, 20, 95, 117, 118, 119, 120], "tanhgrad": 13, "tar": 233, "target": [0, 179, 180], "task_id": 101, "tdpp": 229, "temp": [4, 24, 25, 39, 100], "temp1": 4, "temp2": 4, "temp3": 4, "temp_spac": 200, "tensor": [0, 4, 6, 7, 8, 47, 53, 66, 76, 77, 79, 92, 93, 96, 125, 160, 161, 167, 192, 198, 207, 214, 231], "tensor0": 207, "tensor1": [207, 214], "tensor2": [207, 214], "tensor_c_shap": 212, "tensor_c_shape_s": 212, "tensor_data": 214, "tensor_data_typ": 214, "tensor_element_num": 214, "tensor_list_c_element_shap": 212, "tensor_list_c_element_shape_s": 212, "tensor_num": 214, "tensorarray": [65, 227, 228], "tensorarrayread": [65, 207, 209, 227, 228], "tensorarrayread_": 207, "tensorarrayread_p": 207, "tensorarraywrit": [65, 207, 227, 228], "tensorarraywrite_": 207, "tensorarraywrite_p": 207, "tensorc": [172, 204, 205], "tensorlist": [212, 213, 214], "tensorlist_s": 213, "tensorlistfromtensor": [65, 227, 228, 229], "tensorlistgetitem": [65, 227, 228, 229], "tensorlistgetitem_": 211, "tensorlistgetitem_p": 211, "tensorlistreserv": [65, 227, 228, 229], "tensorlistsetitem": [65, 227, 228, 229], "tensorlistsetitem_": 213, "tensorlistsetitem_p": 213, "tensorliststack": [65, 227, 228, 229], "tensorliststack_": 214, "tensorliststack_p": 214, "tensorscatteradd": [65, 227, 228, 229], "test": 55, "test_fp_affine_": 20, "test_matmul": 231, "testadderl2fp32": 16, "testaddersmcfp32": 16, "testconvl2fp32": 47, "testconvsmcfp32": 47, "testconvtransposel2fp32": 48, "testconvtransposesmcfp32": 48, "testcropandresizesmcfp32": 53, "testcropsmcfp32": 52, "testgrul2fp32": 95, "testgrusmcfp32": 95, "testreducel2fp32": 154, "testreducesmcfp32": 154, "testresizefp32smc": 157, "testreversesequencefp32": 159, "testreversev2": 160, "testscatterelementsl2": 167, "testscatterelementssmc": 167, "text": [10, 11, 12, 13, 21, 23, 26, 29, 31, 32, 35, 37, 43, 44, 45, 47, 49, 50, 54, 55, 58, 61, 63, 64, 67, 68, 69, 70, 71, 73, 74, 75, 76, 77, 78, 80, 81, 84, 88, 90, 91, 92, 93, 96, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 109, 112, 113, 114, 115, 117, 121, 123, 124, 125, 128, 130, 131, 134, 135, 136, 138, 139, 140, 142, 144, 146, 147, 148, 150, 158, 161, 162, 163, 164, 165, 170, 171, 174, 179, 180, 181, 185, 187, 188, 189, 190, 192, 193, 195, 196, 197, 199, 200, 201, 202, 203, 204, 205, 207, 208, 209, 210, 211, 214, 215, 218, 219, 221, 222, 224, 225], "textbf": 69, "tfrac": 12, "the": [37, 38, 113, 159, 181, 233], "third": 233, "thread_num": 101, "tile": [65, 227, 228], "tile_data0": [62, 131], "tile_data1": [62, 131], "tile_data2": 62, "tile_dim": 215, "tile_num": 215, "tilefus": 229, "time": [0, 20, 29, 31, 35, 74, 115, 121, 125, 130, 131, 141, 142, 143, 146, 150, 159, 165, 180, 188, 189, 195, 196, 202, 203, 218, 219], "titl": 4, "tmp_dst_data": 154, "tmp_input": 154, "tmp_input_shap": 154, "tmp_loss": 39, "tmp_output": 154, "tmp_src_data": 154, "tmp_str": 56, "to": [163, 233], "toolchain": 233, "top": [29, 35, 36, 50, 57, 183, 184, 216], "top_i": 158, "topk": [24, 25, 216], "topk_fus": 216, "topk_node_list_": 216, "topkfus": [65, 227, 228, 229], "topkparamet": 216, "total_copy_s": [72, 198, 223], "total_el": 170, "total_num": 154, "total_num_": 154, "total_w": 134, "total_weight": [134, 135], "tr": 4, "tr_axi": 4, "train": 0, "train_data": 0, "trainable_param": 0, "transform": 0, "transpos": [58, 65, 227, 228, 229], "tril": [65, 227, 228], "triu": [65, 227, 228], "true": [0, 4, 12, 23, 24, 25, 26, 30, 34, 39, 40, 54, 60, 70, 74, 75, 76, 77, 92, 93, 104, 105, 110, 111, 112, 117, 126, 136, 138, 142, 161, 170, 171, 204, 216, 225], "tupl": [76, 77, 161], "twiddl": [76, 77, 161], "twod": 0, "txt": [0, 231, 233], "type_": 172, "type_s": [52, 79, 160, 205, 207], "type_size_": [78, 80, 159, 160], "typecast": 0, "typedef": [16, 30, 47, 48, 53, 60, 78, 95, 117, 126, 136, 154, 157, 159, 160, 167, 177, 216], "typeerror": [76, 77, 161], "u_t": 171, "uint32": 224, "uint64_t": [136, 186], "uint8": [0, 76, 77, 96, 161, 229], "uint8_t": [60, 96], "underset": [24, 25], "uniformr": [65, 227, 228, 229], "uniqu": [65, 227, 228, 229], "unit": [12, 33, 68, 87, 91, 94], "unknown_type_offset": 214, "unsign": [148, 149, 170], "unsort": 222, "unsorted_segment_sum": 222, "unsortedsegmentsum": [65, 227, 228, 229], "unsqueez": [4, 65, 227, 228, 229], "unstack": [65, 227, 228, 229], "unstack_": 224, "unstack_p": 224, "up_div": 66, "up_slop": 126, "updat": [167, 168, 169, 206], "update_t": 23, "updates_data": 167, "updates_i": 168, "upstream": [34, 40, 64], "use_regular_nm": 60, "user": 233, "util": [13, 45, 74, 75, 88, 89, 110, 111, 116, 122, 127, 129, 130, 140, 156, 168, 169, 178, 192, 193, 199, 215], "v3": 4, "v_": [14, 15, 124], "v_t": [14, 15], "val_i": 46, "val_r": 46, "valu": [24, 25, 46, 78, 79, 96, 147, 187], "value_imag": 46, "value_r": 46, "valueerror": [76, 77, 161], "values_ptr": 187, "values_tensor": 96, "var": [15, 33, 102], "var_": 15, "var_t": 15, "varianc": [33, 34, 87, 94, 102], "variance_c": 87, "variance_j": 33, "vec": 0, "vecatan": 229, "vector": [0, 231], "version": 233, "vision": 0, "void": [0, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175, 176, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 231], "vr": 4, "vscode": 231, "w_": [14, 35, 47, 48, 95, 117, 124, 134, 135, 171, 185], "w_i": [124, 125], "w_n": [39, 40], "w_o": [124, 125], "w_scale": 60, "w_t": [14, 171], "we": 56, "weight": [14, 15, 16, 23, 39, 40, 47, 48, 55, 57, 116, 134, 135, 171], "weight_": 23, "weight_decay": 171, "weight_defin": [39, 40], "weight_g": 95, "weight_h": 117, "weight_i": 117, "weight_r": 95, "weight_shap": [16, 47, 48], "weight_t": 23, "weight_threshold": 57, "where": [4, 65, 73, 101, 227, 228, 229], "which": [37, 38], "whl": 233, "width": [0, 35, 36, 59, 158, 183, 184, 218, 219], "width_scal": 158, "will": 56, "win": [124, 125], "win_amd64": 233, "win_h": [31, 124, 125], "win_length": [30, 126], "win_w": [31, 124, 125], "window": [32, 231, 233], "window_h": 32, "window_typ": [30, 126], "window_w": 32, "windowtyp": [30, 126], "with": 173, "word": 177, "workspac": [16, 24, 25, 30, 39, 47, 182, 186], "workspace_": [16, 47, 48, 49, 50], "workspaceparam": [30, 126], "world": 55, "www": 233, "x1": [0, 18, 53, 62, 131, 162, 180, 203, 231], "x1_data": [62, 131], "x1_dim": [18, 203], "x1_size": [62, 131], "x1_tensor": 231, "x2": [18, 53, 62, 131, 162, 180, 203, 231], "x2_data": [62, 131], "x2_dim": [18, 203], "x2_size": [62, 131], "x2_tensor": 231, "x86": 233, "x_": [24, 25, 69, 87, 94, 97, 114, 135, 158, 182, 186], "x_1": [62, 179], "x_2": [62, 179], "x_b": 58, "x_data": 58, "x_h": 32, "x_i": [66, 146, 173], "x_j": 102, "x_lefts_": [53, 157], "x_n": [39, 40], "x_rights_": [53, 157], "x_scale": 60, "x_shape": 204, "x_t": [95, 117, 118, 119, 120], "x_tensor": [6, 7, 8], "x_w": 32, "x_weights_": [53, 157], "xmax": 145, "xmin": 145, "xxx": [231, 233, 238], "xxx_add_sub_xxx": 235, "xxx_add_xxx": 235, "xxx_mul_xxx": 235, "xxx_xxx_p": 235, "xxx_xxx_s": 235, "xxx_xxx_xxx": 235, "xz": 233, "y1": [53, 162], "y2": [53, 162], "y_": [87, 94, 97, 114, 158, 182], "y_bottoms_": [53, 157], "y_h": 32, "y_i": [21, 24, 25, 134, 135, 173, 186], "y_k": 69, "y_n": [39, 40], "y_scale": 60, "y_shape": 204, "y_tops_": [53, 157], "y_w": 32, "y_weights_": [53, 157], "yhft": [0, 4, 231], "ymax": 145, "ymin": 145, "z_t": 95, "zero": [66, 74, 146], "zero_point": 74, "zerolik": 226, "zeroslik": [65, 227, 228, 229], "zoneout": 117, "zoneout_cell_": 117, "zoneout_hidden_": 117, "zp": [66, 146], "zsu_23_4": 0}, "titles": ["GRAY_CNN(\u56fe\u7247\u7070\u5ea6\u5316\u5904\u7406+\u56fe\u7247\u8bc6\u522b)", "AI+DSP\u5e94\u7528\u793a\u4f8b", "AI\u8f85\u52a9\u5f00\u53d1\u793a\u4f8b", "DSP\u5e94\u7528\u793a\u4f8b", "RDSAR\uff08\u8ddd\u79bb-\u591a\u666e\u52d2SAR\u6210\u50cf\u7b97\u6cd5\uff09", "\u5e94\u7528\u5f00\u53d1\u793a\u4f8b", "ComplexAbs", "FFT", "IFFT", "\u81ea\u5b9a\u4e49\u7b97\u5b50\u5217\u8868", "Abs", "Absgrad", "Activation", "ActivationGrad", "Adam", "AdamWeightDecay", "Adder", "Addfusion", "Addgrad", "AddN", "Affine", "All", "AllGather", "ApplyMomentum", "Argmax", "Argmin", "Assert", "Assign", "AssignAdd", "Attention", "AudioSpectrogram", "Avgpooling", "AvgPoolingGrad", "BatchNorm", "Batchnormgrad", "BatchToSpace", "BatchToSpaceND", "Biasadd", "Biasaddgrad", "Binarycrossentropy", "Binarycrossentropygrad", "BroadcastTo", "Cast", "Ceil", "Clip", "Concat", "ConstantOfShape", "Conv2d", "Conv2dTranspose", "Conv2DBackpropFilterFusion", "Conv2DBackpropInputFusion", "Cos", "Crop", "CropAndResize", "Cumsum", "CustomExtractFeatures", "Customnormalize", "CustomPredict", "DeconvGradFilter", "DepthToSpace", "DetectionPostProcess", "DivFusion", "Divgrad", "Dropout", "Dropoutgrad", "DSP Library C API Reference", "DynamicQuant", "Eltwise", "Elu", "EmbeddingLookup", "Equal", "Erf", "ExpandDims", "ExpFusion", "FakeQuantWithMinMaxVars", "FakeQuantWithMinMaxVarsPerChannel", "FFTImag", "FFTReal", "Fill", "FillV2", "Flatten", "FlattenGrad", "Floor", "FloorDiv", "Floormod", "FormatTranspose", "FullConnection", "FusedBatchNorm", "Gather", "GatherNd", "GatherD", "GLU", "Greater", "Greaterequal", "GroupNormFusion", "GRU", "HashtableLookup", "InstanceNorm", "InvertPermutation", "Isfinite", "L2norm", "LayerNormFusion", "Layernormgrad", "LeakyReLu", "Less", "Lessequal", "LinSpace", "Log", "Log1p", "LogGrad", "LogicalNot", "LogicalOr", "LogicalAnd", "LogSoftmax", "LpNormalization", "Lrn", "LshProjection", "LSTM", "LstmGrad", "LstmGradData", "LstmGradWeight", "MatMulFusion", "Maximum", "Maximumgrad", "MaxPoolFusion", "MaxPoolGrad", "MFCC", "Minimum", "Minimumgrad", "Mod", "Mul", "Mulgrad", "Neg", "NegGrad", "NLLLoss", "NLLLossGrad", "NonMaxSuppression", "NonZero", "NotEqual", "OneHot", "OnesLike", "PadFusion", "PowFusion", "PowerGrad", "PReLUFusion", "Priorbox", "QuantDTypeCast", "RaggedRange", "RandomNormal", "RandomStandardNormal", "Range", "Rank", "RealDiv", "Reciprocal", "Reduce", "ReduceScatter", "Reshape", "Resize", "Resizegrad", "ReverseSequence", "ReverseV2", "RFFT", "ROIPooling", "Round", "Rsqrt", "Rsqrtgrad", "ScaleFusion", "ScatterElements", "ScatterNd", "ScatterNdUpdate", "Select", "SGD", "Shape", "SigmoidCrossEntropyWithLogitsGrad", "SigmoidCrossEntropyWithLogits", "Sin", "Size", "Skipgram", "Slice", "SmoothL1Loss", "Smoothl1lossgrad", "Softmax", "SoftmaxCrossEntropyWithLogits", "SpaceToBatch", "SpaceToBatchND", "SpaceToDepth", "SparseSoftmaxCrossEntropyWithLogits", "SparseFillEmptyRows", "Sparsereshape", "Sparsesegmentsum", "Sparsetodense", "Splice", "Split", "SplitWithOverlap", "Sqrt", "Sqrtgrad", "Square", "Squaredifference", "Squeeze", "Stack", "Stridedslice", "Stridedslicegrad", "Subfusion", "Subgrad", "Switch", "Switchlayer", "TensorScatterAdd", "Tensorarray", "Tensorarrayread", "TensorarrayWrite", "Tensorlistfromtensor", "Tensorlistgetitem", "Tensorlistreserve", "Tensorlistsetitem", "Tensorliststack", "Tile", "TopkFusion", "Transpose", "Tril", "Triu", "UniformReal", "Unique", "UnsortedSegmentSum", "UnSqueeze", "Unstack", "Where", "ZerosLike", "\u7b97\u5b50\u5e93\u652f\u6301", "\u7b97\u5b50\u5e93\u652f\u6301", "\u7b97\u5b50\u5e93\u652f\u6301\u60c5\u51b5", "MindSpore Signal+ \u4f7f\u7528\u624b\u518c", "HelloDSP", "\u5feb\u901f\u5165\u95e8", "\u73af\u5883\u5b89\u88c5", "\u6574\u4f53\u6982\u89c8", "DSP\u7b97\u5b50C\u63a5\u53e3\u547d\u540d\u89c4\u8303", "\u53c2\u8003\u8d44\u6599", "\u5b98\u65b9\u8d44\u6599", "MindSpore Signal+ \u8c03\u5ea6\u65b9\u6848", "MindSpore Signal+ \u4f7f\u7528\u624b\u518c"], "titleterms": {"abs": 10, "absgrad": 11, "activ": 12, "activationgrad": 13, "adam": 14, "adamweightdecay": 15, "adder": 16, "addfus": 17, "addgrad": 18, "addn": 19, "affin": 20, "ai": [1, 2], "all": 21, "allgath": 22, "api": 65, "applymomentum": 23, "argmax": 24, "argmin": 25, "assert": 26, "assign": 27, "assignadd": 28, "attent": 29, "audiospectrogram": 30, "avgpool": 31, "avgpoolinggrad": 32, "batchnorm": 33, "batchnormgrad": 34, "batchtospac": 35, "batchtospacend": 36, "biasadd": 37, "biasaddgrad": 38, "binarycrossentropi": 39, "binarycrossentropygrad": 40, "broadcastto": 41, "cast": 42, "cc": 231, "ceil": 43, "clip": 44, "cmake": 233, "complexab": 6, "concat": 45, "conda": 233, "constantofshap": 46, "conv2d": 47, "conv2dbackpropfilterfus": 49, "conv2dbackpropinputfus": 50, "conv2dtranspos": 48, "cos": 51, "crop": 52, "cropandres": 53, "cumsum": 54, "customextractfeatur": 55, "customnorm": 56, "custompredict": 57, "deconvgradfilt": 58, "depthtospac": 59, "detectionpostprocess": 60, "divfus": 61, "divgrad": 62, "dropout": 63, "dropoutgrad": 64, "dsp": [1, 3, 65, 235], "dynamicqu": 66, "eltwis": 67, "elu": 68, "embeddinglookup": 69, "equal": 70, "erf": 71, "expanddim": 72, "expfus": 73, "fakequantwithminmaxvar": 74, "fakequantwithminmaxvarsperchannel": 75, "fft": 7, "fftimag": 76, "fftreal": 77, "fill": 78, "fillv2": 79, "flatten": 80, "flattengrad": 81, "floor": 82, "floordiv": 83, "floormod": 84, "formattranspos": 85, "fullconnect": 86, "fusedbatchnorm": 87, "gather": 88, "gatherd": 90, "gathernd": 89, "glu": 91, "gray_cnn": 0, "greater": 92, "greaterequ": 93, "groupnormfus": 94, "gru": 95, "hashtablelookup": 96, "hellodsp": 231, "ide": 233, "ifft": 8, "instancenorm": 97, "invertpermut": 98, "isfinit": 99, "l2norm": 100, "layernormfus": 101, "layernormgrad": 102, "leakyrelu": 103, "less": 104, "lessequ": 105, "librari": 65, "linspac": 106, "lite": 231, "log": 107, "log1p": 108, "loggrad": 109, "logicaland": 112, "logicalnot": 110, "logicalor": 111, "logsoftmax": 113, "lpnormal": 114, "lrn": 115, "lshproject": 116, "lstm": 117, "lstmgrad": 118, "lstmgraddata": 119, "lstmgradweight": 120, "main": 231, "matlab": 4, "matmulfus": 121, "maximum": 122, "maximumgrad": 123, "maxpoolfus": 124, "maxpoolgrad": 125, "mfcc": 126, "mindradar": 233, "mindspor": [4, 230, 231, 233, 238, 239], "minimum": 127, "minimumgrad": 128, "mod": 129, "mt7004": 231, "mul": 130, "mulgrad": 131, "neg": 132, "neggrad": 133, "netron": 233, "nllloss": 134, "nlllossgrad": 135, "nonmaxsuppress": 136, "nonzero": 137, "notequ": 138, "onehot": 139, "oneslik": 140, "padfus": 141, "powergrad": 143, "powfus": 142, "prelufus": 144, "priorbox": 145, "python": [0, 231, 233], "quantdtypecast": 146, "radar": 233, "raggedrang": 147, "randomnorm": 148, "randomstandardnorm": 149, "rang": 150, "rank": 151, "rdsar": 4, "realdiv": 152, "reciproc": 153, "reduc": 154, "reducescatt": 155, "refer": 65, "reshap": 156, "resiz": 157, "resizegrad": 158, "reversesequ": 159, "reversev2": 160, "rfft": 161, "roipool": 162, "round": 163, "rsqrt": 164, "rsqrtgrad": 165, "sar": 4, "scalefus": 166, "scatterel": 167, "scatternd": 168, "scatterndupd": 169, "select": 170, "sgd": 171, "shape": 172, "sigmoidcrossentropywithlogit": 174, "sigmoidcrossentropywithlogitsgrad": 173, "signal": [4, 230, 238, 239], "sin": 175, "size": 176, "skipgram": 177, "slice": 178, "smoothl1loss": 179, "smoothl1lossgrad": 180, "softmax": 181, "softmaxcrossentropywithlogit": 182, "spacetobatch": 183, "spacetobatchnd": 184, "spacetodepth": 185, "sparsefillemptyrow": 187, "sparsereshap": 188, "sparsesegmentsum": 189, "sparsesoftmaxcrossentropywithlogit": 186, "sparsetodens": 190, "splice": 191, "split": 192, "splitwithoverlap": 193, "sqrt": 194, "sqrtgrad": 195, "squar": 196, "squarediffer": 197, "squeez": 198, "stack": 199, "stridedslic": 200, "stridedslicegrad": 201, "subfus": 202, "subgrad": 203, "switch": 204, "switchlay": 205, "tensorarray": 207, "tensorarrayread": 208, "tensorarraywrit": 209, "tensorlistfromtensor": 210, "tensorlistgetitem": 211, "tensorlistreserv": 212, "tensorlistsetitem": 213, "tensorliststack": 214, "tensorscatteradd": 206, "tile": 215, "topkfus": 216, "transpos": 217, "tril": 218, "triu": 219, "uniformr": 220, "uniqu": 221, "unsortedsegmentsum": 222, "unsqueez": 223, "unstack": 224, "where": 225, "yhft": 233, "zeroslik": 226}}) \ No newline at end of file