diff --git a/src/plugins/intel_gpu/include/intel_gpu/plugin/primitives_list.hpp b/src/plugins/intel_gpu/include/intel_gpu/plugin/primitives_list.hpp index ceba5be5a5d..2377670585f 100644 --- a/src/plugins/intel_gpu/include/intel_gpu/plugin/primitives_list.hpp +++ b/src/plugins/intel_gpu/include/intel_gpu/plugin/primitives_list.hpp @@ -94,7 +94,7 @@ REGISTER_FACTORY(v0, Unsqueeze); REGISTER_FACTORY(v1, Add); REGISTER_FACTORY(v1, AvgPool); REGISTER_FACTORY(v1, BatchToSpace); -REGISTER_FACTORY(v1, BinaryConvolution); +// REGISTER_FACTORY(v1, BinaryConvolution); Supported via BinaryConvolution->Convolution conversion REGISTER_FACTORY(v1, Broadcast); REGISTER_FACTORY(v1, ConvertLike); REGISTER_FACTORY(v1, Convolution); diff --git a/src/plugins/intel_gpu/include/intel_gpu/primitives/binary_convolution.hpp b/src/plugins/intel_gpu/include/intel_gpu/primitives/binary_convolution.hpp deleted file mode 100644 index 5ba028bda78..00000000000 --- a/src/plugins/intel_gpu/include/intel_gpu/primitives/binary_convolution.hpp +++ /dev/null @@ -1,124 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once -#include "primitive.hpp" -#include "openvino/core/coordinate_diff.hpp" -#include "openvino/core/strides.hpp" -#include - -namespace cldnn { - -/// @brief Performs forward spatial binary_convolution with weight sharing. -struct binary_convolution : public primitive_base { - CLDNN_DECLARE_PRIMITIVE(binary_convolution) - - binary_convolution() : primitive_base("", {}) {} - - /// @brief Constructs binary_convolution primitive. - /// @param id This primitive id. - /// @param input Input primitive id. - /// @param weights List of primitive ids containing weights data. - /// @param pad Defines logical pad value added to input tensor - /// @param stride Defines shift in input buffer between adjacent calculations of output values. - /// @param dilation Defines gaps in the input - dilation rate k=1 is normal binary_convolution, - /// k=2 means skipping one pixel per input, k=4 means skipping 3 pixels. - /// As an example in one dimension, a filter w of size 3 would compute over input x the following: w[0]*x[0] + w[1]*x[1] + w[2]*x[2] for dilation of 1. - /// For dilation 2 the filter would instead compute w[0]*x[0] + w[1]*x[2] + w[2]*x[4]. - /// @param output_size User-defined output data size of the primitive (w/o padding). - /// @param groups Number of feature groups (grouped convolution). If more than 1 then weights/bias count needs to be 1. - /// @param pad_value Logical value of padding. Can be one of 3 values: 1 - pad bits equal to 1; -1 -> pad bits equal to 0; 0 -> pad is not counted - /// @param calc_precision Precision of intermediate accumulators - binary_convolution(const primitive_id& id, - const input_info& input, - const std::vector& weights, - ov::Strides stride = {1, 1}, - ov::CoordinateDiff pad = {0, 0}, - ov::Strides dilation = {1, 1}, - tensor output_size = {0, 0, 0, 0}, - int groups = 1, - float pad_value = 0.0f, - data_types calc_precision = data_types::f32, - const padding& output_padding = padding()) - : primitive_base(id, {input}, {output_padding}, {optional_data_type {calc_precision}}), - pad(pad), - stride(stride), - dilation(dilation), - output_size(output_size), - groups(groups), - pad_value(pad_value), - weights(weights) {} - - /// @brief Defines logical pad value added to input tensor - ov::CoordinateDiff pad; - /// @brief Defines shift in input buffer between adjacent calculations of output values. - ov::Strides stride; - /// @brief Defines gaps in the input - dilation rate k=1 is normal binary_convolution, k=2 means skipping one pixel per input, k=4 means skipping 3 pixels. - /// As an example in one dimension, a filter w of size 3 would compute over input x the following: w[0]*x[0] + w[1]*x[1] + w[2]*x[2] for dilation of 1. - /// For dilation 2 the filter would instead compute w[0]*x[0] + w[1]*x[2] + w[2]*x[4]. - ov::Strides dilation; - /// @brief User-defined output data size of the primitive (w/o padding). - tensor output_size; - /// @brief Number of feature groups (grouped convolution). If more than 1 then weights/bias count needs to be 1. - int groups = 1; - /// @brief Logical value of padding. Can be one of 3 values: 1 - pad bits equal to 1; -1 -> pad bits equal to 0; 0 -> pad is not counted - float pad_value = 0.0f; - /// @brief List of primitive ids containing weights data. - const primitive_id_arr weights; - - size_t hash() const override { - size_t seed = primitive::hash(); - seed = hash_range(seed, pad.begin(), pad.end()); - seed = hash_range(seed, stride.begin(), stride.end()); - seed = hash_range(seed, dilation.begin(), dilation.end()); - seed = hash_combine(seed, groups); - seed = hash_combine(seed, pad_value); - seed = hash_combine(seed, weights.size()); - return seed; - } - - bool operator==(const primitive& rhs) const override { - if (!compare_common_params(rhs)) - return false; - - auto rhs_casted = downcast(rhs); - - return pad == rhs_casted.pad && - stride == rhs_casted.stride && - dilation == rhs_casted.dilation && - groups == rhs_casted.groups && - pad_value == rhs_casted.pad_value && - weights.size() == rhs_casted.weights.size(); - } - - void save(BinaryOutputBuffer& ob) const override { - primitive_base::save(ob); - ob << pad; - ob << stride; - ob << dilation; - ob << output_size; - ob << groups; - ob << pad_value; - ob << weights; - } - - void load(BinaryInputBuffer& ib) override { - primitive_base::load(ib); - ib >> pad; - ib >> stride; - ib >> dilation; - ib >> output_size; - ib >> groups; - ib >> pad_value; - ib >> *const_cast(&weights); - } - - std::vector> get_dependencies() const override { - std::vector> ret; - ret.reserve(weights.size()); - for (auto& w : weights) ret.push_back(std::ref(w)); - return ret; - } -}; -} // namespace cldnn diff --git a/src/plugins/intel_gpu/include/intel_gpu/runtime/format.hpp b/src/plugins/intel_gpu/include/intel_gpu/runtime/format.hpp index ea3e2aec027..f155dcea481 100644 --- a/src/plugins/intel_gpu/include/intel_gpu/runtime/format.hpp +++ b/src/plugins/intel_gpu/include/intel_gpu/runtime/format.hpp @@ -123,7 +123,6 @@ struct format { bs_fs_fsv8_bsv16, ///< format used only for fully connected bs_f_bsv16, ///< format used only for fully connected weights fp16 batch=1 : bs - batch slice ///< (responses slice), bsv16 - 16 values of single batch slice, f - flattened plane of (fyx) - b_fs_yx_32fp, ///< format for data for binary convolutions winograd_2x3_s1_data, ///< format used for input for winograd convolution, F(2,3) -- filter 3x3 with stride 1 nv12, ///< format for media nv12 input image_2d_rgba, ///< format for image2d RGBA, always allocates memory for 4 feature maps (even when only 3 are used) @@ -219,7 +218,6 @@ struct format { os_is_yx_osv32_isv4_swizzled_by_2, ///< format for weights for IMAD convolutions os_is_yx_osv32_isv4, ///< format for weights for IMAD convolutions os_is_zyx_osv32_isv4, ///< format for weights for IMAD convolutions - os_is_yx_osv32_isv32p, ///< format for weights for binary convolutions lstm_weights_dio, ///< dynamic_lstm, direction, ///< than IO (I - input size, O - 4 * hidden_size) os_is_osv32_isv32_swizzled_by_4, ///< format for weights for 1x1 IMAD convolution diff --git a/src/plugins/intel_gpu/include/intel_gpu/runtime/layout.hpp b/src/plugins/intel_gpu/include/intel_gpu/runtime/layout.hpp index 679f4c51ea6..f3e2af08a33 100644 --- a/src/plugins/intel_gpu/include/intel_gpu/runtime/layout.hpp +++ b/src/plugins/intel_gpu/include/intel_gpu/runtime/layout.hpp @@ -49,12 +49,6 @@ struct data_type_traits { } static ov::element::Type max_type(ov::element::Type t1, ov::element::Type t2) { - if (t1 == ov::element::u1) - return t2; - - if (t2 == ov::element::u1) - return t1; - if (t1.bitwidth() < t2.bitwidth()) return t2; diff --git a/src/plugins/intel_gpu/src/graph/binary_convolution.cpp b/src/plugins/intel_gpu/src/graph/binary_convolution.cpp deleted file mode 100644 index ff8a34c3364..00000000000 --- a/src/plugins/intel_gpu/src/graph/binary_convolution.cpp +++ /dev/null @@ -1,119 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// -#include "binary_convolution_inst.h" -#include "convolution_inst.h" -#include "primitive_type_base.h" -#include "intel_gpu/runtime/error_handler.hpp" -#include "json_object.h" -#include - -namespace cldnn { -GPU_DEFINE_PRIMITIVE_TYPE_ID(binary_convolution) - -layout binary_convolution_inst::calc_output_layout(binary_convolution_node const& node, kernel_impl_params const& impl_param) { - auto desc = impl_param.typed_desc(); - - auto output_type = *desc->output_data_types[0]; - auto output_size = desc->output_size; - auto layout = cldnn::layout{output_type, format::bfyx, output_size}; - if (impl_param.has_fused_primitives()) { - layout = impl_param.get_fused_output_layout(); - } - - auto users = node.get_users(); - if (users.size() == 1 && users.front()->is_type()) { - auto conv_groups = (int32_t)users.front()->as().get_groups(); - - bool next_is_dw = conv_groups > 1 && conv_groups == output_size.feature[0]; - - if ((layout.data_type == data_types::f16 || layout.data_type == data_types::f32) && next_is_dw) { - layout.format = cldnn::format::b_fs_yx_fsv16; - } - } - - return layout; -} - -std::string binary_convolution_inst::to_string(binary_convolution_node const& node) { - auto desc = node.get_primitive(); - auto strd = desc->stride; - auto dilation = desc->dilation; - auto node_info = node.desc_to_json(); - - std::stringstream primitive_description; - json_composite conv_info; - conv_info.add("stride", cldnn::to_string(strd)); - conv_info.add("pad", cldnn::to_string(desc->pad)); - conv_info.add("dilation", cldnn::to_string(dilation)); - conv_info.add("out size", desc->output_size.to_string()); - - node_info->add("binary convolution info", conv_info); - node_info->dump(primitive_description); - - return primitive_description.str(); -} - -binary_convolution_inst::typed_primitive_inst(network& network, binary_convolution_node const& node) - : parent(network, node) { - auto stride = argument->stride; - auto pad = argument->pad; - - auto input_layout = node.input().get_output_layout(); - auto output_layout = node.get_output_layout(); - auto output_size = output_layout.get_tensor(); - - CLDNN_ERROR_NOT_EQUAL(node.id(), - "Input number of dimensions", - input_layout.get_rank(), - "output number of dimensions", - output_layout.get_rank(), - "Input/output rank mismatch"); - CLDNN_ERROR_NOT_EQUAL(node.id(), - "Stride number of dimensions", - stride.size(), - "output number of dimensions", - output_layout.get_spatial_rank(), - "stride/output dims mismatch"); - CLDNN_ERROR_NOT_EQUAL(node.id(), - "pad number of dimensions", - pad.size(), - "input number of dimensions", - input_layout.get_spatial_rank(), - "Input offset/ input size mismatch"); - - auto filter_inst = node.weights().get_output_layout(); // convolution filter - - - CLDNN_ERROR_NOT_EQUAL(node.id(), - "Weights number of dimensions", - filter_inst.get_rank(), - "output number of dimensions", - output_layout.get_rank(), - "Weights/output dims mismatch"); - CLDNN_ERROR_NOT_EQUAL(node.id(), - "Convolution padding mode", - node.get_output_layout().data_padding.filling_value(), - "padding value", - 0.0f, - "Unknown padding mode."); - CLDNN_ERROR_NOT_EQUAL(node.id(), - "Output feature size", - output_size.feature.size(), - "expected feature size", - 1, - "Only one-dimensional features are supported"); - CLDNN_ERROR_NOT_EQUAL(node.id(), - "Output batch size", - output_size.batch.size(), - "expected output size", - 1, - "Only one-dimensional batch size are supported"); - CLDNN_ERROR_NOT_EQUAL(node.id(), - "Weights feature maps number", - input_layout.feature(), - "input feature maps number", - filter_inst.feature(), - "Weights/ifm mismatch"); -} -} // namespace cldnn diff --git a/src/plugins/intel_gpu/src/graph/graph_optimizer/post_optimize_weights.cpp b/src/plugins/intel_gpu/src/graph/graph_optimizer/post_optimize_weights.cpp index 4f48148c395..3f568b96eba 100644 --- a/src/plugins/intel_gpu/src/graph/graph_optimizer/post_optimize_weights.cpp +++ b/src/plugins/intel_gpu/src/graph/graph_optimizer/post_optimize_weights.cpp @@ -7,7 +7,6 @@ #include "implementation_map.hpp" #include "convolution_inst.h" -#include "binary_convolution_inst.h" #include "deconvolution_inst.h" #include "deformable_convolution_inst.h" #include "fully_connected_inst.h" @@ -124,8 +123,6 @@ void post_optimize_weights::run(program& p) { for (auto& node : p.get_processing_order()) { if (node->is_type()) { optimize_weights(node->as(), p); - } else if (node->is_type()) { - optimize_weights(node->as(), p); } else if (node->is_type()) { optimize_weights(node->as(), p); } else if (node->is_type()) { diff --git a/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_padding.cpp b/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_padding.cpp index 499c8eee49f..87193ff304a 100644 --- a/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_padding.cpp +++ b/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_padding.cpp @@ -134,12 +134,6 @@ void prepare_padding::run(program& p) { else needed_padding = prim_node.input().get_output_layout().data_padding; - add_required_padding(prim_node, needed_padding); - } else if (node->is_type()) { - auto& prim_node = node->as(); - - auto needed_padding = prim_node.input().get_output_layout().data_padding; - add_required_padding(prim_node, needed_padding); } } @@ -168,8 +162,7 @@ void prepare_padding::run(program& p) { conv_layout.format != cldnn::format::b_fs_zyx_fsv16 && conv_layout.format != cldnn::format::bs_fs_yx_bsv16_fsv16 && conv_layout.format != cldnn::format::b_fs_yx_fsv4 && - conv_layout.format != cldnn::format::fs_b_yx_fsv32 && - conv_layout.format != cldnn::format::b_fs_yx_32fp) { + conv_layout.format != cldnn::format::fs_b_yx_fsv32) { continue; } @@ -257,77 +250,4 @@ void prepare_padding::run(program& p) { needed_padding = padding::max(prev_prim_output_layout.data_padding, needed_padding); p.apply_needed_padding(node, conv_input_node, needed_padding); } - - for (auto& pair : p.nodes_map) { - if (pair.second->type() != binary_convolution::type_id()) - continue; - - auto& node = pair.second->as(); - if (node.get_dependencies().empty()) - continue; - - if (node.is_dynamic()) continue; - auto conv = node.get_primitive(); - auto& conv_input_node = node.get_dependency(0); - auto conv_layout = node.get_output_layout(); - - // right now output padding optimization is only available for bfyx format and data type = float32 - if (conv_layout.format != cldnn::format::bfyx && conv_layout.format != cldnn::format::b_fs_yx_32fp) - continue; - - // We shoudn't apply any padding to nodes which are marked as outputs or have type as data - if (conv_input_node.is_output() || conv_input_node.is_type()) - continue; - - // Calculating input padding needed for convolution - auto& filter_node = node.as().weights(); - auto filter_prim = filter_node.get_primitive(); - - layout filter_layout = filter_node.get_output_layout(); - - // convolution have only one input primitive - auto prev_prim_output_layout = conv_input_node.get_output_layout(); - - // Compute initial required paddings for primitive used as input for convolution. - auto pad = conv->pad; - auto stride = conv->stride; - auto dilation = conv->dilation; - - auto stride_z = stride.size() >= 3 ? stride[stride.size() - 3] : 1; - auto stride_y = stride.size() >= 2 ? stride[stride.size() - 2] : 1; - auto stride_x = stride.size() >= 1 ? stride[stride.size() - 1] : 1; - - auto dilation_z = dilation.size() >= 3 ? dilation[dilation.size() - 3] : 1; - auto dilation_y = dilation.size() >= 2 ? dilation[dilation.size() - 2] : 1; - auto dilation_x = dilation.size() >= 1 ? dilation[dilation.size() - 1] : 1; - - auto pad_z = pad.size() >= 3 ? pad[pad.size() - 3] : 0; - auto pad_y = pad.size() >= 2 ? pad[pad.size() - 2] : 0; - auto pad_x = pad.size() >= 1 ? pad[pad.size() - 1] : 0; - - auto input_limit_x = -pad_x + (conv_layout.spatial(0) - 1) * stride_x + - (filter_layout.spatial(0) - 1) * dilation_x + 1; - auto input_limit_y = -pad_y + (conv_layout.spatial(1) - 1) * stride_y + - (filter_layout.spatial(1) - 1) * dilation_y + 1; - auto input_limit_z = -pad_z + (conv_layout.spatial(2) - 1) * stride_z + - (filter_layout.spatial(2) - 1) * dilation_z + 1; - - auto padding_begin_x = std::max(pad_x, 0); - auto padding_begin_y = std::max(pad_y, 0); - auto padding_begin_z = std::max(pad_z, 0); - auto padding_end_x = std::max( - static_cast(input_limit_x) - prev_prim_output_layout.spatial(0), - 0); - auto padding_end_y = std::max( - static_cast(input_limit_y) - prev_prim_output_layout.spatial(1), - 0); - auto padding_end_z = std::max( - static_cast(input_limit_z) - prev_prim_output_layout.spatial(2), - 0); - - cldnn::padding needed_padding({0, 0, padding_begin_x, padding_begin_y, padding_begin_z}, {0, 0, padding_end_x, padding_end_y, padding_end_z}, 0); - needed_padding = padding::max(prev_prim_output_layout.data_padding, needed_padding); - - p.apply_needed_padding(node, conv_input_node, needed_padding); - } } diff --git a/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_primitive_fusing.cpp b/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_primitive_fusing.cpp index 167b77c74af..fb61e577a36 100644 --- a/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_primitive_fusing.cpp +++ b/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_primitive_fusing.cpp @@ -9,7 +9,6 @@ #include "proposal_inst.h" #include "roi_pooling_inst.h" #include "quantize_inst.h" -#include "binary_convolution_inst.h" #include "activation_inst.h" #include "batch_to_space_inst.h" #include "crop_inst.h" @@ -545,25 +544,6 @@ void prepare_primitive_fusing::fuse_simple_primitives(program &p) { return data_type_traits::is_i8_u8(in_dt); }; - auto bin_conv_supports_eltw_fusings = [](binary_convolution_node& conv_node) -> bool { - auto& eltw_node = static_cast(*conv_node.get_users().front()); - auto& eltw_prim = *eltw_node.get_primitive(); - - if (eltw_node.get_dependencies().size() < 2) - return false; - - auto const_layout = eltw_node.get_input_layout(1); - auto conv_layout = conv_node.get_output_layout(); - auto per_channel_eltwise = const_layout.feature() == conv_layout.feature(); - - if (eltw_node.get_dependency(1).is_constant() && per_channel_eltwise && - (eltw_prim.mode == eltwise_mode::sum || eltw_prim.mode == eltwise_mode::prod) && - all_ones(conv_node.get_primitive()->dilation)) - return true; - - return false; - }; - auto fc_supports_fusings = [&](fully_connected_node& node) -> bool { if (_lo.get_optimization_attributes().use_onednn_impls && _lo.get_preferred_impl_type(node, format::any /*dummy*/) == impl_types::onednn) { @@ -734,9 +714,7 @@ void prepare_primitive_fusing::fuse_simple_primitives(program &p) { } } - bool should_fuse = input.is_type(); - - should_fuse |= input.is_type() && conv_supports_fusings(input.as()); + bool should_fuse = input.is_type() && conv_supports_fusings(input.as()); should_fuse |= input.is_type() && fc_supports_fusings(input.as()); @@ -849,18 +827,7 @@ void prepare_primitive_fusing::fuse_simple_primitives(program &p) { quantize_node.get_per_tensor_output_shift() && quantize_node.get_per_tensor_output_range(); - auto& input_lo = quantize_node.get_dependency(1); - auto& input_hi = quantize_node.get_dependency(2); - bool should_fuse = input_data.is_type() && - ((out_dt == data_types::u1 && - quantize_node.get_dependencies().size() == 5 && - ((in_layout.feature() == input_lo.get_output_layout().feature() && - in_layout.feature() == input_hi.get_output_layout().feature()) || - (input_lo.get_output_layout().feature() == 1 && - input_hi.get_output_layout().feature() == 1)))) && - all_ones(input_data.as().get_primitive()->dilation); - - should_fuse |= input_data.is_type() && conv_supports_fusings(input_data.as()) && + bool should_fuse = input_data.is_type() && conv_supports_fusings(input_data.as()) && quantize_node.get_scale_shift_opt() && ((out_dt == data_types::f32 || out_dt == data_types::f16) || in_layout.format == format::b_fs_yx_fsv16 || @@ -954,8 +921,6 @@ void prepare_primitive_fusing::fuse_simple_primitives(program &p) { for (size_t i = 0; i < parents.size(); i++) { can_fuse_parents[i] = (parents[i].first->is_type() && conv_supports_fusings(parents[i].first->as())) || - (parents[i].first->is_type() && - bin_conv_supports_eltw_fusings(parents[i].first->as())) || (parents[i].first->is_type() && mvn_supports_fusings(parents[i].first->as(), true)) || (parents[i].first->is_type()) || @@ -1197,7 +1162,6 @@ void prepare_primitive_fusing::fuse_constant_transposes(program& p) { if (next_node->is_type() || next_node->is_type() || next_node->is_type() || - next_node->is_type() || next_node->is_type()) { size_t weights_offset = next_node->get_primitive()->input_size(); std::vector valid_weights_indices = {next_node->get_primitive()->input_size()}; diff --git a/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_quantization.cpp b/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_quantization.cpp index f55d99b6a5f..13e21ab4c21 100644 --- a/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_quantization.cpp +++ b/src/plugins/intel_gpu/src/graph/graph_optimizer/prepare_quantization.cpp @@ -6,13 +6,11 @@ #include "pooling_inst.h" #include "quantize_inst.h" #include "reorder_inst.h" -#include "binary_convolution_inst.h" #include "eltwise_inst.h" #include "data_inst.h" #include "pass_manager.h" #include "program_helpers.h" #include "to_string_utils.h" -#include "intel_gpu/runtime/error_handler.hpp" #include #include @@ -23,24 +21,6 @@ using namespace cldnn; namespace { -template -bool check_binarization(memory::ptr mem_input_low, memory::ptr mem_input_high, program& p) { - bool is_binarization = true; - const auto& stream = p.get_stream(); - mem_lock data_input_low_lock{mem_input_low, stream}; - mem_lock data_input_high_lock{mem_input_high, stream}; - auto data_input_low = data_input_low_lock.data(); - auto data_input_high = data_input_high_lock.data(); - const size_t number_mem_layout_elements = mem_input_high->get_layout().count(); - for (size_t i = 0; i < number_mem_layout_elements; i++) { - if (data_input_high[i] != data_input_low[i]) { - is_binarization = false; - break; - } - } - return is_binarization; -} - inline float clamp(float val) { return std::max(std::numeric_limits::lowest(), std::min(std::numeric_limits::max(), val)); } @@ -322,50 +302,12 @@ void prepare_quantization::handle_quantize_node(program& p, quantize_node& quant if (optimize_quantize(p, quantize_node)) return; - if (quantize_node.get_primitive()->levels == 2) { - prepare_packed_quantize(p, quantize_node); - } else if (quantize_node.get_primitive()->levels <= 256 && !quantize_node.get_scale_shift_opt() && !quantize_node.is_constant()) { + auto l = quantize_node.get_primitive()->levels; + if (l > 2 && l <= 256 && !quantize_node.get_scale_shift_opt() && !quantize_node.is_constant()) { prepare_scale_shift_opt(p, quantize_node); } } -void prepare_quantization::prepare_packed_quantize(program& p, quantize_node& quantize_node) { - program_node &input_low_node = quantize_node.get_dependency(1); - program_node &input_high_node = quantize_node.get_dependency(2); - - if (quantize_node.is_output() || !input_low_node.is_type() || !input_high_node.is_type()) { - return; - } - - auto &input_low = input_low_node.as(); - auto &input_high = input_high_node.as(); - - auto mem_input_low = input_low.get_attached_memory_ptr(); - auto mem_input_high = input_high.get_attached_memory_ptr(); - - bool is_binarization = true; - switch (mem_input_high->get_layout().data_type) { - case data_types::f32: { - is_binarization = check_binarization(mem_input_low, mem_input_high, p); - break; - } - case data_types::f16: { - is_binarization = check_binarization(mem_input_low, mem_input_high, p); - break; - } - default: - CLDNN_ERROR_MESSAGE(quantize_node.id(), "prepare_quantization: Unsupported precision of quantize inputs"); - } - - auto output_dt = quantize_node.get_output_layout().data_type; - if (is_binarization) { - output_dt = data_types::u1; - } - - quantize_node.typed_desc()->output_data_types = {optional_data_type{output_dt}}; - quantize_node.recalc_output_layout(); -} - void prepare_quantization::prepare_dequantize_merge(program& p, eltwise_node& eltwise_node) { for (size_t i = 1; i < eltwise_node.get_dependencies().size(); i++) { if (!eltwise_node.get_dependency(i).is_type()) { diff --git a/src/plugins/intel_gpu/src/graph/graph_optimizer/remove_redundant_reorders.cpp b/src/plugins/intel_gpu/src/graph/graph_optimizer/remove_redundant_reorders.cpp index 362cf5bf63d..efdaee8a3ab 100644 --- a/src/plugins/intel_gpu/src/graph/graph_optimizer/remove_redundant_reorders.cpp +++ b/src/plugins/intel_gpu/src/graph/graph_optimizer/remove_redundant_reorders.cpp @@ -7,7 +7,6 @@ #include "pass_manager.h" #include "program_helpers.h" -#include "binary_convolution_inst.h" #include "reshape_inst.h" #include "convert_color_inst.h" #include "one_hot_inst.h" diff --git a/src/plugins/intel_gpu/src/graph/graph_optimizer/reorder_inputs.cpp b/src/plugins/intel_gpu/src/graph/graph_optimizer/reorder_inputs.cpp index 769134e440b..24f414af467 100644 --- a/src/plugins/intel_gpu/src/graph/graph_optimizer/reorder_inputs.cpp +++ b/src/plugins/intel_gpu/src/graph/graph_optimizer/reorder_inputs.cpp @@ -9,7 +9,6 @@ #include "intel_gpu/runtime/debug_configuration.hpp" #include "intel_gpu/runtime/utils.hpp" #include "program_helpers.h" -#include "binary_convolution_inst.h" #include "mvn_inst.h" #include "to_string_utils.h" #include "pooling_inst.h" @@ -742,30 +741,6 @@ void reorder_inputs::run(program& p, layout_optimizer& lo, reorder_factory& rf) } }; - const auto reorder_input_and_weights_binary_convolution = [&p, &rf](typed_program_node& binary_conv_node) { - auto& input = binary_conv_node.input(); - auto input_layout = input.get_output_layout(); - auto new_layout = input_layout; - new_layout.data_type = data_types::u1; - - auto reorder = rf.get_reorder(input.id(), input_layout, new_layout); - - if (reorder.first) { - p.add_intermediate(reorder.first, binary_conv_node, 0, !reorder.second); - } - - auto& weights = binary_conv_node.weights(); - auto weights_layout = weights.get_output_layout(); - if (!weights.is_type() && !weights.is_constant()) { - auto new_layout = layout{ weights_layout.get_partial_shape(), data_types::u1, format::b_fs_yx_32fp }; - auto reorder = rf.get_reorder(weights.id(), weights_layout, new_layout); - if (reorder.first) { - p.add_intermediate(reorder.first, binary_conv_node, 1, !reorder.second); - p.get_or_create(reorder.first).recalc_output_layouts(false); - } - } - }; - const auto reorder_input_and_weights_deconvolution = [&p, &lo, &rf](typed_program_node& deconv_node) { auto& input = deconv_node.input(); auto input_layout = input.get_output_layout(); @@ -928,10 +903,9 @@ void reorder_inputs::run(program& p, layout_optimizer& lo, reorder_factory& rf) }; for (auto& prim : p.get_processing_order()) { - program_helpers::do_for_types( + program_helpers::do_for_types( *prim, reorder_input_detection_output, - reorder_input_and_weights_binary_convolution, reorder_input_and_weights_deconvolution, reorder_convolution, reorder_input_fully_connected, diff --git a/src/plugins/intel_gpu/src/graph/impls/ocl/binary_convolution.cpp b/src/plugins/intel_gpu/src/graph/impls/ocl/binary_convolution.cpp deleted file mode 100644 index 41ad3b6d92d..00000000000 --- a/src/plugins/intel_gpu/src/graph/impls/ocl/binary_convolution.cpp +++ /dev/null @@ -1,89 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "primitive_base.hpp" - -#include "binary_convolution_inst.h" -#include "kernel_selector/kernels/binary_convolution/binary_convolution_kernel_selector.h" -#include "kernel_selector/kernels/binary_convolution/binary_convolution_params.h" - -namespace cldnn { -namespace ocl { - -struct binary_convolution_impl : typed_primitive_impl_ocl { - using parent = typed_primitive_impl_ocl; - using parent::parent; - using kernel_selector_t = kernel_selector::binary_convolution_kernel_selector; - using kernel_params_t = std::pair; - - DECLARE_OBJECT_TYPE_SERIALIZATION(cldnn::ocl::binary_convolution_impl) - - std::unique_ptr clone() const override { - return make_unique(*this); - } - -protected: - kernel_arguments_data get_arguments(const typed_primitive_inst& instance) const override { - kernel_arguments_data args = parent::get_arguments(instance); - - args.weights = instance.weights_memory(); - return args; - } - -public: - static kernel_params_t get_kernel_params(const kernel_impl_params& impl_param) { - const auto& primitive = impl_param.typed_desc(); - const auto& weights_layout = (*impl_param.weights_layout).convert_to_weights_layout(false); - const auto& weights_size = weights_layout.get_tensor(); - - const auto& groups = primitive->groups; - const auto& stride = primitive->stride; - const auto& dilation = primitive->dilation; - const auto& pad = primitive->pad; - - auto params = get_weights_bias_default_params(impl_param); - auto optional_params = get_default_weights_bias_optional_params(impl_param.get_program()); - - params.pad_value = primitive->pad_value; - params.out_dt = to_data_type(*primitive->output_data_types[0]); - params.groups = static_cast(groups); - params.filterSize = { - (uint32_t)weights_size.spatial[0], - (uint32_t)weights_size.spatial[1], - (uint32_t)weights_size.spatial[2], - }; - - uint32_t pad_z = std::max(pad.size() >= 3 ? pad[pad.size() - 3] : 0, 0); - uint32_t pad_y = std::max(pad.size() >= 2 ? pad[pad.size() - 2] : 0, 0); - uint32_t pad_x = std::max(pad.size() >= 1 ? pad[pad.size() - 1] : 0, 0); - params.padding = {pad_x, pad_y, pad_z}; - - uint32_t stride_z = stride.size() >= 3 ? static_cast(stride[stride.size() - 3]) : 1; - uint32_t stride_y = stride.size() >= 2 ? static_cast(stride[stride.size() - 2]) : 1; - uint32_t stride_x = stride.size() >= 1 ? static_cast(stride[stride.size() - 1]) : 1; - params.stride = {stride_x, stride_y, stride_z}; - - uint32_t dilation_z = dilation.size() >= 3 ? static_cast(dilation[dilation.size() - 3]) : 1; - uint32_t dilation_y = dilation.size() >= 2 ? static_cast(dilation[dilation.size() - 2]) : 1; - uint32_t dilation_x = dilation.size() >= 1 ? static_cast(dilation[dilation.size() - 1]) : 1; - params.dilation = {dilation_x, dilation_y, dilation_z}; - - return {params, optional_params}; - } -}; - -namespace detail { - -attach_binary_convolution_impl::attach_binary_convolution_impl() { - implementation_map::add(impl_types::ocl, typed_primitive_impl_ocl::create, { - std::make_tuple(data_types::u1, format::b_fs_yx_32fp), - }); -} - -} // namespace detail -} // namespace ocl -} // namespace cldnn - -BIND_BINARY_BUFFER_WITH_TYPE(cldnn::ocl::binary_convolution_impl) -BIND_BINARY_BUFFER_WITH_TYPE(cldnn::binary_convolution) diff --git a/src/plugins/intel_gpu/src/graph/impls/ocl/kernel_selector_helper.cpp b/src/plugins/intel_gpu/src/graph/impls/ocl/kernel_selector_helper.cpp index e9b98a57e00..0ec066f0099 100644 --- a/src/plugins/intel_gpu/src/graph/impls/ocl/kernel_selector_helper.cpp +++ b/src/plugins/intel_gpu/src/graph/impls/ocl/kernel_selector_helper.cpp @@ -121,8 +121,6 @@ namespace cldnn { kernel_selector::data_type to_data_type(data_types dt) { switch (dt) { - case cldnn::data_types::u1: - return kernel_selector::data_type::BINARY; case cldnn::data_types::i4: return kernel_selector::data_type::INT4; case cldnn::data_types::u4: @@ -146,8 +144,6 @@ kernel_selector::data_type to_data_type(data_types dt) { data_types from_data_type(kernel_selector::data_type dt) { switch (dt) { - case kernel_selector::data_type::BINARY: - return cldnn::data_types::u1; case kernel_selector::data_type::INT4: return cldnn::data_types::i4; case kernel_selector::data_type::UINT4: @@ -171,8 +167,6 @@ data_types from_data_type(kernel_selector::data_type dt) { kernel_selector::weights_type to_weights_type(data_types dt) { switch (dt) { - case cldnn::data_types::u1: - return kernel_selector::weights_type::BINARY; case cldnn::data_types::u4: return kernel_selector::weights_type::UINT4; case cldnn::data_types::i4: @@ -194,8 +188,6 @@ kernel_selector::weights_type to_weights_type(data_types dt) { data_types from_weights_type(kernel_selector::weights_type dt) { switch (dt) { - case kernel_selector::weights_type::BINARY: - return data_types::u1; case kernel_selector::weights_type::INT4: return data_types::i4; case kernel_selector::weights_type::UINT4: @@ -255,8 +247,6 @@ kernel_selector::data_layout to_data_layout(format f) { return kernel_selector::data_layout::bs_f_bsv8__af8; case format::winograd_2x3_s1_data: return kernel_selector::data_layout::winograd_2x3_s1_data; - case format::b_fs_yx_32fp: - return kernel_selector::data_layout::b_fs_yx_32fp; case format::bfzyx: return kernel_selector::data_layout::bfzyx; case format::bzyxf: @@ -360,8 +350,6 @@ cldnn::format from_data_layout(kernel_selector::data_layout l) { return cldnn::format::bs_f_bsv16; case kernel_selector::data_layout::winograd_2x3_s1_data: return cldnn::format::winograd_2x3_s1_data; - case kernel_selector::data_layout::b_fs_yx_32fp: - return cldnn::format::b_fs_yx_32fp; case kernel_selector::data_layout::bfzyx: return cldnn::format::bfzyx; case kernel_selector::data_layout::fs_b_yx_fsv32: @@ -538,9 +526,6 @@ kernel_selector::weights_layout to_weights_layout(format f, bool is_grouped) { return kernel_selector::weights_layout::os_is_yx_osv32_isv4; case format::os_is_zyx_osv32_isv4: return kernel_selector::weights_layout::os_is_zyx_osv32_isv4; - case format::b_fs_yx_32fp: - case format::os_is_yx_osv32_isv32p: - return kernel_selector::weights_layout::os_is_yx_osv32_isv32p; case format::os_is_yx_isv16_osv16: return kernel_selector::weights_layout::os_is_yx_isv16_osv16; case format::os_is_y_x8_osv8_isv4_swizzled_by_4: @@ -865,8 +850,6 @@ cldnn::format::type from_weights_layout(kernel_selector::weights_layout l) { return format::os_is_zyx_osv32_isv4; case kernel_selector::weights_layout::os_is_y_x8_osv8_isv4_swizzled_by_4: return cldnn::format::os_is_y_x8_osv8_isv4_swizzled_by_4; - case kernel_selector::weights_layout::os_is_yx_osv32_isv32p: - return cldnn::format::os_is_yx_osv32_isv32p; case kernel_selector::weights_layout::oizyx: return cldnn::format::oizyx; case kernel_selector::weights_layout::iozyx: diff --git a/src/plugins/intel_gpu/src/graph/impls/ocl/quantize.cpp b/src/plugins/intel_gpu/src/graph/impls/ocl/quantize.cpp index ef3ea9b4316..2164c922a2a 100644 --- a/src/plugins/intel_gpu/src/graph/impls/ocl/quantize.cpp +++ b/src/plugins/intel_gpu/src/graph/impls/ocl/quantize.cpp @@ -51,7 +51,6 @@ public: get_default_optional_params(impl_param.get_program()); quantize_params.levels = arg.get_levels(); - quantize_params.packed_binary_output = arg.get_packed_binary_output(); quantize_params.scale_shift_opt = arg.get_scale_shift_opt(); quantize_params.has_post_scale = arg.get_need_post_scale(); quantize_params.has_post_shift = arg.get_need_post_shift(); @@ -90,8 +89,6 @@ public: void update_dispatch_data(const kernel_impl_params& impl_param) override { auto quantize_params = get_default_params(impl_param); - const auto& output_layout = impl_param.get_output_layout(); - quantize_params.packed_binary_output = output_layout.data_type == data_types::u1; (_kernel_data.update_dispatch_data_func)(quantize_params, _kernel_data); } }; diff --git a/src/plugins/intel_gpu/src/graph/impls/ocl/register.cpp b/src/plugins/intel_gpu/src/graph/impls/ocl/register.cpp index 3cc96ee0073..b03944ae774 100644 --- a/src/plugins/intel_gpu/src/graph/impls/ocl/register.cpp +++ b/src/plugins/intel_gpu/src/graph/impls/ocl/register.cpp @@ -12,7 +12,6 @@ void register_implementations() { REGISTER_OCL(activation); REGISTER_OCL(adaptive_pooling); REGISTER_OCL(arg_max_min); - REGISTER_OCL(binary_convolution); REGISTER_OCL(border); REGISTER_OCL(broadcast); REGISTER_OCL(bucketize); diff --git a/src/plugins/intel_gpu/src/graph/impls/ocl/register.hpp b/src/plugins/intel_gpu/src/graph/impls/ocl/register.hpp index d591499148e..17731a3f634 100644 --- a/src/plugins/intel_gpu/src/graph/impls/ocl/register.hpp +++ b/src/plugins/intel_gpu/src/graph/impls/ocl/register.hpp @@ -7,7 +7,6 @@ #include "intel_gpu/primitives/activation.hpp" #include "intel_gpu/primitives/arg_max_min.hpp" #include "intel_gpu/primitives/batch_to_space.hpp" -#include "intel_gpu/primitives/binary_convolution.hpp" #include "intel_gpu/primitives/border.hpp" #include "intel_gpu/primitives/broadcast.hpp" #include "intel_gpu/primitives/bucketize.hpp" @@ -93,7 +92,6 @@ REGISTER_OCL(activation); REGISTER_OCL(adaptive_pooling); REGISTER_OCL(arg_max_min); REGISTER_OCL(batch_to_space); -REGISTER_OCL(binary_convolution); REGISTER_OCL(border); REGISTER_OCL(broadcast); REGISTER_OCL(bucketize); diff --git a/src/plugins/intel_gpu/src/graph/include/binary_convolution_inst.h b/src/plugins/intel_gpu/src/graph/include/binary_convolution_inst.h deleted file mode 100644 index 1a910bcf81e..00000000000 --- a/src/plugins/intel_gpu/src/graph/include/binary_convolution_inst.h +++ /dev/null @@ -1,67 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once -#include "intel_gpu/primitives/binary_convolution.hpp" -#include "primitive_inst.h" - -#include -#include -#include - -namespace cldnn { - -template <> -struct typed_program_node : public typed_program_node_base { - using parent = typed_program_node_base; - -public: - typed_program_node(std::shared_ptr prim, program& prog) - : parent(prim, prog) {} - - program_node& input() const { return get_dependency(0); } - program_node& weights() const { return get_dependency(1); } - - std::unique_ptr get_kernel_impl_params(const std::vector& in_layouts, const std::vector& out_layouts) const override { - auto params = parent::get_kernel_impl_params(in_layouts, out_layouts); - params->weights_layout = optional_layout(weights().get_output_layout()); - return params; - } -}; - -using binary_convolution_node = typed_program_node; - -template <> -class typed_primitive_inst : public typed_primitive_inst_base { - using parent = typed_primitive_inst_base; - using parent::parent; - -public: - static layout calc_output_layout(binary_convolution_node const& node, kernel_impl_params const& impl_param); - static std::string to_string(binary_convolution_node const& node); - typed_primitive_inst(network& network, binary_convolution_node const& node); - - bool need_reset_input_memory(size_t idx = 0) const override { - if (idx != 0) - return false; - - auto input_layout = _deps[0].first->_impl_params->get_output_layout(0); - return input_layout.data_padding ? true : false; - } - - bool need_reset_output_memory() const override { - bool res = parent::need_reset_output_memory(); - auto output_layout = _impl_params->get_output_layout(0); - if (output_layout.data_padding) { - return true; - } - return res; - } - - memory::ptr weights_memory() const { return dep_memory_ptr(1); } -}; - -using binary_convolution_inst = typed_primitive_inst; - -} // namespace cldnn diff --git a/src/plugins/intel_gpu/src/graph/include/layout_optimizer.h b/src/plugins/intel_gpu/src/graph/include/layout_optimizer.h index e9ed4ec959a..aee3bff0a6d 100644 --- a/src/plugins/intel_gpu/src/graph/include/layout_optimizer.h +++ b/src/plugins/intel_gpu/src/graph/include/layout_optimizer.h @@ -14,7 +14,6 @@ #include "convolution_inst.h" #include "deconvolution_inst.h" #include "detection_output_inst.h" -#include "binary_convolution_inst.h" #include "quantize_inst.h" #include diff --git a/src/plugins/intel_gpu/src/graph/include/pass_manager.h b/src/plugins/intel_gpu/src/graph/include/pass_manager.h index d6409eab31e..8973cd74aa4 100644 --- a/src/plugins/intel_gpu/src/graph/include/pass_manager.h +++ b/src/plugins/intel_gpu/src/graph/include/pass_manager.h @@ -149,7 +149,6 @@ public: private: void run(program& p) override; void handle_quantize_node(program& p, quantize_node& quantize_node); - void prepare_packed_quantize(program& p, quantize_node& quantize_node); void prepare_dequantize_merge(program& p, eltwise_node& eltwise_node); void remove_fake_reorders(program& p, reorder_node& reorder_node); void prepare_asymmetric_quantization(program& p, convolution_node& convolution_node); diff --git a/src/plugins/intel_gpu/src/graph/include/quantize_inst.h b/src/plugins/intel_gpu/src/graph/include/quantize_inst.h index 8d213f68bc3..fa616980fe0 100644 --- a/src/plugins/intel_gpu/src/graph/include/quantize_inst.h +++ b/src/plugins/intel_gpu/src/graph/include/quantize_inst.h @@ -142,7 +142,6 @@ public: program_node& input(size_t index = 0) const { return get_dependency(index); } int get_levels() const { return get_primitive()->levels; } - bool get_packed_binary_output() const { return get_output_layout().data_type == data_types::u1; } bool get_scale_shift_opt() const { return get_primitive()->scale_shift_opt; } bool get_need_pre_shift() const { return get_primitive()->need_pre_shift; } bool get_need_post_scale() const { return get_primitive()->need_post_scale; } @@ -201,7 +200,9 @@ class typed_primitive_inst : public typed_primitive_inst_base - static std::vector calc_output_layouts(quantize_node const& node, kernel_impl_params const& impl_param); + static std::vector calc_output_layouts(quantize_node const& node, kernel_impl_params const& impl_param) { + return forward_input0_shape(impl_param); + } static layout calc_output_layout(quantize_node const& node, kernel_impl_params const& impl_param); static std::string to_string(quantize_node const& node); diff --git a/src/plugins/intel_gpu/src/graph/layout_optimizer.cpp b/src/plugins/intel_gpu/src/graph/layout_optimizer.cpp index ca4569a7df7..eb02ba9779c 100644 --- a/src/plugins/intel_gpu/src/graph/layout_optimizer.cpp +++ b/src/plugins/intel_gpu/src/graph/layout_optimizer.cpp @@ -450,9 +450,6 @@ bool layout_optimizer::can_fuse_reorder_to_prev(program_node& prev, reorder_node || fmt_next == format::bs_fs_yx_bsv32_fsv16 || fmt_next == format::bs_fs_yx_bsv32_fsv32)) return true; - if (prev.is_type() && fmt_next == format::b_fs_yx_fsv16) - return true; - if (prev.is_type() && !data_type_traits::is_floating_point(dt_prev) && data_type_traits::is_floating_point(dt_next) && @@ -1741,8 +1738,6 @@ format layout_optimizer::get_preferred_format(program_node& node) { expected = _forcing_map.at(node.id()).first; } else if (node.is_type()) { expected = get_expected_format(node.as()); - } else if (node.is_type()) { - expected = cldnn::format::b_fs_yx_32fp; } else if (node.is_type()) { expected = get_expected_format(node.as()); } else if (node.is_type() || node.is_type()) { diff --git a/src/plugins/intel_gpu/src/graph/network.cpp b/src/plugins/intel_gpu/src/graph/network.cpp index c8a081dadbc..2aaca0c73b2 100644 --- a/src/plugins/intel_gpu/src/graph/network.cpp +++ b/src/plugins/intel_gpu/src/graph/network.cpp @@ -212,41 +212,6 @@ void dump(memory::ptr mem, stream& stream, std::ofstream& file_stream, bool dump file_stream << buffer.str(); } -template <> -void dump(memory::ptr mem, stream& stream, std::ofstream& file_stream, bool dump_raw) { - auto&& l = mem->get_layout(); - - file_stream << "shape: "; - file_stream << l.batch() << " "; - file_stream << l.feature() << " "; - file_stream << l.spatial(1) << " "; - file_stream << l.spatial(0) << " "; - file_stream << "(" << l.batch() * l.feature() * l.spatial(1) * l.spatial(0) << ")" << std::endl; - - mem_lock lock(mem, stream); - auto mem_ptr = lock.data(); - - if (!dump_raw) { - for (cldnn::tensor::value_type b = 0; b < l.batch(); ++b) { - for (cldnn::tensor::value_type f = 0; f < (cldnn::tensor::value_type)ceil_div(l.feature(), 32); ++f) { - for (cldnn::tensor::value_type z = 0; z < l.spatial(2); ++z) { - for (cldnn::tensor::value_type y = 0; y < l.spatial(1); ++y) { - for (cldnn::tensor::value_type x = 0; x < l.spatial(0); ++x) { - cldnn::tensor t(cldnn::batch(b), cldnn::feature(f), cldnn::spatial(x, y, z, 0)); - size_t input_it = mem->get_layout().get_linear_offset(t); - file_stream << mem_ptr[input_it] << std::endl; - } - } - } - } - } - } else { - for (size_t i = 0; i < lock.size(); ++i) { - file_stream << std::fixed << std::setprecision(6) << mem_ptr[i] << std::endl; - } - } -} - void log_memory_to_file(memory::ptr mem, layout data_layout, stream& stream, std::string layerName, bool dump_raw) { std::cout << "Dump " << (dump_raw ? "raw " : "") << layerName << std::endl; GPU_DEBUG_GET_INSTANCE(debug_config); @@ -266,8 +231,6 @@ void log_memory_to_file(memory::ptr mem, layout data_layout, stream& stream, std dump(actual_mem, stream, file_stream, dump_raw); else if (mem_dt == cldnn::data_types::f16) dump(actual_mem, stream, file_stream, dump_raw); - else if (mem_dt == cldnn::data_types::u1) - dump(actual_mem, stream, file_stream, dump_raw); else if (mem_dt == cldnn::data_types::i64) dump(actual_mem, stream, file_stream, dump_raw); else if (mem_dt == cldnn::data_types::i32) diff --git a/src/plugins/intel_gpu/src/graph/program.cpp b/src/plugins/intel_gpu/src/graph/program.cpp index 9bd5d57090c..7035005be33 100644 --- a/src/plugins/intel_gpu/src/graph/program.cpp +++ b/src/plugins/intel_gpu/src/graph/program.cpp @@ -28,7 +28,6 @@ #include "softmax_inst.h" #include "permute_inst.h" #include "custom_gpu_primitive_inst.h" -#include "binary_convolution_inst.h" #include "resample_inst.h" #include "reshape_inst.h" #include "quantize_inst.h" @@ -299,28 +298,7 @@ bool program::analyze_output_size_handling_need() { // Calculate output size and compare with specified. for (const auto& node : processing_order) { - if (node->is_type()) { - auto& prim_node = node->as(); - const auto& prim = prim_node.get_primitive(); - - tensor specified_output_range( - {0, 0, prim->output_size.spatial[0], prim->output_size.spatial[1], prim->output_size.spatial[2]}, - 1); - - auto filter_size = prim_node.weights().get_output_layout().get_tensor(); - - auto primInputSize = prim_node.input().get_output_layout().get_tensor(); - auto calc_output_range = - calc_sliding_window_output_range(primInputSize, - filter_size, - prim->pad, - prim->stride, - prim->dilation, - true, - 1); - if (specified_output_range != calc_output_range) - handling_needed = true; - } else if (node->is_type()) { + if (node->is_type()) { auto& prim_node = node->as(); const auto& prim = prim_node.get_primitive(); @@ -1439,7 +1417,6 @@ void program::set_layout_optimizer_attributes(layout_optimizer& lo) { prim.type() != cldnn::permute::type_id() && prim.type() != cldnn::reshape::type_id() && prim.type() != cldnn::detection_output::type_id() && - prim.type() != cldnn::binary_convolution::type_id() && prim.type() != cldnn::quantize::type_id() && prim.type() != cldnn::custom_gpu_primitive::type_id() && prim.type() != cldnn::concatenation::type_id() && diff --git a/src/plugins/intel_gpu/src/graph/quantize.cpp b/src/plugins/intel_gpu/src/graph/quantize.cpp index dde4aa4b7ec..c4125ae40ee 100644 --- a/src/plugins/intel_gpu/src/graph/quantize.cpp +++ b/src/plugins/intel_gpu/src/graph/quantize.cpp @@ -3,7 +3,6 @@ // #include "quantize_inst.h" -#include "binary_convolution_inst.h" #include "primitive_type_base.h" #include "intel_gpu/runtime/memory.hpp" #include "json_object.h" @@ -22,30 +21,9 @@ layout quantize_inst::calc_output_layout(quantize_node const& node, kernel_impl_ if (desc->output_data_types[0]) out_dt = *desc->output_data_types[0]; - if (out_dt == data_types::u1) { - output_format = format::b_fs_yx_32fp; - } - return layout{out_dt, output_format, input_layout.get_tensor()}; } -template -std::vector quantize_inst::calc_output_layouts(quantize_node const&, kernel_impl_params const& impl_param) { - auto desc = impl_param.typed_desc(); - - auto input_layout = impl_param.get_input_layout(); - auto output_format = input_layout.format; - auto out_dt = desc->output_data_types[0].value_or(input_layout.data_type); - - if (out_dt == data_types::u1) { - output_format = format::b_fs_yx_32fp; - } - - return { layout{input_layout.get(), out_dt, output_format} }; -} - -template std::vector quantize_inst::calc_output_layouts(quantize_node const& node, const kernel_impl_params& impl_param); - std::string quantize_inst::to_string(quantize_node const& node) { auto desc = node.get_primitive(); auto node_info = node.desc_to_json(); diff --git a/src/plugins/intel_gpu/src/kernel_selector/cache/cache.json b/src/plugins/intel_gpu/src/kernel_selector/cache/cache.json index a9662f4606e..59574741bb4 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/cache/cache.json +++ b/src/plugins/intel_gpu/src/kernel_selector/cache/cache.json @@ -10894,28 +10894,17 @@ "4424123045426419379": ["convolution_gpu_bfyx_f16_depthwise", 0], "4163001530200549687": ["convolution_gpu_bfyx_f16", 8], "6890722566263723898": ["convolution_gpu_bfyx_f16_depthwise", 0], - "13967737018625834884": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4904008439880070743": ["convolution_gpu_bfyx_f16_depthwise", 1], - "12495525202846933706": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4591223941823315334": ["convolution_gpu_bfyx_f16_depthwise", 2], - "544756362416159697": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8054350595915663704": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11732173920945220656": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7719296864138745692": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11084677377269310947": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3900078181903132788": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12063794501602674144": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10104159986220401403": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2769623751530494205": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "5912445578783112178": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4676013859334121048": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8428605412862257526": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10635621202663297160": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821402568274932830": ["binary_convolution_gpu_1x1", 0], "14214799641428760795": ["convolution_gpu_bfyx_direct_10_12_16", 1], "2419835076951229610": ["convolution_gpu_bfyx_f16_depthwise", 2], "10657672650587258853": ["convolution_gpu_bfyx_f16", 5], - "13401815977163875034": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "1118760218381327639": ["convolution_gpu_bfyx_f16_depthwise", 2], "13626797216057420236": ["convolution_gpu_bfyx_f16", 6], "2506095387855338923": ["convolution_gpu_bfyx_f16", 3], @@ -10941,52 +10930,22 @@ "8980088396308495358": ["convolution_gpu_bfyx_f16", 7], "6051363798671277490": ["convolution_gpu_bfyx_f16", 5], "15256882419569076308": ["convolution_gpu_bfyx_gemm_like", 2], - "5264667632252570871": ["binary_convolution_gpu_1x1", 0], - "13666815947927457789": ["binary_convolution_gpu_generic", 0], - "17044275001224274100": ["binary_convolution_gpu_1x1", 0], - "12262607945825744026": ["binary_convolution_gpu_1x1", 0], - "14361360851358312136": ["binary_convolution_gpu_1x1", 0], - "3860430324097549563": ["binary_convolution_gpu_generic", 0], - "7128208160650643266": ["binary_convolution_gpu_1x1", 0], - "1062027263129762214": ["binary_convolution_gpu_1x1", 0], - "16561633756389098051": ["binary_convolution_gpu_generic", 0], - "18052555090835437234": ["binary_convolution_gpu_1x1", 0], - "6603476907029730789": ["binary_convolution_gpu_generic", 0], - "6440401458387219749": ["binary_convolution_gpu_1x1", 0], - "7943355244336393643": ["binary_convolution_gpu_1x1", 0], - "11265761184374928749": ["binary_convolution_gpu_generic", 0], - "1225084982500358091": ["binary_convolution_gpu_1x1", 0], - "9666917304428574817": ["binary_convolution_gpu_generic", 0], - "15210383919838660019": ["binary_convolution_gpu_1x1", 0], - "12329467286607927665": ["binary_convolution_gpu_1x1", 0], - "13821628145640330381": ["binary_convolution_gpu_generic", 0], "10766710068843786211": ["fully_connected_gpu_bfyx_ref", 2], "10399951843541697656": ["convolution_gpu_bfyx_to_bfyx_f16", 8], "6121182450365731169": ["convolution_gpu_bfyx_f16_depthwise", 0], "1420839373798024197": ["convolution_gpu_bfyx_f16", 6], - "13139718073646557611": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6766478895508954889": ["convolution_gpu_bfyx_f16_depthwise", 2], - "17134103923720311191": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13452284390313542161": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2726108976392323449": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15079819271991253405": ["convolution_gpu_bfyx_f16_depthwise", 0], - "9323432656779660443": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13582860960891838539": ["convolution_gpu_bfyx_f16_depthwise", 1], - "1055817291271670229": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2823755942522128459": ["convolution_gpu_bfyx_f16_depthwise", 2], - "3384212664007545715": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "2453671492344359798": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10377368418548257894": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821668718539890122": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3048467327118481877": ["convolution_gpu_bfyx_f16_depthwise", 2], - "275456580066174196": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "18142781007687401165": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11817977686815992972": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "656647291151035001": ["convolution_gpu_bfyx_f16", 4], "14551802214127931636": ["convolution_gpu_bfyx_f16_depthwise", 2], "13523379689227815262": ["convolution_gpu_bfyx_f16", 5], "10093198489340308880": ["convolution_gpu_bfyx_f16", 5], - "12285668048424773773": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15179725479322599748": ["convolution_gpu_bfyx_direct_10_12_16", 0], "5314501484112365200": ["convolution_gpu_bfyx_f16_depthwise", 2], "11234282887624973651": ["convolution_gpu_bfyx_f16_1x1", 2], @@ -15152,28 +15111,17 @@ "10677387047764489263": ["convolution_gpu_bfyx_f16_depthwise", 1], "1537866870296831307": ["convolution_gpu_bfyx_f16", 6], "6121043402577263178": ["convolution_gpu_bfyx_f16_depthwise", 1], - "6925053265869446926": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6571448459512229759": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5033302751957212880": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17763423818624479514": ["convolution_gpu_bfyx_f16_depthwise", 1], - "14590866505568013579": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4444924555401610608": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6203626494792050078": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2300190318489790800": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2242155068249197061": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2904120218680757524": ["convolution_gpu_bfyx_f16_depthwise", 0], - "12956000960440491758": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8351838766968536267": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8556976994485015619": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "15984235378444812956": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5573407848022795004": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15497405578993446736": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13977494186365957972": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13526783681740823304": ["binary_convolution_gpu_1x1", 0], "11430675853825242111": ["convolution_gpu_bfyx_os_iyx_osv16", 522], "229385769741075054": ["convolution_gpu_bfyx_f16_depthwise", 2], "16642117060176841433": ["convolution_gpu_bfyx_f16", 5], - "10567925043930198424": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "12594060950826322919": ["convolution_gpu_bfyx_f16_depthwise", 2], "4859984633862705344": ["convolution_gpu_bfyx_f16", 7], "6643541161570220487": ["convolution_gpu_bfyx_f16", 4], @@ -15198,36 +15146,21 @@ "4669930370801439013": ["convolution_gpu_bfyx_f16", 4], "2049835121645334394": ["convolution_gpu_bfyx_f16", 1], "4179197899143727062": ["convolution_gpu_bfyx_f16", 1], - "17629208725190652410": ["binary_convolution_gpu_1x1", 0], - "10203558295793180608": ["binary_convolution_gpu_1x1", 0], - "14083006767377408735": ["binary_convolution_gpu_1x1", 0], - "11002601216030213097": ["binary_convolution_gpu_1x1", 0], "791829835282095596": ["convolution_gpu_bfyx_to_bfyx_f16", 8], "13741392821104156137": ["convolution_gpu_bfyx_f16_depthwise", 2], "14407270906917824601": ["convolution_gpu_bfyx_f16", 6], - "10520976832008005001": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8640243682990139429": ["convolution_gpu_bfyx_f16_depthwise", 1], - "5483210158429664653": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2044520988682161997": ["convolution_gpu_bfyx_f16_depthwise", 1], - "1305091083986203859": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5659956897985857329": ["convolution_gpu_bfyx_f16_depthwise", 0], - "4005952778869826841": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17364300506911036439": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9812558313251709379": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5598344570994891971": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13865812989618108181": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "3791901918413409048": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8565954012969407126": ["convolution_gpu_bfyx_f16_depthwise", 0], - "10881232647513304568": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4973139580034915617": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5367180062414144278": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15140881728515527701": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15616026263121816018": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7285564639878424393": ["convolution_gpu_bfyx_f16", 8], "2742457992410896516": ["convolution_gpu_bfyx_f16_depthwise", 1], "5183001506630431534": ["convolution_gpu_bfyx_f16", 5], "8365841447443821412": ["convolution_gpu_bfyx_f16", 5], - "11741754254612323251": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "616934627583263600": ["convolution_gpu_bfyx_os_iyx_osv16", 551], "15327993174794686756": ["convolution_gpu_bfyx_f16_depthwise", 2], "12722030162332410659": ["convolution_gpu_bfyx_f16_1x1", 1], @@ -23738,28 +23671,17 @@ "4424123045426419379": ["convolution_gpu_bfyx_f16_depthwise", 2], "4163001530200549687": ["convolution_gpu_bfyx_f16", 8], "6890722566263723898": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13967737018625834884": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4904008439880070743": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12495525202846933706": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4591223941823315334": ["convolution_gpu_bfyx_f16_depthwise", 2], - "544756362416159697": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8054350595915663704": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11732173920945220656": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7719296864138745692": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11084677377269310947": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3900078181903132788": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12063794501602674144": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10104159986220401403": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2769623751530494205": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "5912445578783112178": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4676013859334121048": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8428605412862257526": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10635621202663297160": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821402568274932830": ["binary_convolution_gpu_1x1", 0], "14214799641428760795": ["convolution_gpu_bfyx_direct_10_12_16", 1], "2419835076951229610": ["convolution_gpu_bfyx_f16_depthwise", 2], "10657672650587258853": ["convolution_gpu_bfyx_f16", 6], - "13401815977163875034": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "1118760218381327639": ["convolution_gpu_bfyx_f16_depthwise", 2], "13626797216057420236": ["convolution_gpu_bfyx_f16", 8], "2506095387855338923": ["convolution_gpu_bfyx_f16", 2], @@ -23785,52 +23707,22 @@ "8980088396308495358": ["convolution_gpu_bfyx_f16", 2], "6051363798671277490": ["convolution_gpu_bfyx_f16", 2], "15256882419569076308": ["convolution_gpu_bfyx_gemm_like", 2], - "5264667632252570871": ["binary_convolution_gpu_1x1", 0], - "13666815947927457789": ["binary_convolution_gpu_generic", 0], - "17044275001224274100": ["binary_convolution_gpu_1x1", 0], - "12262607945825744026": ["binary_convolution_gpu_1x1", 0], - "14361360851358312136": ["binary_convolution_gpu_1x1", 0], - "3860430324097549563": ["binary_convolution_gpu_generic", 0], - "7128208160650643266": ["binary_convolution_gpu_1x1", 0], - "1062027263129762214": ["binary_convolution_gpu_1x1", 0], - "16561633756389098051": ["binary_convolution_gpu_generic", 0], - "18052555090835437234": ["binary_convolution_gpu_1x1", 0], - "6603476907029730789": ["binary_convolution_gpu_generic", 0], - "6440401458387219749": ["binary_convolution_gpu_1x1", 0], - "7943355244336393643": ["binary_convolution_gpu_1x1", 0], - "11265761184374928749": ["binary_convolution_gpu_generic", 0], - "1225084982500358091": ["binary_convolution_gpu_1x1", 0], - "9666917304428574817": ["binary_convolution_gpu_generic", 0], - "15210383919838660019": ["binary_convolution_gpu_1x1", 0], - "12329467286607927665": ["binary_convolution_gpu_1x1", 0], - "13821628145640330381": ["binary_convolution_gpu_generic", 0], "10766710068843786211": ["fully_connected_gpu_bfyx_ref", 2], "10399951843541697656": ["convolution_gpu_bfyx_to_bfyx_f16", 8], "6121182450365731169": ["convolution_gpu_bfyx_f16_depthwise", 2], "1420839373798024197": ["convolution_gpu_bfyx_f16", 8], - "13139718073646557611": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6766478895508954889": ["convolution_gpu_bfyx_f16_depthwise", 2], - "17134103923720311191": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13452284390313542161": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2726108976392323449": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15079819271991253405": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9323432656779660443": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13582860960891838539": ["convolution_gpu_bfyx_f16_depthwise", 2], - "1055817291271670229": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2823755942522128459": ["convolution_gpu_bfyx_f16_depthwise", 2], - "3384212664007545715": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "2453671492344359798": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10377368418548257894": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821668718539890122": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3048467327118481877": ["convolution_gpu_bfyx_f16_depthwise", 2], - "275456580066174196": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "18142781007687401165": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11817977686815992972": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "656647291151035001": ["convolution_gpu_bfyx_f16", 7], "14551802214127931636": ["convolution_gpu_bfyx_f16_depthwise", 2], "13523379689227815262": ["convolution_gpu_bfyx_f16", 8], "10093198489340308880": ["convolution_gpu_bfyx_f16", 5], - "12285668048424773773": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15179725479322599748": ["convolution_gpu_bfyx_direct_10_12_16", 1], "5314501484112365200": ["convolution_gpu_bfyx_f16_depthwise", 2], "11234282887624973651": ["convolution_gpu_bfyx_f16", 8], @@ -37425,28 +37317,17 @@ "10677387047764489263": ["convolution_gpu_bfyx_f16_depthwise", 0], "1537866870296831307": ["convolution_gpu_bfyx_f16", 6], "6121043402577263178": ["convolution_gpu_bfyx_f16_depthwise", 1], - "6925053265869446926": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6571448459512229759": ["convolution_gpu_bfyx_f16_depthwise", 1], - "5033302751957212880": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17763423818624479514": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14590866505568013579": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4444924555401610608": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6203626494792050078": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2300190318489790800": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2242155068249197061": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2904120218680757524": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12956000960440491758": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8351838766968536267": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8556976994485015619": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "15984235378444812956": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5573407848022795004": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15497405578993446736": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13977494186365957972": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13526783681740823304": ["binary_convolution_gpu_1x1", 0], "11430675853825242111": ["convolution_gpu_bfyx_os_iyx_osv16", 1078], "229385769741075054": ["convolution_gpu_bfyx_f16_depthwise", 2], "16642117060176841433": ["convolution_gpu_bfyx_f16", 6], - "10567925043930198424": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "12594060950826322919": ["convolution_gpu_bfyx_f16_depthwise", 2], "4859984633862705344": ["convolution_gpu_bfyx_f16", 8], "6643541161570220487": ["convolution_gpu_bfyx_f16", 3], @@ -37471,36 +37352,21 @@ "4669930370801439013": ["convolution_gpu_bfyx_f16", 2], "2049835121645334394": ["convolution_gpu_bfyx_f16", 5], "4179197899143727062": ["convolution_gpu_bfyx_f16", 5], - "17629208725190652410": ["binary_convolution_gpu_1x1", 0], - "10203558295793180608": ["binary_convolution_gpu_1x1", 0], - "14083006767377408735": ["binary_convolution_gpu_1x1", 0], - "11002601216030213097": ["binary_convolution_gpu_1x1", 0], "791829835282095596": ["convolution_gpu_bfyx_to_bfyx_f16", 8], "13741392821104156137": ["convolution_gpu_bfyx_f16_depthwise", 2], "14407270906917824601": ["convolution_gpu_bfyx_f16", 8], - "10520976832008005001": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8640243682990139429": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5483210158429664653": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2044520988682161997": ["convolution_gpu_bfyx_f16_depthwise", 2], - "1305091083986203859": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5659956897985857329": ["convolution_gpu_bfyx_f16_depthwise", 2], - "4005952778869826841": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17364300506911036439": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9812558313251709379": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5598344570994891971": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13865812989618108181": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "3791901918413409048": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8565954012969407126": ["convolution_gpu_bfyx_f16_depthwise", 2], - "10881232647513304568": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4973139580034915617": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5367180062414144278": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15140881728515527701": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15616026263121816018": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7285564639878424393": ["convolution_gpu_bfyx_f16", 7], "2742457992410896516": ["convolution_gpu_bfyx_f16_depthwise", 2], "5183001506630431534": ["convolution_gpu_bfyx_f16", 7], "8365841447443821412": ["convolution_gpu_bfyx_f16", 8], - "11741754254612323251": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "616934627583263600": ["convolution_gpu_bfyx_os_iyx_osv16", 547], "15327993174794686756": ["convolution_gpu_bfyx_f16_depthwise", 2], "12722030162332410659": ["convolution_gpu_bfyx_f16_1x1", 2], @@ -65402,28 +65268,17 @@ "4424123045426419379": ["convolution_gpu_bfyx_f16_depthwise", 0], "4163001530200549687": ["convolution_gpu_bfyx_f16", 7], "6890722566263723898": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13967737018625834884": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4904008439880070743": ["convolution_gpu_bfyx_f16_depthwise", 0], - "12495525202846933706": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4591223941823315334": ["convolution_gpu_bfyx_f16_depthwise", 2], - "544756362416159697": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8054350595915663704": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11732173920945220656": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7719296864138745692": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11084677377269310947": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3900078181903132788": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12063794501602674144": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10104159986220401403": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2769623751530494205": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "5912445578783112178": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4676013859334121048": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8428605412862257526": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10635621202663297160": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821402568274932830": ["binary_convolution_gpu_1x1", 0], "14214799641428760795": ["convolution_gpu_bfyx_direct_10_12_16", 2], "2419835076951229610": ["convolution_gpu_bfyx_f16_depthwise", 2], "10657672650587258853": ["convolution_gpu_bfyx_f16", 7], - "13401815977163875034": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "1118760218381327639": ["convolution_gpu_bfyx_f16_depthwise", 2], "13626797216057420236": ["convolution_gpu_bfyx_f16", 7], "2506095387855338923": ["convolution_gpu_bfyx_f16", 1], @@ -65449,52 +65304,22 @@ "8980088396308495358": ["convolution_gpu_bfyx_f16", 4], "6051363798671277490": ["convolution_gpu_bfyx_f16", 4], "15256882419569076308": ["convolution_gpu_bfyx_gemm_like", 2], - "5264667632252570871": ["binary_convolution_gpu_1x1", 0], - "13666815947927457789": ["binary_convolution_gpu_generic", 0], - "17044275001224274100": ["binary_convolution_gpu_1x1", 0], - "12262607945825744026": ["binary_convolution_gpu_1x1", 0], - "14361360851358312136": ["binary_convolution_gpu_1x1", 0], - "3860430324097549563": ["binary_convolution_gpu_generic", 0], - "7128208160650643266": ["binary_convolution_gpu_1x1", 0], - "1062027263129762214": ["binary_convolution_gpu_1x1", 0], - "16561633756389098051": ["binary_convolution_gpu_generic", 0], - "18052555090835437234": ["binary_convolution_gpu_1x1", 0], - "6603476907029730789": ["binary_convolution_gpu_generic", 0], - "6440401458387219749": ["binary_convolution_gpu_1x1", 0], - "7943355244336393643": ["binary_convolution_gpu_1x1", 0], - "11265761184374928749": ["binary_convolution_gpu_generic", 0], - "1225084982500358091": ["binary_convolution_gpu_1x1", 0], - "9666917304428574817": ["binary_convolution_gpu_generic", 0], - "15210383919838660019": ["binary_convolution_gpu_1x1", 0], - "12329467286607927665": ["binary_convolution_gpu_1x1", 0], - "13821628145640330381": ["binary_convolution_gpu_generic", 0], "10766710068843786211": ["fully_connected_gpu_bfyx_ref", 2], "10399951843541697656": ["convolution_gpu_bfyx_to_bfyx_f16", 6], "6121182450365731169": ["convolution_gpu_bfyx_f16_depthwise", 0], "1420839373798024197": ["convolution_gpu_bfyx_f16", 8], - "13139718073646557611": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6766478895508954889": ["convolution_gpu_bfyx_f16_depthwise", 0], - "17134103923720311191": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13452284390313542161": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2726108976392323449": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15079819271991253405": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9323432656779660443": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13582860960891838539": ["convolution_gpu_bfyx_f16_depthwise", 2], - "1055817291271670229": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2823755942522128459": ["convolution_gpu_bfyx_f16_depthwise", 2], - "3384212664007545715": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "2453671492344359798": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10377368418548257894": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821668718539890122": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3048467327118481877": ["convolution_gpu_bfyx_f16_depthwise", 2], - "275456580066174196": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "18142781007687401165": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11817977686815992972": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "656647291151035001": ["convolution_gpu_bfyx_f16_1x1", 2], "14551802214127931636": ["convolution_gpu_bfyx_f16_depthwise", 2], "13523379689227815262": ["convolution_gpu_bfyx_f16", 4], "10093198489340308880": ["convolution_gpu_bfyx_f16", 5], - "12285668048424773773": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15179725479322599748": ["convolution_gpu_bfyx_direct_10_12_16", 2], "5314501484112365200": ["convolution_gpu_bfyx_f16_depthwise", 2], "11234282887624973651": ["convolution_gpu_bfyx_f16", 7], @@ -69720,28 +69545,17 @@ "10677387047764489263": ["convolution_gpu_bfyx_f16_depthwise", 2], "1537866870296831307": ["convolution_gpu_bfyx_f16", 4], "6121043402577263178": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6925053265869446926": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6571448459512229759": ["convolution_gpu_bfyx_f16_depthwise", 0], - "5033302751957212880": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17763423818624479514": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14590866505568013579": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4444924555401610608": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6203626494792050078": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2300190318489790800": ["convolution_gpu_bfyx_f16_depthwise", 1], - "2242155068249197061": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2904120218680757524": ["convolution_gpu_bfyx_f16_depthwise", 1], - "12956000960440491758": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8351838766968536267": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8556976994485015619": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "15984235378444812956": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5573407848022795004": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15497405578993446736": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13977494186365957972": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13526783681740823304": ["binary_convolution_gpu_1x1", 0], "11430675853825242111": ["convolution_gpu_bfyx_os_iyx_osv16", 201], "229385769741075054": ["convolution_gpu_bfyx_f16_depthwise", 1], "16642117060176841433": ["convolution_gpu_bfyx_f16", 7], - "10567925043930198424": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "12594060950826322919": ["convolution_gpu_bfyx_f16_depthwise", 2], "4859984633862705344": ["convolution_gpu_bfyx_f16", 6], "6643541161570220487": ["convolution_gpu_bfyx_f16", 5], @@ -69766,36 +69580,21 @@ "4669930370801439013": ["convolution_gpu_bfyx_f16", 3], "2049835121645334394": ["convolution_gpu_bfyx_f16", 5], "4179197899143727062": ["convolution_gpu_bfyx_f16", 4], - "17629208725190652410": ["binary_convolution_gpu_1x1", 0], - "10203558295793180608": ["binary_convolution_gpu_1x1", 0], - "14083006767377408735": ["binary_convolution_gpu_1x1", 0], - "11002601216030213097": ["binary_convolution_gpu_1x1", 0], "791829835282095596": ["convolution_gpu_bfyx_to_bfyx_f16", 8], "13741392821104156137": ["convolution_gpu_bfyx_f16_depthwise", 2], "14407270906917824601": ["convolution_gpu_bfyx_f16", 8], - "10520976832008005001": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8640243682990139429": ["convolution_gpu_bfyx_f16_depthwise", 0], - "5483210158429664653": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2044520988682161997": ["convolution_gpu_bfyx_f16_depthwise", 0], - "1305091083986203859": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5659956897985857329": ["convolution_gpu_bfyx_f16_depthwise", 0], - "4005952778869826841": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17364300506911036439": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9812558313251709379": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5598344570994891971": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13865812989618108181": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "3791901918413409048": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8565954012969407126": ["convolution_gpu_bfyx_f16_depthwise", 2], - "10881232647513304568": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4973139580034915617": ["convolution_gpu_bfyx_f16_depthwise", 1], - "5367180062414144278": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15140881728515527701": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15616026263121816018": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7285564639878424393": ["convolution_gpu_bfyx_f16_1x1", 1], "2742457992410896516": ["convolution_gpu_bfyx_f16_depthwise", 2], "5183001506630431534": ["convolution_gpu_bfyx_f16", 7], "8365841447443821412": ["convolution_gpu_bfyx_f16", 3], - "11741754254612323251": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "616934627583263600": ["convolution_gpu_bfyx_os_iyx_osv16", 545], "15327993174794686756": ["convolution_gpu_bfyx_f16_depthwise", 2], "12722030162332410659": ["convolution_gpu_bfyx_f16", 6], @@ -75070,28 +74869,17 @@ "4424123045426419379": ["convolution_gpu_bfyx_f16_depthwise", 1], "4163001530200549687": ["convolution_gpu_bfyx_f16", 8], "6890722566263723898": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13967737018625834884": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4904008439880070743": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12495525202846933706": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4591223941823315334": ["convolution_gpu_bfyx_f16_depthwise", 2], - "544756362416159697": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8054350595915663704": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11732173920945220656": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7719296864138745692": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11084677377269310947": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3900078181903132788": ["convolution_gpu_bfyx_f16_depthwise", 1], - "12063794501602674144": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10104159986220401403": ["convolution_gpu_bfyx_f16_depthwise", 1], - "2769623751530494205": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "5912445578783112178": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4676013859334121048": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8428605412862257526": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10635621202663297160": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821402568274932830": ["binary_convolution_gpu_1x1", 0], "14214799641428760795": ["convolution_gpu_bfyx_os_iyx_osv16", 1032], "2419835076951229610": ["convolution_gpu_bfyx_f16_depthwise", 2], "10657672650587258853": ["convolution_gpu_bfyx_f16", 6], - "13401815977163875034": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "1118760218381327639": ["convolution_gpu_bfyx_f16_depthwise", 2], "13626797216057420236": ["convolution_gpu_bfyx_f16", 6], "2506095387855338923": ["convolution_gpu_bfyx_f16", 1], @@ -75117,52 +74905,22 @@ "8980088396308495358": ["convolution_gpu_bfyx_f16", 3], "6051363798671277490": ["convolution_gpu_bfyx_f16", 2], "15256882419569076308": ["convolution_gpu_bfyx_gemm_like", 2], - "5264667632252570871": ["binary_convolution_gpu_1x1", 0], - "13666815947927457789": ["binary_convolution_gpu_generic", 0], - "17044275001224274100": ["binary_convolution_gpu_1x1", 0], - "12262607945825744026": ["binary_convolution_gpu_1x1", 0], - "14361360851358312136": ["binary_convolution_gpu_1x1", 0], - "3860430324097549563": ["binary_convolution_gpu_generic", 0], - "7128208160650643266": ["binary_convolution_gpu_1x1", 0], - "1062027263129762214": ["binary_convolution_gpu_1x1", 0], - "16561633756389098051": ["binary_convolution_gpu_generic", 0], - "18052555090835437234": ["binary_convolution_gpu_1x1", 0], - "6603476907029730789": ["binary_convolution_gpu_generic", 0], - "6440401458387219749": ["binary_convolution_gpu_1x1", 0], - "7943355244336393643": ["binary_convolution_gpu_1x1", 0], - "11265761184374928749": ["binary_convolution_gpu_generic", 0], - "1225084982500358091": ["binary_convolution_gpu_1x1", 0], - "9666917304428574817": ["binary_convolution_gpu_generic", 0], - "15210383919838660019": ["binary_convolution_gpu_1x1", 0], - "12329467286607927665": ["binary_convolution_gpu_1x1", 0], - "13821628145640330381": ["binary_convolution_gpu_generic", 0], "10766710068843786211": ["fully_connected_gpu_bfyx_ref", 1], "10399951843541697656": ["convolution_gpu_bfyx_to_bfyx_f16", 6], "6121182450365731169": ["convolution_gpu_bfyx_f16_depthwise", 1], "1420839373798024197": ["convolution_gpu_bfyx_f16", 8], - "13139718073646557611": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6766478895508954889": ["convolution_gpu_bfyx_f16_depthwise", 2], - "17134103923720311191": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13452284390313542161": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2726108976392323449": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15079819271991253405": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9323432656779660443": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13582860960891838539": ["convolution_gpu_bfyx_f16_depthwise", 2], - "1055817291271670229": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2823755942522128459": ["convolution_gpu_bfyx_f16_depthwise", 2], - "3384212664007545715": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "2453671492344359798": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10377368418548257894": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821668718539890122": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3048467327118481877": ["convolution_gpu_bfyx_f16_depthwise", 1], - "275456580066174196": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "18142781007687401165": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11817977686815992972": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "656647291151035001": ["convolution_gpu_bfyx_f16_1x1", 2], "14551802214127931636": ["convolution_gpu_bfyx_f16_depthwise", 2], "13523379689227815262": ["convolution_gpu_bfyx_f16", 6], "10093198489340308880": ["convolution_gpu_bfyx_f16", 4], - "12285668048424773773": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15179725479322599748": ["convolution_gpu_bfyx_os_iyx_osv16", 168], "5314501484112365200": ["convolution_gpu_bfyx_f16_depthwise", 2], "11234282887624973651": ["convolution_gpu_bfyx_f16", 7], @@ -79463,28 +79221,17 @@ "10677387047764489263": ["convolution_gpu_bfyx_f16_depthwise", 2], "1537866870296831307": ["convolution_gpu_bfyx_f16", 5], "6121043402577263178": ["convolution_gpu_bfyx_f16_depthwise", 1], - "6925053265869446926": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6571448459512229759": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5033302751957212880": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17763423818624479514": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14590866505568013579": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4444924555401610608": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6203626494792050078": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2300190318489790800": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2242155068249197061": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2904120218680757524": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12956000960440491758": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8351838766968536267": ["convolution_gpu_bfyx_f16_depthwise", 1], - "8556976994485015619": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "15984235378444812956": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5573407848022795004": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15497405578993446736": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13977494186365957972": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13526783681740823304": ["binary_convolution_gpu_1x1", 0], "11430675853825242111": ["convolution_gpu_bfyx_os_iyx_osv16", 886], "229385769741075054": ["convolution_gpu_bfyx_f16_depthwise", 1], "16642117060176841433": ["convolution_gpu_bfyx_f16", 7], - "10567925043930198424": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "12594060950826322919": ["convolution_gpu_bfyx_f16_depthwise", 1], "4859984633862705344": ["convolution_gpu_bfyx_f16", 7], "6643541161570220487": ["convolution_gpu_bfyx_f16", 3], @@ -79509,36 +79256,21 @@ "4669930370801439013": ["convolution_gpu_bfyx_f16", 8], "2049835121645334394": ["convolution_gpu_bfyx_f16", 5], "4179197899143727062": ["convolution_gpu_bfyx_f16", 7], - "17629208725190652410": ["binary_convolution_gpu_1x1", 0], - "10203558295793180608": ["binary_convolution_gpu_1x1", 0], - "14083006767377408735": ["binary_convolution_gpu_1x1", 0], - "11002601216030213097": ["binary_convolution_gpu_1x1", 0], "791829835282095596": ["convolution_gpu_bfyx_to_bfyx_f16", 8], "13741392821104156137": ["convolution_gpu_bfyx_f16_depthwise", 2], "14407270906917824601": ["convolution_gpu_bfyx_f16", 5], - "10520976832008005001": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8640243682990139429": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5483210158429664653": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2044520988682161997": ["convolution_gpu_bfyx_f16_depthwise", 2], - "1305091083986203859": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5659956897985857329": ["convolution_gpu_bfyx_f16_depthwise", 2], - "4005952778869826841": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17364300506911036439": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9812558313251709379": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5598344570994891971": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13865812989618108181": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "3791901918413409048": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8565954012969407126": ["convolution_gpu_bfyx_f16_depthwise", 2], - "10881232647513304568": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4973139580034915617": ["convolution_gpu_bfyx_f16_depthwise", 1], - "5367180062414144278": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15140881728515527701": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15616026263121816018": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7285564639878424393": ["convolution_gpu_bfyx_f16", 6], "2742457992410896516": ["convolution_gpu_bfyx_f16_depthwise", 2], "5183001506630431534": ["convolution_gpu_bfyx_f16", 6], "8365841447443821412": ["convolution_gpu_bfyx_f16", 5], - "11741754254612323251": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "616934627583263600": ["convolution_gpu_bfyx_os_iyx_osv16", 544], "15327993174794686756": ["convolution_gpu_bfyx_f16_depthwise", 1], "12722030162332410659": ["convolution_gpu_bfyx_f16", 6], @@ -89452,28 +89184,17 @@ "4424123045426419379": ["convolution_gpu_bfyx_f16_depthwise", 0], "4163001530200549687": ["convolution_gpu_bfyx_f16", 8], "6890722566263723898": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13967737018625834884": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4904008439880070743": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12495525202846933706": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4591223941823315334": ["convolution_gpu_bfyx_f16_depthwise", 2], - "544756362416159697": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8054350595915663704": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11732173920945220656": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7719296864138745692": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11084677377269310947": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3900078181903132788": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12063794501602674144": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10104159986220401403": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2769623751530494205": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "5912445578783112178": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4676013859334121048": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8428605412862257526": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10635621202663297160": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821402568274932830": ["binary_convolution_gpu_1x1", 0], "14214799641428760795": ["convolution_gpu_bfyx_direct_10_12_16", 2], "2419835076951229610": ["convolution_gpu_bfyx_f16_depthwise", 2], "10657672650587258853": ["convolution_gpu_bfyx_f16", 4], - "13401815977163875034": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "1118760218381327639": ["convolution_gpu_bfyx_f16_depthwise", 2], "13626797216057420236": ["convolution_gpu_bfyx_f16", 6], "2506095387855338923": ["convolution_gpu_bfyx_f16", 2], @@ -89499,52 +89220,22 @@ "8980088396308495358": ["convolution_gpu_bfyx_f16", 2], "6051363798671277490": ["convolution_gpu_bfyx_f16", 1], "15256882419569076308": ["convolution_gpu_bfyx_gemm_like", 2], - "5264667632252570871": ["binary_convolution_gpu_1x1", 0], - "13666815947927457789": ["binary_convolution_gpu_generic", 0], - "17044275001224274100": ["binary_convolution_gpu_1x1", 0], - "12262607945825744026": ["binary_convolution_gpu_1x1", 0], - "14361360851358312136": ["binary_convolution_gpu_1x1", 0], - "3860430324097549563": ["binary_convolution_gpu_generic", 0], - "7128208160650643266": ["binary_convolution_gpu_1x1", 0], - "1062027263129762214": ["binary_convolution_gpu_1x1", 0], - "16561633756389098051": ["binary_convolution_gpu_generic", 0], - "18052555090835437234": ["binary_convolution_gpu_1x1", 0], - "6603476907029730789": ["binary_convolution_gpu_generic", 0], - "6440401458387219749": ["binary_convolution_gpu_1x1", 0], - "7943355244336393643": ["binary_convolution_gpu_1x1", 0], - "11265761184374928749": ["binary_convolution_gpu_generic", 0], - "1225084982500358091": ["binary_convolution_gpu_1x1", 0], - "9666917304428574817": ["binary_convolution_gpu_generic", 0], - "15210383919838660019": ["binary_convolution_gpu_1x1", 0], - "12329467286607927665": ["binary_convolution_gpu_1x1", 0], - "13821628145640330381": ["binary_convolution_gpu_generic", 0], "10766710068843786211": ["fully_connected_gpu_bfyx_ref", 2], "10399951843541697656": ["convolution_gpu_bfyx_to_bfyx_f16", 8], "6121182450365731169": ["convolution_gpu_bfyx_f16_depthwise", 2], "1420839373798024197": ["convolution_gpu_bfyx_f16", 8], - "13139718073646557611": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6766478895508954889": ["convolution_gpu_bfyx_f16_depthwise", 2], - "17134103923720311191": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13452284390313542161": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2726108976392323449": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15079819271991253405": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9323432656779660443": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13582860960891838539": ["convolution_gpu_bfyx_f16_depthwise", 2], - "1055817291271670229": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2823755942522128459": ["convolution_gpu_bfyx_f16_depthwise", 2], - "3384212664007545715": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "2453671492344359798": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10377368418548257894": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821668718539890122": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3048467327118481877": ["convolution_gpu_bfyx_f16_depthwise", 2], - "275456580066174196": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "18142781007687401165": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11817977686815992972": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "656647291151035001": ["convolution_gpu_bfyx_f16_1x1", 2], "14551802214127931636": ["convolution_gpu_bfyx_f16_depthwise", 2], "13523379689227815262": ["convolution_gpu_bfyx_f16", 4], "10093198489340308880": ["convolution_gpu_bfyx_f16", 2], - "12285668048424773773": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15179725479322599748": ["convolution_gpu_bfyx_direct_10_12_16", 2], "5314501484112365200": ["convolution_gpu_bfyx_f16_depthwise", 2], "11234282887624973651": ["convolution_gpu_bfyx_f16", 7], @@ -93768,28 +93459,17 @@ "10677387047764489263": ["convolution_gpu_bfyx_f16_depthwise", 2], "1537866870296831307": ["convolution_gpu_bfyx_f16", 8], "6121043402577263178": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6925053265869446926": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6571448459512229759": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5033302751957212880": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17763423818624479514": ["convolution_gpu_bfyx_f16_depthwise", 1], - "14590866505568013579": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4444924555401610608": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6203626494792050078": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2300190318489790800": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2242155068249197061": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2904120218680757524": ["convolution_gpu_bfyx_f16_depthwise", 2], - "12956000960440491758": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8351838766968536267": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8556976994485015619": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "15984235378444812956": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5573407848022795004": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15497405578993446736": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13977494186365957972": ["convolution_gpu_bfyx_f16_depthwise", 1], - "13526783681740823304": ["binary_convolution_gpu_1x1", 0], "11430675853825242111": ["convolution_gpu_bfyx_os_iyx_osv16", 576], "229385769741075054": ["convolution_gpu_bfyx_f16_depthwise", 1], "16642117060176841433": ["convolution_gpu_bfyx_f16", 8], - "10567925043930198424": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "12594060950826322919": ["convolution_gpu_bfyx_f16_depthwise", 2], "4859984633862705344": ["convolution_gpu_bfyx_f16", 7], "6643541161570220487": ["convolution_gpu_bfyx_f16", 2], @@ -93814,36 +93494,21 @@ "4669930370801439013": ["convolution_gpu_bfyx_f16", 5], "2049835121645334394": ["convolution_gpu_bfyx_f16", 5], "4179197899143727062": ["convolution_gpu_bfyx_f16", 8], - "17629208725190652410": ["binary_convolution_gpu_1x1", 0], - "10203558295793180608": ["binary_convolution_gpu_1x1", 0], - "14083006767377408735": ["binary_convolution_gpu_1x1", 0], - "11002601216030213097": ["binary_convolution_gpu_1x1", 0], "791829835282095596": ["convolution_gpu_bfyx_to_bfyx_f16", 6], "13741392821104156137": ["convolution_gpu_bfyx_f16_depthwise", 2], "14407270906917824601": ["convolution_gpu_bfyx_f16", 8], - "10520976832008005001": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8640243682990139429": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5483210158429664653": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2044520988682161997": ["convolution_gpu_bfyx_f16_depthwise", 0], - "1305091083986203859": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5659956897985857329": ["convolution_gpu_bfyx_f16_depthwise", 2], - "4005952778869826841": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17364300506911036439": ["convolution_gpu_bfyx_f16_depthwise", 1], - "9812558313251709379": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5598344570994891971": ["convolution_gpu_bfyx_f16_depthwise", 1], - "13865812989618108181": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "3791901918413409048": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8565954012969407126": ["convolution_gpu_bfyx_f16_depthwise", 2], - "10881232647513304568": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4973139580034915617": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5367180062414144278": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15140881728515527701": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15616026263121816018": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7285564639878424393": ["convolution_gpu_bfyx_f16", 8], "2742457992410896516": ["convolution_gpu_bfyx_f16_depthwise", 2], "5183001506630431534": ["convolution_gpu_bfyx_f16", 8], "8365841447443821412": ["convolution_gpu_bfyx_f16", 3], - "11741754254612323251": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "616934627583263600": ["convolution_gpu_bfyx_os_iyx_osv16", 89], "15327993174794686756": ["convolution_gpu_bfyx_f16_depthwise", 2], "12722030162332410659": ["convolution_gpu_bfyx_f16", 7], @@ -103558,28 +103223,17 @@ "4424123045426419379": ["convolution_gpu_bfyx_f16_depthwise", 1], "4163001530200549687": ["convolution_gpu_bfyx_f16", 6], "6890722566263723898": ["convolution_gpu_bfyx_f16_depthwise", 1], - "13967737018625834884": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4904008439880070743": ["convolution_gpu_bfyx_f16_depthwise", 1], - "12495525202846933706": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4591223941823315334": ["convolution_gpu_bfyx_f16_depthwise", 2], - "544756362416159697": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8054350595915663704": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11732173920945220656": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7719296864138745692": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11084677377269310947": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3900078181903132788": ["convolution_gpu_bfyx_f16_depthwise", 1], - "12063794501602674144": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10104159986220401403": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2769623751530494205": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "5912445578783112178": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4676013859334121048": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8428605412862257526": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10635621202663297160": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821402568274932830": ["binary_convolution_gpu_1x1", 0], "14214799641428760795": ["convolution_gpu_bfyx_direct_10_12_16", 0], "2419835076951229610": ["convolution_gpu_bfyx_f16_depthwise", 2], "10657672650587258853": ["convolution_gpu_bfyx_f16", 1], - "13401815977163875034": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "1118760218381327639": ["convolution_gpu_bfyx_f16_depthwise", 2], "13626797216057420236": ["convolution_gpu_bfyx_f16", 3], "2506095387855338923": ["convolution_gpu_bfyx_f16", 0], @@ -103605,52 +103259,22 @@ "8980088396308495358": ["convolution_gpu_bfyx_f16", 6], "6051363798671277490": ["convolution_gpu_bfyx_f16", 3], "15256882419569076308": ["convolution_gpu_bfyx_gemm_like", 2], - "5264667632252570871": ["binary_convolution_gpu_1x1", 0], - "13666815947927457789": ["binary_convolution_gpu_generic", 0], - "17044275001224274100": ["binary_convolution_gpu_1x1", 0], - "12262607945825744026": ["binary_convolution_gpu_1x1", 0], - "14361360851358312136": ["binary_convolution_gpu_1x1", 0], - "3860430324097549563": ["binary_convolution_gpu_generic", 0], - "7128208160650643266": ["binary_convolution_gpu_1x1", 0], - "1062027263129762214": ["binary_convolution_gpu_1x1", 0], - "16561633756389098051": ["binary_convolution_gpu_generic", 0], - "18052555090835437234": ["binary_convolution_gpu_1x1", 0], - "6603476907029730789": ["binary_convolution_gpu_generic", 0], - "6440401458387219749": ["binary_convolution_gpu_1x1", 0], - "7943355244336393643": ["binary_convolution_gpu_1x1", 0], - "11265761184374928749": ["binary_convolution_gpu_generic", 0], - "1225084982500358091": ["binary_convolution_gpu_1x1", 0], - "9666917304428574817": ["binary_convolution_gpu_generic", 0], - "15210383919838660019": ["binary_convolution_gpu_1x1", 0], - "12329467286607927665": ["binary_convolution_gpu_1x1", 0], - "13821628145640330381": ["binary_convolution_gpu_generic", 0], "10766710068843786211": ["fully_connected_gpu_bfyx_ref", 0], "10399951843541697656": ["convolution_gpu_bfyx_to_bfyx_f16", 6], "6121182450365731169": ["convolution_gpu_bfyx_f16_depthwise", 2], "1420839373798024197": ["convolution_gpu_bfyx_f16", 6], - "13139718073646557611": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6766478895508954889": ["convolution_gpu_bfyx_f16_depthwise", 0], - "17134103923720311191": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13452284390313542161": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2726108976392323449": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15079819271991253405": ["convolution_gpu_bfyx_f16_depthwise", 1], - "9323432656779660443": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13582860960891838539": ["convolution_gpu_bfyx_f16_depthwise", 2], - "1055817291271670229": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2823755942522128459": ["convolution_gpu_bfyx_f16_depthwise", 2], - "3384212664007545715": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "2453671492344359798": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "10377368418548257894": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14821668718539890122": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "3048467327118481877": ["convolution_gpu_bfyx_f16_depthwise", 2], - "275456580066174196": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "18142781007687401165": ["convolution_gpu_bfyx_f16_depthwise", 2], - "11817977686815992972": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "656647291151035001": ["convolution_gpu_bfyx_f16_1x1", 0], "14551802214127931636": ["convolution_gpu_bfyx_f16_depthwise", 2], "13523379689227815262": ["convolution_gpu_bfyx_f16", 0], "10093198489340308880": ["convolution_gpu_bfyx_f16", 5], - "12285668048424773773": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15179725479322599748": ["convolution_gpu_bfyx_direct_10_12_16", 0], "5314501484112365200": ["convolution_gpu_bfyx_f16_depthwise", 2], "11234282887624973651": ["convolution_gpu_bfyx_f16_1x1", 0], @@ -107791,28 +107415,17 @@ "10677387047764489263": ["convolution_gpu_bfyx_f16_depthwise", 2], "1537866870296831307": ["convolution_gpu_bfyx_f16", 3], "6121043402577263178": ["convolution_gpu_bfyx_f16_depthwise", 2], - "6925053265869446926": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "6571448459512229759": ["convolution_gpu_bfyx_f16_depthwise", 1], - "5033302751957212880": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17763423818624479514": ["convolution_gpu_bfyx_f16_depthwise", 2], - "14590866505568013579": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4444924555401610608": ["convolution_gpu_bfyx_f16_depthwise", 1], - "6203626494792050078": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2300190318489790800": ["convolution_gpu_bfyx_f16_depthwise", 2], - "2242155068249197061": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2904120218680757524": ["convolution_gpu_bfyx_f16_depthwise", 1], - "12956000960440491758": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8351838766968536267": ["convolution_gpu_bfyx_f16_depthwise", 2], - "8556976994485015619": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "15984235378444812956": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5573407848022795004": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15497405578993446736": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "13977494186365957972": ["convolution_gpu_bfyx_f16_depthwise", 2], - "13526783681740823304": ["binary_convolution_gpu_1x1", 0], "11430675853825242111": ["convolution_gpu_bfyx_os_iyx_osv16", 95], "229385769741075054": ["convolution_gpu_bfyx_f16_depthwise", 1], "16642117060176841433": ["convolution_gpu_bfyx_f16", 0], - "10567925043930198424": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "12594060950826322919": ["convolution_gpu_bfyx_f16_depthwise", 2], "4859984633862705344": ["convolution_gpu_bfyx_f16", 3], "6643541161570220487": ["convolution_gpu_bfyx_f16", 0], @@ -107837,36 +107450,21 @@ "4669930370801439013": ["convolution_gpu_bfyx_f16_1x1", 2], "2049835121645334394": ["convolution_gpu_bfyx_f16", 2], "4179197899143727062": ["convolution_gpu_bfyx_f16", 2], - "17629208725190652410": ["binary_convolution_gpu_1x1", 0], - "10203558295793180608": ["binary_convolution_gpu_1x1", 0], - "14083006767377408735": ["binary_convolution_gpu_1x1", 0], - "11002601216030213097": ["binary_convolution_gpu_1x1", 0], "791829835282095596": ["convolution_gpu_bfyx_to_bfyx_f16", 6], "13741392821104156137": ["convolution_gpu_bfyx_f16_depthwise", 2], "14407270906917824601": ["convolution_gpu_bfyx_f16", 8], - "10520976832008005001": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8640243682990139429": ["convolution_gpu_bfyx_f16_depthwise", 2], - "5483210158429664653": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "2044520988682161997": ["convolution_gpu_bfyx_f16_depthwise", 0], - "1305091083986203859": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5659956897985857329": ["convolution_gpu_bfyx_f16_depthwise", 2], - "4005952778869826841": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "17364300506911036439": ["convolution_gpu_bfyx_f16_depthwise", 2], - "9812558313251709379": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "5598344570994891971": ["convolution_gpu_bfyx_f16_depthwise", 1], - "13865812989618108181": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], - "3791901918413409048": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "8565954012969407126": ["convolution_gpu_bfyx_f16_depthwise", 2], - "10881232647513304568": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "4973139580034915617": ["convolution_gpu_bfyx_f16_depthwise", 1], - "5367180062414144278": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "15140881728515527701": ["convolution_gpu_bfyx_f16_depthwise", 2], - "15616026263121816018": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "7285564639878424393": ["convolution_gpu_bfyx_f16", 5], "2742457992410896516": ["convolution_gpu_bfyx_f16_depthwise", 1], "5183001506630431534": ["convolution_gpu_bfyx_f16", 0], "8365841447443821412": ["convolution_gpu_bfyx_f16", 8], - "11741754254612323251": ["binary_convolution_gpu_1x1_b_fs_yx_fsv16", 0], "616934627583263600": ["convolution_gpu_bfyx_os_iyx_osv16", 183], "15327993174794686756": ["convolution_gpu_bfyx_f16_depthwise", 2], "12722030162332410659": ["convolution_gpu_bfyx_f16_1x1", 0], @@ -114131,4 +113729,4 @@ "8519379094225608238": ["fully_connected_gpu_fb_io_ref", 0] } } -} \ No newline at end of file +} diff --git a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_1x1.cl b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_1x1.cl deleted file mode 100644 index 4e4165bd99c..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_1x1.cl +++ /dev/null @@ -1,215 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "include/batch_headers/sub_group_block_read.cl" -#include "include/batch_headers/sub_group_block_write.cl" -#include "include/batch_headers/sub_group_shuffle.cl" -#include "include/batch_headers/fetch_data.cl" - -#define OC_BLOCK_SIZE 32 - -#define GET_WEI(data, id) _sub_group_shuffle(data, id) -#define ALIGNED_BLOCK_READ(ptr, byte_offset) as_uint(_sub_group_block_read((const __global uint*)(ptr) + (byte_offset))) -#define ALIGNED_BLOCK_WRITE(ptr, byte_offset, val) _sub_group_block_write((__global uint*)(ptr) + (byte_offset), as_uint(val)) -#define ALIGNED_BLOCK_READ2(ptr, byte_offset) as_uint2(_sub_group_block_read2((const __global uint*)(ptr) + (byte_offset))) - -REQD_SUB_GROUP_SIZE(SUB_GROUP_SIZE) -__attribute__((reqd_work_group_size(SUB_GROUP_SIZE, 1, 1))) -KERNEL(binary_convolution_1x1)(const __global INPUT0_TYPE* input, - __global OUTPUT_TYPE* output, - const __global FILTER_TYPE* weights -#if HAS_FUSED_OPS_DECLS - , FUSED_OPS_DECLS -#endif -) -{ - const int xy = get_group_id(0); - const int f_block = get_global_id(1); - const int b = get_global_id(2); - const int lid = get_sub_group_local_id(); -#if PADDED_INPUT - const int x = (xy * XY_BLOCK_SIZE + lid) % OUTPUT_SIZE_X; - const int y = (xy * XY_BLOCK_SIZE + lid) / OUTPUT_SIZE_X; - const uint input_offset = INPUT0_OFFSET - + b*INPUT0_FEATURE_NUM_PACKED*INPUT0_FEATURE_PITCH - + y*INPUT0_Y_PITCH; -#else - const int x = (xy * XY_BLOCK_SIZE + lid) % OUTPUT_SIZE_X; - const int y = (xy * XY_BLOCK_SIZE + lid) / OUTPUT_SIZE_X; - const uint input_offset = INPUT0_OFFSET - + b*INPUT0_FEATURE_NUM_PACKED*INPUT0_FEATURE_PITCH - + xy*XY_BLOCK_SIZE; -#endif - typedef MAKE_VECTOR_TYPE(FILTER_TYPE, 2) wei_t; - -#if BINARY_PACKED_OUTPUT - const uint dst_index = OUTPUT_OFFSET - + b*OUTPUT_FEATURE_NUM_PACKED*OUTPUT_FEATURE_PITCH - + f_block*OUTPUT_FEATURE_PITCH; -#else - const uint dst_index = OUTPUT_OFFSET - + b*OUTPUT_BATCH_PITCH - + f_block*OC_BLOCK_SIZE*OUTPUT_FEATURE_PITCH; -#endif - const uint filter_offset = f_block*OC_BLOCK_SIZE*INPUT0_FEATURE_NUM_PACKED; - - int dst_buf[OC_BLOCK_SIZE] = { 0 }; // 32 OC - - for (int k = 0; k < INPUT0_FEATURE_NUM_PACKED; ++k) - { - // Load 16 input elements from feature map by subgroup -#if PADDED_INPUT - INPUT0_TYPE src = input[input_offset + k*INPUT0_FEATURE_PITCH + x]; -#else - INPUT0_TYPE src = ALIGNED_BLOCK_READ(input, input_offset + k*INPUT0_FEATURE_PITCH); -#endif - - // Load 32 OC x 32 ICP. Each WI has lid-th and (lid+16)-th channels - wei_t wei = ALIGNED_BLOCK_READ2(weights, filter_offset + k * OC_BLOCK_SIZE); - - // Shuffle 32 OC x 32 ICP of weights in each WI - const wei_t wei0 = GET_WEI(wei, 0); - const wei_t wei1 = GET_WEI(wei, 1); - const wei_t wei2 = GET_WEI(wei, 2); - const wei_t wei3 = GET_WEI(wei, 3); - const wei_t wei4 = GET_WEI(wei, 4); - const wei_t wei5 = GET_WEI(wei, 5); - const wei_t wei6 = GET_WEI(wei, 6); - const wei_t wei7 = GET_WEI(wei, 7); - const wei_t wei8 = GET_WEI(wei, 8); - const wei_t wei9 = GET_WEI(wei, 9); - const wei_t wei10 = GET_WEI(wei, 10); - const wei_t wei11 = GET_WEI(wei, 11); - const wei_t wei12 = GET_WEI(wei, 12); - const wei_t wei13 = GET_WEI(wei, 13); - const wei_t wei14 = GET_WEI(wei, 14); - const wei_t wei15 = GET_WEI(wei, 15); - -#if LEFTOVERS_IC - if (k == INPUT0_FEATURE_NUM_PACKED - 1) - { - dst_buf[0] += popcount((wei0.s0 ^ src) & FILTER_MASK); - dst_buf[1] += popcount((wei1.s0 ^ src) & FILTER_MASK); - dst_buf[2] += popcount((wei2.s0 ^ src) & FILTER_MASK); - dst_buf[3] += popcount((wei3.s0 ^ src) & FILTER_MASK); - dst_buf[4] += popcount((wei4.s0 ^ src) & FILTER_MASK); - dst_buf[5] += popcount((wei5.s0 ^ src) & FILTER_MASK); - dst_buf[6] += popcount((wei6.s0 ^ src) & FILTER_MASK); - dst_buf[7] += popcount((wei7.s0 ^ src) & FILTER_MASK); - dst_buf[8] += popcount((wei8.s0 ^ src) & FILTER_MASK); - dst_buf[9] += popcount((wei9.s0 ^ src) & FILTER_MASK); - dst_buf[10] += popcount((wei10.s0 ^ src) & FILTER_MASK); - dst_buf[11] += popcount((wei11.s0 ^ src) & FILTER_MASK); - dst_buf[12] += popcount((wei12.s0 ^ src) & FILTER_MASK); - dst_buf[13] += popcount((wei13.s0 ^ src) & FILTER_MASK); - dst_buf[14] += popcount((wei14.s0 ^ src) & FILTER_MASK); - dst_buf[15] += popcount((wei15.s0 ^ src) & FILTER_MASK); - -#if OUTPUT_FEATURE_NUM > 16 - dst_buf[16] += popcount((wei0.s1 ^ src) & FILTER_MASK); - dst_buf[17] += popcount((wei1.s1 ^ src) & FILTER_MASK); - dst_buf[18] += popcount((wei2.s1 ^ src) & FILTER_MASK); - dst_buf[19] += popcount((wei3.s1 ^ src) & FILTER_MASK); - dst_buf[20] += popcount((wei4.s1 ^ src) & FILTER_MASK); - dst_buf[21] += popcount((wei5.s1 ^ src) & FILTER_MASK); - dst_buf[22] += popcount((wei6.s1 ^ src) & FILTER_MASK); - dst_buf[23] += popcount((wei7.s1 ^ src) & FILTER_MASK); - dst_buf[24] += popcount((wei8.s1 ^ src) & FILTER_MASK); - dst_buf[25] += popcount((wei9.s1 ^ src) & FILTER_MASK); - dst_buf[26] += popcount((wei10.s1 ^ src) & FILTER_MASK); - dst_buf[27] += popcount((wei11.s1 ^ src) & FILTER_MASK); - dst_buf[28] += popcount((wei12.s1 ^ src) & FILTER_MASK); - dst_buf[29] += popcount((wei13.s1 ^ src) & FILTER_MASK); - dst_buf[30] += popcount((wei14.s1 ^ src) & FILTER_MASK); - dst_buf[31] += popcount((wei15.s1 ^ src) & FILTER_MASK); -#endif - break; - } -#endif - dst_buf[0] += popcount(wei0.s0 ^ src); - dst_buf[1] += popcount(wei1.s0 ^ src); - dst_buf[2] += popcount(wei2.s0 ^ src); - dst_buf[3] += popcount(wei3.s0 ^ src); - dst_buf[4] += popcount(wei4.s0 ^ src); - dst_buf[5] += popcount(wei5.s0 ^ src); - dst_buf[6] += popcount(wei6.s0 ^ src); - dst_buf[7] += popcount(wei7.s0 ^ src); - dst_buf[8] += popcount(wei8.s0 ^ src); - dst_buf[9] += popcount(wei9.s0 ^ src); - dst_buf[10] += popcount(wei10.s0 ^ src); - dst_buf[11] += popcount(wei11.s0 ^ src); - dst_buf[12] += popcount(wei12.s0 ^ src); - dst_buf[13] += popcount(wei13.s0 ^ src); - dst_buf[14] += popcount(wei14.s0 ^ src); - dst_buf[15] += popcount(wei15.s0 ^ src); - -#if OUTPUT_FEATURE_NUM > 16 - dst_buf[16] += popcount(wei0.s1 ^ src); - dst_buf[17] += popcount(wei1.s1 ^ src); - dst_buf[18] += popcount(wei2.s1 ^ src); - dst_buf[19] += popcount(wei3.s1 ^ src); - dst_buf[20] += popcount(wei4.s1 ^ src); - dst_buf[21] += popcount(wei5.s1 ^ src); - dst_buf[22] += popcount(wei6.s1 ^ src); - dst_buf[23] += popcount(wei7.s1 ^ src); - dst_buf[24] += popcount(wei8.s1 ^ src); - dst_buf[25] += popcount(wei9.s1 ^ src); - dst_buf[26] += popcount(wei10.s1 ^ src); - dst_buf[27] += popcount(wei11.s1 ^ src); - dst_buf[28] += popcount(wei12.s1 ^ src); - dst_buf[29] += popcount(wei13.s1 ^ src); - dst_buf[30] += popcount(wei14.s1 ^ src); - dst_buf[31] += popcount(wei15.s1 ^ src); -#endif - } - - // Load data for fused operations (scales, biases, quantization thresholds, etc) -#if CUSTOM_FUSED_OPS - FUSED_OPS_PREPARE_DATA; -#endif - - UNIT_TYPE dst[OC_BLOCK_SIZE]; - for (int oc = 0; oc < OC_BLOCK_SIZE; oc++) - { - CONV_RESULT_TYPE res = TO_CONV_RESULT_TYPE(INPUT0_FEATURE_NUM - 2*dst_buf[oc]); -#if CUSTOM_FUSED_OPS - DO_ELTWISE_FUSED_OPS; -// Don't save floating-point intermediate result, since packed one is already computed -#if !BINARY_PACKED_OUTPUT - dst[oc] = res; -#endif -#elif HAS_FUSED_OPS - FUSED_OPS; - dst[oc] = FUSED_OPS_RESULT; -#endif - - } - - bool in_x = x < OUTPUT_SIZE_X; - bool in_y = y < OUTPUT_SIZE_Y; -#if BINARY_PACKED_OUTPUT - -#if PADDED_OUTPUT - if (in_x && in_y) - output[dst_index + y*OUTPUT_Y_PITCH + x] = TO_OUTPUT_TYPE(packed_res); -#else - if (xy * XY_BLOCK_SIZE < OUTPUT_SIZE_X*OUTPUT_SIZE_Y) - ALIGNED_BLOCK_WRITE(output, dst_index + xy*XY_BLOCK_SIZE, TO_OUTPUT_TYPE(packed_res)); - else if (in_x && in_y) - output[dst_index + y*OUTPUT_Y_PITCH + x] = TO_OUTPUT_TYPE(packed_res); - -#endif - -#else - - for (int oc = 0; oc < OC_BLOCK_SIZE; oc++) - { - bool in_fm = f_block*OC_BLOCK_SIZE + oc < OUTPUT_FEATURE_NUM; - if (in_x && in_y && in_fm) - output[dst_index + oc*OUTPUT_FEATURE_PITCH + y*OUTPUT_Y_PITCH + x] = TO_OUTPUT_TYPE(dst[oc]); - } - -#endif - -} diff --git a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_1x1_b_fs_yx_fsv16.cl b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_1x1_b_fs_yx_fsv16.cl deleted file mode 100644 index 32586d3134a..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_1x1_b_fs_yx_fsv16.cl +++ /dev/null @@ -1,168 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "include/batch_headers/sub_group_block_read.cl" -#include "include/batch_headers/sub_group_block_write.cl" -#include "include/batch_headers/sub_group_shuffle.cl" -#include "include/batch_headers/fetch_data.cl" -#include "include/unit_type.cl" - -#define OC_BLOCK_SIZE 16 - -#define GET_SRC(data, id) _sub_group_shuffle(data, id) -#define ALIGNED_BLOCK_READ(ptr, byte_offset) as_uint(_sub_group_block_read((const __global uint*)(ptr) + (byte_offset))) -#define ALIGNED_BLOCK_READ2(ptr, byte_offset) as_uint2(_sub_group_block_read2((const __global uint*)(ptr) + (byte_offset))) - -REQD_SUB_GROUP_SIZE(SUB_GROUP_SIZE) -__attribute__((reqd_work_group_size(SUB_GROUP_SIZE, 1, 1))) -KERNEL(binary_convolution_1x1_b_fs_yx_fsv16)(const __global INPUT0_TYPE* input, - __global OUTPUT_TYPE* output, - const __global FILTER_TYPE* weights -#if HAS_FUSED_OPS_DECLS - , FUSED_OPS_DECLS -#endif -) -{ - const int xy = get_group_id(0); - const int f_block = get_global_id(1); - const int b = get_global_id(2); - const int lid = get_sub_group_local_id(); -#if PADDED_INPUT - const int x = (xy * XY_BLOCK_SIZE + lid) % OUTPUT_SIZE_X; - const int y = (xy * XY_BLOCK_SIZE + lid) / OUTPUT_SIZE_X; - const uint input_offset = INPUT0_OFFSET - + b*INPUT0_FEATURE_NUM_PACKED*INPUT0_FEATURE_PITCH - + y*INPUT0_Y_PITCH; -#else - const int x = (xy * XY_BLOCK_SIZE + lid) % OUTPUT_SIZE_X; - const int y = (xy * XY_BLOCK_SIZE + lid) / OUTPUT_SIZE_X; - const uint input_offset = INPUT0_OFFSET - + b*INPUT0_FEATURE_NUM_PACKED*INPUT0_FEATURE_PITCH - + xy*XY_BLOCK_SIZE; -#endif - const uint output_x_pitch = OC_BLOCK_SIZE; - const uint output_y_pitch = output_x_pitch * (OUTPUT_PAD_BEFORE_SIZE_X + OUTPUT_SIZE_X + OUTPUT_PAD_AFTER_SIZE_X); - const uint output_total_f_size = OUTPUT_PAD_BEFORE_FEATURE_NUM + OUTPUT_FEATURE_NUM + OUTPUT_PAD_AFTER_FEATURE_NUM; - const uint output_fs_pitch = output_y_pitch * (OUTPUT_PAD_BEFORE_SIZE_Y + OUTPUT_SIZE_Y + OUTPUT_PAD_AFTER_SIZE_Y); - const uint output_b_pitch = output_fs_pitch * ((output_total_f_size + OC_BLOCK_SIZE - 1) / OC_BLOCK_SIZE); - const uint dst_index = OUTPUT_OFFSET*OC_BLOCK_SIZE - + b*output_b_pitch - + f_block*output_fs_pitch; - - const uint filter_offset = ((f_block/2)*2)*OC_BLOCK_SIZE*INPUT0_FEATURE_NUM_PACKED + (f_block%2)*16; - - int dst_buf[OC_BLOCK_SIZE] = { 0 }; // 16 X - - for (int k = 0; k < INPUT0_FEATURE_NUM_PACKED; ++k) - { - // Load 16 input elements from feature map by subgroup -#if PADDED_INPUT - INPUT0_TYPE src = input[input_offset + k*INPUT0_FEATURE_PITCH + x]; -#else - INPUT0_TYPE src = ALIGNED_BLOCK_READ(input, input_offset + k*INPUT0_FEATURE_PITCH); -#endif - - // Load 32 OC x 32 ICP. Each WI has lid-th and (lid+16)-th channels - FILTER_TYPE wei = ALIGNED_BLOCK_READ(weights, filter_offset + k * OC_BLOCK_SIZE*2); - - // Shuffle 2 OC x 32 ICP x 16 X of src - const INPUT0_TYPE src0 = GET_SRC(src, 0); - const INPUT0_TYPE src1 = GET_SRC(src, 1); - const INPUT0_TYPE src2 = GET_SRC(src, 2); - const INPUT0_TYPE src3 = GET_SRC(src, 3); - const INPUT0_TYPE src4 = GET_SRC(src, 4); - const INPUT0_TYPE src5 = GET_SRC(src, 5); - const INPUT0_TYPE src6 = GET_SRC(src, 6); - const INPUT0_TYPE src7 = GET_SRC(src, 7); - const INPUT0_TYPE src8 = GET_SRC(src, 8); - const INPUT0_TYPE src9 = GET_SRC(src, 9); - const INPUT0_TYPE src10 = GET_SRC(src, 10); - const INPUT0_TYPE src11 = GET_SRC(src, 11); - const INPUT0_TYPE src12 = GET_SRC(src, 12); - const INPUT0_TYPE src13 = GET_SRC(src, 13); - const INPUT0_TYPE src14 = GET_SRC(src, 14); - const INPUT0_TYPE src15 = GET_SRC(src, 15); - -#if LEFTOVERS_IC - if (k == INPUT0_FEATURE_NUM_PACKED - 1) - { - dst_buf[0] += popcount((wei ^ src0) & FILTER_MASK); - dst_buf[1] += popcount((wei ^ src1) & FILTER_MASK); - dst_buf[2] += popcount((wei ^ src2) & FILTER_MASK); - dst_buf[3] += popcount((wei ^ src3) & FILTER_MASK); - dst_buf[4] += popcount((wei ^ src4) & FILTER_MASK); - dst_buf[5] += popcount((wei ^ src5) & FILTER_MASK); - dst_buf[6] += popcount((wei ^ src6) & FILTER_MASK); - dst_buf[7] += popcount((wei ^ src7) & FILTER_MASK); - dst_buf[8] += popcount((wei ^ src8) & FILTER_MASK); - dst_buf[9] += popcount((wei ^ src9) & FILTER_MASK); - dst_buf[10] += popcount((wei ^ src10) & FILTER_MASK); - dst_buf[11] += popcount((wei ^ src11) & FILTER_MASK); - dst_buf[12] += popcount((wei ^ src12) & FILTER_MASK); - dst_buf[13] += popcount((wei ^ src13) & FILTER_MASK); - dst_buf[14] += popcount((wei ^ src14) & FILTER_MASK); - dst_buf[15] += popcount((wei ^ src15) & FILTER_MASK); - break; - } -#endif - dst_buf[0] += popcount(wei ^ src0); - dst_buf[1] += popcount(wei ^ src1); - dst_buf[2] += popcount(wei ^ src2); - dst_buf[3] += popcount(wei ^ src3); - dst_buf[4] += popcount(wei ^ src4); - dst_buf[5] += popcount(wei ^ src5); - dst_buf[6] += popcount(wei ^ src6); - dst_buf[7] += popcount(wei ^ src7); - dst_buf[8] += popcount(wei ^ src8); - dst_buf[9] += popcount(wei ^ src9); - dst_buf[10] += popcount(wei ^ src10); - dst_buf[11] += popcount(wei ^ src11); - dst_buf[12] += popcount(wei ^ src12); - dst_buf[13] += popcount(wei ^ src13); - dst_buf[14] += popcount(wei ^ src14); - dst_buf[15] += popcount(wei ^ src15); - } - - // Load data for fused operations (scales, biases, quantization thresholds, etc) -#if CUSTOM_FUSED_OPS - FUSED_OPS_PREPARE_DATA; -#endif - - OUTPUT_TYPE dst[OC_BLOCK_SIZE]; - __attribute__((opencl_unroll_hint(OC_BLOCK_SIZE))) - for (int oc = 0; oc < OC_BLOCK_SIZE; oc++) - { - CONV_RESULT_TYPE res = TO_CONV_RESULT_TYPE(INPUT0_FEATURE_NUM - 2*dst_buf[oc]); -#if CUSTOM_FUSED_OPS - DO_ELTWISE_FUSED_OPS; - dst[oc] = res; -#elif HAS_FUSED_OPS - FUSED_OPS; - dst[oc] = TO_OUTPUT_TYPE(FUSED_OPS_RESULT); -#endif - } - -#if LEFTOVERS_OC - bool in_fm = f_block*OC_BLOCK_SIZE + lid < OUTPUT_FEATURE_NUM; - __attribute__((opencl_unroll_hint(SUB_GROUP_SIZE))) - for (int ox = 0; ox < SUB_GROUP_SIZE; ox++) { - int xi = (xy * XY_BLOCK_SIZE+ox) % OUTPUT_SIZE_X; - int yi = (xy * XY_BLOCK_SIZE+ox) / OUTPUT_SIZE_X; - bool in_x = xi < OUTPUT_SIZE_X; - bool in_y = yi < OUTPUT_SIZE_Y; - if (in_x && in_y && in_fm) { - output[dst_index + yi*output_y_pitch + xi*output_x_pitch + lid] = dst[ox]; - } - } -#else - for (int ox = 0; ox < SUB_GROUP_SIZE; ox++) { - int xi = (xy * XY_BLOCK_SIZE+ox) % OUTPUT_SIZE_X; - int yi = (xy * XY_BLOCK_SIZE+ox) / OUTPUT_SIZE_X; - bool in_x = xi < OUTPUT_SIZE_X; - bool in_y = yi < OUTPUT_SIZE_Y; - if (in_x && in_y) - UNIT_BLOCK_WRITE(output, dst_index + yi*output_y_pitch + xi*output_x_pitch, dst[ox]); - } -#endif -} diff --git a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_generic.cl b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_generic.cl deleted file mode 100644 index 9536247e721..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_generic.cl +++ /dev/null @@ -1,201 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "include/batch_headers/sub_group_block_read.cl" -#include "include/batch_headers/sub_group_shuffle.cl" -#include "include/batch_headers/fetch_data.cl" - -#define OC_BLOCK_SIZE 32 - -#define ALIGNED_BLOCK_READ(ptr, byte_offset) as_uint(_sub_group_block_read((const __global uint*)(ptr) + (byte_offset))) -#define ALIGNED_BLOCK_READ2(ptr, byte_offset) as_uint2(_sub_group_block_read2((const __global uint*)(ptr) + (byte_offset))) - -#if BINARY_PACKED_OUTPUT - #define BUFFER_TYPE UNIT_TYPE -#else - #define BUFFER_TYPE OUTPUT_TYPE -#endif - -REQD_SUB_GROUP_SIZE(SUB_GROUP_SIZE) -__attribute__((reqd_work_group_size(SUB_GROUP_SIZE, 1, 1))) -KERNEL(binary_convolution_generic)(const __global INPUT0_TYPE* input, - __global OUTPUT_TYPE* output, - const __global FILTER_TYPE* weights -#if HAS_FUSED_OPS_DECLS - , FUSED_OPS_DECLS -#endif -) -{ - const int f_block = get_global_id(1); - const int lid = get_sub_group_local_id(); - const int b = get_global_id(2); - - const int xy = get_group_id(0); - const int x = (xy % X_BLOCKS) * OUTPUT_X_BLOCK_SIZE; - const int y = (xy / X_BLOCKS); - - const int input_x = x * STRIDE_SIZE_X - PADDING_SIZE_X; - const int input_y = y * STRIDE_SIZE_Y - PADDING_SIZE_Y; - - const uint input_offset = INPUT0_OFFSET - + b*INPUT0_FEATURE_NUM_PACKED*INPUT0_FEATURE_PITCH - + input_y*INPUT0_Y_PITCH - + input_x*INPUT0_X_PITCH; - - typedef MAKE_VECTOR_TYPE(FILTER_TYPE, 2) data_t; - -#if BINARY_PACKED_OUTPUT - const uint dst_index = OUTPUT_OFFSET - + b*OUTPUT_FEATURE_NUM_PACKED*OUTPUT_FEATURE_PITCH - + f_block*OUTPUT_FEATURE_PITCH - + y*OUTPUT_Y_PITCH - + x; -#else - const uint dst_index = OUTPUT_OFFSET - + b*OUTPUT_BATCH_PITCH - + f_block*OC_BLOCK_SIZE*OUTPUT_FEATURE_PITCH - + y*OUTPUT_Y_PITCH - + x; -#endif - const uint filter_offset = f_block*OC_BLOCK_SIZE*INPUT0_FEATURE_NUM_PACKED*FILTER_SIZE_Y*FILTER_SIZE_X; - - int dst_buf[SUB_GROUP_SIZE*2] = { 0 }; // 2 OC x 16 X - -#if EXCLUDE_PAD - int real_ks = 0; - // calc real kernel size for out_x = x+lid - for (int kh = 0; kh < FILTER_SIZE_Y; kh++) - { - for (int kw = 0; kw < FILTER_SIZE_X; kw++) - { - real_ks += ((input_x + kw + lid*STRIDE_SIZE_X >= 0) && - (input_x + kw + lid*STRIDE_SIZE_X < INPUT0_SIZE_X) && - (input_y + kh >= 0) && - (input_y + kh < INPUT0_SIZE_Y)) ? 1 : 0; - } - } -#endif - - for (int k = 0; k < INPUT0_FEATURE_NUM_PACKED; ++k) - { - for (int kh = 0; kh < FILTER_SIZE_Y; kh++) - { - INPUT0_TYPE line_cache[INPUT_ELEMENTS_PER_WI]; - for (int i = 0; i < INPUT_ELEMENTS_PER_WI; i++) - { - line_cache[i] = PAD_VALUE; - } - - if (input_y + kh >= 0 && input_y + kh < INPUT0_SIZE_Y) - { - for (int i = 0; i < INPUT_ELEMENTS_PER_WI; i++) - { - if (input_x + i*SUB_GROUP_SIZE >= 0 && input_x + (i+1)*SUB_GROUP_SIZE < INPUT0_SIZE_X) - line_cache[i] = ALIGNED_BLOCK_READ(input, input_offset + kh*INPUT0_Y_PITCH + k*INPUT0_FEATURE_PITCH + i*SUB_GROUP_SIZE); - else if (input_x + i*SUB_GROUP_SIZE + lid >= 0 && input_x + i*SUB_GROUP_SIZE + lid < INPUT0_SIZE_X) - line_cache[i] = input[input_offset + kh*INPUT0_Y_PITCH + k*INPUT0_FEATURE_PITCH + i*SUB_GROUP_SIZE + lid]; - } - } - - __attribute__((opencl_unroll_hint(FILTER_SIZE_X))) - for (int kw = 0; kw < FILTER_SIZE_X; kw++) - { - // Load 32 OC x 32 ICP. Each WI has lid-th and (lid+16)-th channels - data_t wei = ALIGNED_BLOCK_READ2(weights, filter_offset + OC_BLOCK_SIZE*(k*FILTER_SIZE_Y*FILTER_SIZE_X + kh*FILTER_SIZE_X + kw)); - - // Single WI in subgroup calcs 2 OC x 16 X elements - __attribute__((opencl_unroll_hint(SUB_GROUP_SIZE))) - for (int i = 0; i < SUB_GROUP_SIZE; i++) - { - INPUT0_TYPE src = _sub_group_shuffle(line_cache[(kw + i*STRIDE_SIZE_X) / SUB_GROUP_SIZE], - (kw + i*STRIDE_SIZE_X) % SUB_GROUP_SIZE); -#if EXCLUDE_PAD - int compute = ((input_x + kw + i*STRIDE_SIZE_X >= 0) && - (input_x + kw + i*STRIDE_SIZE_X < INPUT0_SIZE_X) && - (input_y + kh >= 0) && - (input_y + kh < INPUT0_SIZE_Y)) ? 1 : 0; - - if (!compute) - continue; -#endif - -#if LEFTOVERS_IC - if (k == INPUT0_FEATURE_NUM_PACKED - 1) - { - dst_buf[0*SUB_GROUP_SIZE + i] += popcount((wei.s0 ^ src) & FILTER_MASK); - dst_buf[1*SUB_GROUP_SIZE + i] += popcount((wei.s1 ^ src) & FILTER_MASK); - continue; - } -#endif - - dst_buf[0*SUB_GROUP_SIZE + i] += popcount(wei.s0 ^ src); - dst_buf[1*SUB_GROUP_SIZE + i] += popcount(wei.s1 ^ src); - } - } - } - } - -#if EXCLUDE_PAD - -#endif - // Load data for fused operations (scales, biases, quantization thresholds, etc) -#if CUSTOM_FUSED_OPS - FUSED_OPS_PREPARE_DATA; -#endif - - BUFFER_TYPE dst[SUB_GROUP_SIZE*2]; - - __attribute__((opencl_unroll_hint(SUB_GROUP_SIZE*2))) - for (int i = 0; i < SUB_GROUP_SIZE*2; i++) - { -#if EXCLUDE_PAD - CONV_RESULT_TYPE res = TO_CONV_RESULT_TYPE(INPUT0_FEATURE_NUM*_sub_group_shuffle(real_ks, i%SUB_GROUP_SIZE) - 2*dst_buf[i]); -#else - CONV_RESULT_TYPE res = TO_CONV_RESULT_TYPE(INPUT0_FEATURE_NUM*FILTER_SIZE_Y*FILTER_SIZE_X - 2*dst_buf[i]); -#endif - -#if CUSTOM_FUSED_OPS - DO_ELTWISE_FUSED_OPS; - dst[i] = res; -#elif HAS_FUSED_OPS - FUSED_OPS; - dst[i] = FUSED_OPS_RESULT; -#else - dst[i] = res; -#endif - - } - -#if BINARY_PACKED_OUTPUT - int packed_out[SUB_GROUP_SIZE]; - -#if CUSTOM_FUSED_OPS - DO_CHANNEL_PACK_OPS; -#else - #error "BINARY_PACKED_OUTPUT should be true only if node has fused quantize with bin output" -#endif - - bool in_x = (x + lid) < OUTPUT_SIZE_X; - bool in_y = y < OUTPUT_SIZE_Y; - if (in_x && in_y) - output[dst_index + lid] = packed_out[lid]; - -#else - - for (int oc = 0; oc < 2; oc++) - { - for (int ow = 0; ow < SUB_GROUP_SIZE; ow++) - { - bool in_x = (x + ow) < OUTPUT_SIZE_X; - bool in_y = y < OUTPUT_SIZE_Y; - bool in_fm = f_block*OC_BLOCK_SIZE + oc*SUB_GROUP_SIZE + lid < OUTPUT_FEATURE_NUM; - if (in_x && in_y && in_fm) - { - output[dst_index + (oc*SUB_GROUP_SIZE + lid)*OUTPUT_FEATURE_PITCH + ow] = TO_OUTPUT_TYPE(dst[oc*SUB_GROUP_SIZE + ow]); - } - } - } - -#endif -} diff --git a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_ref.cl b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_ref.cl deleted file mode 100644 index d7d2c7363a9..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/binary_convolution_gpu_ref.cl +++ /dev/null @@ -1,111 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "include/batch_headers/fetch_data.cl" - -KERNEL(binary_convolution_ref)(const __global INPUT0_TYPE* input, - __global OUTPUT_TYPE* output, - const __global FILTER_TYPE* weights -#if HAS_FUSED_OPS_DECLS - , FUSED_OPS_DECLS -#endif -) -{ - const int b = get_global_id(0); - const int f = get_global_id(1); - const int yx = get_global_id(2); - const int y = yx / OUTPUT_SIZE_X; - const int x = yx % OUTPUT_SIZE_X; - - const int input_x = x * STRIDE_SIZE_X - PADDING_SIZE_X; - const int input_y = y * STRIDE_SIZE_Y - PADDING_SIZE_Y; - - const int output_index = OUTPUT_OFFSET - + b * OUTPUT_BATCH_PITCH - + f * OUTPUT_FEATURE_PITCH - + y * OUTPUT_Y_PITCH - + x * OUTPUT_X_PITCH; - - const int input_index = INPUT0_OFFSET - + b * INPUT0_FEATURE_NUM_PACKED*INPUT0_FEATURE_PITCH; - - const int weights_index = (f / OFM_BLOCK_SIZE) * INPUT0_FEATURE_NUM_PACKED*FILTER_SIZE_X*FILTER_SIZE_Y*OFM_BLOCK_SIZE - + (f % OFM_BLOCK_SIZE); -#if EXCLUDE_PAD - int ks = 0; -#endif - int res_popcnt = 0; - for (int icp = 0; icp < INPUT0_FEATURE_NUM_PACKED; icp++) - { - for (int kh = 0; kh < FILTER_SIZE_Y; kh++) - { - const int input_offset_y = input_y + kh * DILATION_SIZE_Y; - const bool zero_y = input_offset_y >= INPUT0_SIZE_Y || input_offset_y < 0; - - for (int kw = 0; kw < FILTER_SIZE_X; kw++) - { - const int input_offset_x = input_x + kw * DILATION_SIZE_X; - const bool zero_x = input_offset_x >= INPUT0_SIZE_X || input_offset_x < 0; - FILTER_TYPE wei = weights[weights_index + icp*OFM_BLOCK_SIZE*FILTER_SIZE_X*FILTER_SIZE_Y + - kh*FILTER_SIZE_X*OFM_BLOCK_SIZE + kw*OFM_BLOCK_SIZE]; -#if EXCLUDE_PAD - if (!zero_y && !zero_x) - { - INPUT0_TYPE src = input[input_index + - icp * INPUT0_FEATURE_PITCH + - input_offset_y*INPUT0_Y_PITCH + - input_offset_x*INPUT0_X_PITCH]; // 32 packed input channels - -#if LEFTOVERS - if (icp == INPUT0_FEATURE_NUM_PACKED - 1) - res_popcnt += popcount((src ^ wei) & LEFTOVERS_MASK); - else -#endif - res_popcnt += popcount(src ^ wei); - if (icp == 0) - ks++; - } -#else - if (zero_y || zero_x) - { -#if LEFTOVERS - if (icp == INPUT0_FEATURE_NUM_PACKED - 1) - res_popcnt += popcount((PAD_VALUE ^ wei) & LEFTOVERS_MASK); - else -#endif - res_popcnt += popcount(PAD_VALUE ^ wei); - } - else - { - INPUT0_TYPE src = input[input_index + - icp * INPUT0_FEATURE_PITCH + - input_offset_y*INPUT0_Y_PITCH + - input_offset_x*INPUT0_X_PITCH]; // 32 packed input channels -#if LEFTOVERS - if (icp == INPUT0_FEATURE_NUM_PACKED - 1) - res_popcnt += popcount((src ^ wei) & LEFTOVERS_MASK); - else -#endif - res_popcnt += popcount(src ^ wei); - - } -#endif - } - } - } - - -#if EXCLUDE_PAD - UNIT_TYPE res = TO_OUTPUT_TYPE(INPUT0_FEATURE_NUM*ks - 2*res_popcnt); -#else - UNIT_TYPE res = TO_OUTPUT_TYPE(INPUT0_FEATURE_NUM*FILTER_SIZE_X*FILTER_SIZE_Y - 2*res_popcnt); -#endif - -#if HAS_FUSED_OPS - FUSED_OPS; - res = FUSED_OPS_RESULT; -#endif - - output[output_index] = res; -} diff --git a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/quantize_gpu_ref.cl b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/quantize_gpu_ref.cl index 8941cf83fb3..7e89faba03a 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/quantize_gpu_ref.cl +++ b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/quantize_gpu_ref.cl @@ -52,42 +52,6 @@ KERNEL(quantize_ref)( const int v = ((vuwzyx / OUTPUT_SIZE_X) / OUTPUT_SIZE_Y) / OUTPUT_SIZE_Z / OUTPUT_SIZE_W / OUTPUT_SIZE_U; #endif -#if PACKED_BINARY_OUTPUT - const int output_offset = OUTPUT_OFFSET - + b*OUTPUT_FEATURE_NUM_PACKED*OUTPUT_FEATURE_PITCH - + of*OUTPUT_FEATURE_PITCH - + y*OUTPUT_Y_PITCH - + x*OUTPUT_X_PITCH; - - const int threshold_offset = INPUT1_OFFSET - + (b % INPUT1_BATCH_NUM)*INPUT1_BATCH_PITCH - + (y % INPUT1_SIZE_Y)*INPUT1_Y_PITCH - + (x % INPUT1_SIZE_X)*INPUT1_X_PITCH; - - OUTPUT_TYPE res = 0x00000000; -#if SINGLE_OUT_VAL - int high_bit = output_high[0] == UNIT_VAL_ONE ? 1 : 0; - int low_bit = output_low[0] == UNIT_VAL_ONE ? 1 : 0; -#endif - int limit = min((int)OC_BLOCK_SIZE, (int)INPUT0_FEATURE_NUM); - for (int f = 0; f < limit; f++) - { - UNIT_TYPE val = input[INPUT0_GET_INDEX(b, of*OC_BLOCK_SIZE + f, y, x)]; - UNIT_TYPE threshold = input_low[threshold_offset + ((of*OC_BLOCK_SIZE + f) % INPUT1_FEATURE_NUM)*INPUT1_FEATURE_PITCH]; -#if PER_CHANNEL_OUT_VAL - int high_bit = output_high[of*OC_BLOCK_SIZE + f] == UNIT_VAL_ONE ? 1 : 0; - int low_bit = output_low[of*OC_BLOCK_SIZE + f] == UNIT_VAL_ONE ? 1 : 0; -#endif - res |= (((val > threshold) ? high_bit : low_bit) << f); - } - - if (x >= OUTPUT_SIZE_X || y >= OUTPUT_SIZE_Y) - return; - - output[output_offset] = res; - -#else - #if INPUT0_DIMS == 8 const int input_offset = INPUT0_GET_INDEX(b, of, v, u, w, z, y, x); #elif INPUT0_DIMS == 7 @@ -195,6 +159,4 @@ KERNEL(quantize_ref)( * (UNIT_VAL_ONE / (LEVELS-1) * (output_high_val - output_low_val)) + output_low_val)); #endif } - -#endif } diff --git a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/reorder_data_binary.cl b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/reorder_data_binary.cl deleted file mode 100644 index 6b9cb200cbe..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/reorder_data_binary.cl +++ /dev/null @@ -1,89 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "include/reshape_dims.cl" -#include "include/batch_headers/fetch_data.cl" - - -#if !INPUT0_LAYOUT_BFYX && !INPUT0_LAYOUT_B_FS_YX_32FP -#error "Data binary reorder: unsupported input layout" -#endif - -#if !OUTPUT_LAYOUT_BFYX && !OUTPUT_LAYOUT_B_FS_YX_32FP -#error "Data binary reorder: unsupported output layout" -#endif - -#ifdef MEAN_SUBTRACT_IN_BUFFER -#error "Mean subtruction is not supported in binary reorder" -#endif - - -KERNEL (reorder_data_binary)(const __global INPUT_REORDER_TYPE* input, - __global OUTPUT_REORDER_TYPE* output) -{ - const uint b = get_global_id(0); - const uint f = get_global_id(1); - const uint y = ((uint)(get_global_id(2))) / INPUT0_SIZE_X; - const uint x = ((uint)(get_global_id(2))) % INPUT0_SIZE_X; - - -#if BINARY_INPUT && BINARY_OUTPUT - int input_index = INPUT0_OFFSET - + b * INPUT_PACKED_FEATURES_NUM * INPUT0_FEATURE_PITCH - + f * INPUT0_FEATURE_PITCH - + y * INPUT0_Y_PITCH - + x * INPUT0_X_PITCH; - int output_index = OUTPUT_OFFSET - + b * OUTPUT_PACKED_FEATURES_NUM * OUTPUT_FEATURE_PITCH - + f * OUTPUT_FEATURE_PITCH - + y * OUTPUT_Y_PITCH - + x * OUTPUT_X_PITCH; - - output[output_index] = ACTIVATION_FUNC_TYPED(OUTPUT_REORDER, TO_OUTPUT_REORDER_TYPE(input[input_index]), NL_M, NL_N); -#elif BINARY_OUTPUT - int output_index = OUTPUT_OFFSET - + b * OUTPUT_PACKED_FEATURES_NUM * OUTPUT_FEATURE_PITCH - + f * OUTPUT_FEATURE_PITCH - + y * OUTPUT_Y_PITCH - + x * OUTPUT_X_PITCH; - - OUTPUT_TYPE res = 0x00000000; - int limit = min((int)IFM_PACK_SIZE, (int)(INPUT0_FEATURE_NUM - f*IFM_PACK_SIZE)); - for (int c = 0; c < limit; c++) - { - // index of required bit - int input_index = INPUT0_OFFSET - + b * INPUT0_BATCH_PITCH - + (f * IFM_PACK_SIZE + c) * INPUT0_FEATURE_PITCH - + y * INPUT0_Y_PITCH - + x * INPUT0_X_PITCH; - - int bit = input[input_index] > UNIT_VAL_ZERO ? 1 : 0; - res |= (bit << c); - } - output[output_index] = ACTIVATION_FUNC_TYPED(OUTPUT_REORDER, TO_OUTPUT_REORDER_TYPE(res), NL_M, NL_N); -#elif BINARY_INPUT - int input_index = INPUT0_OFFSET - + b * INPUT_PACKED_FEATURES_NUM * INPUT0_FEATURE_PITCH - + f * INPUT0_FEATURE_PITCH - + y * INPUT0_Y_PITCH - + x * INPUT0_X_PITCH; - int res = input[input_index]; - int limit = min((int)IFM_PACK_SIZE, (int)(INPUT0_FEATURE_NUM - f*IFM_PACK_SIZE)); - for (int c = 0; c < limit; c++) - { - int output_index = OUTPUT_OFFSET - + b * OUTPUT_BATCH_PITCH - + (f*IFM_PACK_SIZE + c) * OUTPUT_FEATURE_PITCH - + y * OUTPUT_Y_PITCH - + x * OUTPUT_X_PITCH; - - int bit = (res >> c) & 0x00000001 > 0 ? 1 : -1; - output[output_index] = ACTIVATION_FUNC_TYPED(OUTPUT_REORDER, TO_OUTPUT_REORDER_TYPE(bit), NL_M, NL_N); - } -#else -#error "Binary reorder is used without binary tensors" -#endif - -} diff --git a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/reorder_weights_binary.cl b/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/reorder_weights_binary.cl deleted file mode 100644 index 63115f2fdbc..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/cl_kernels/reorder_weights_binary.cl +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "include/reshape_dims.cl" - -#define OFM_BLOCK_SIZE 32 -#define IFM_PACK_SIZE 32 - -// packed binary oihw to packed binary os_is_yx_osv32_isv32p -KERNEL (reorder_weights_binary)(const __global INPUT0_TYPE* input, __global OUTPUT_TYPE* output) -{ - const unsigned o = get_global_id(0); - const unsigned i = get_global_id(1); - const unsigned y = (uint)get_global_id(2) / OUTPUT_SIZE_X; - const unsigned x = (uint)get_global_id(2) % OUTPUT_SIZE_X; - - int output_index = OUTPUT_OFFSET - + (o % OFM_BLOCK_SIZE) - + (o / OFM_BLOCK_SIZE) * ((OUTPUT_IFM_NUM + IFM_PACK_SIZE - 1) / IFM_PACK_SIZE) * OUTPUT_SIZE_Y * OUTPUT_SIZE_X * OFM_BLOCK_SIZE - + i * OFM_BLOCK_SIZE * OUTPUT_IFM_PITCH - + y * OFM_BLOCK_SIZE * OUTPUT_Y_PITCH - + x * OFM_BLOCK_SIZE * OUTPUT_X_PITCH; - - OUTPUT_TYPE res = 0x00000000; - int limit = min((int)IFM_PACK_SIZE, (int)(INPUT0_IFM_NUM - i*IFM_PACK_SIZE)); - for (int c = 0; c < limit; c++) - { - // index of required bit - int input_index = INPUT0_OFFSET - + o * INPUT0_OFM_PITCH - + (i * IFM_PACK_SIZE + c) * INPUT0_IFM_PITCH - + y * INPUT0_Y_PITCH - + x * INPUT0_X_PITCH; - - const int bit = input_index % IFM_PACK_SIZE; - const int element = input_index / IFM_PACK_SIZE; - res |= ((input[element] & (1 << bit)) >> bit) << c; - } - - output[output_index] = res; -} diff --git a/src/plugins/intel_gpu/src/kernel_selector/common_tools.h b/src/plugins/intel_gpu/src/kernel_selector/common_tools.h index b9f3e69acf8..e9673ca2e59 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/common_tools.h +++ b/src/plugins/intel_gpu/src/kernel_selector/common_tools.h @@ -23,7 +23,6 @@ inline uint32_t BytesPerElement(Datatype dt) { case Datatype::F32: case Datatype::INT32: case Datatype::UINT32: - case Datatype::BINARY: return 4; case Datatype::INT64: return 8; @@ -40,7 +39,6 @@ inline uint32_t BytesPerElement(WeightsType wt) { case WeightsType::F16: return 2; case WeightsType::F32: - case WeightsType::BINARY: case WeightsType::INT32: return 4; default: diff --git a/src/plugins/intel_gpu/src/kernel_selector/common_types.h b/src/plugins/intel_gpu/src/kernel_selector/common_types.h index 6d84aa65f37..e884cddefde 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/common_types.h +++ b/src/plugins/intel_gpu/src/kernel_selector/common_types.h @@ -64,7 +64,6 @@ enum class KernelType { SLICE, STRIDED_SLICE, REVERSE_SEQUENCE, - BINARY_CONVOLUTION, QUANTIZE, LSTM_DYNAMIC_INPUT, LSTM_DYNAMIC_TIMELOOP, @@ -105,7 +104,6 @@ enum class KernelType { //////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// enum class Datatype { UNSUPPORTED, - BINARY, UINT4, INT4, INT8, @@ -124,7 +122,6 @@ enum class Datatype { //////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// enum class WeightsType { UNSUPPORTED, - BINARY, F16, F32, INT8, diff --git a/src/plugins/intel_gpu/src/kernel_selector/jitter.cpp b/src/plugins/intel_gpu/src/kernel_selector/jitter.cpp index ac079ea4487..9c94c9cc33e 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/jitter.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/jitter.cpp @@ -113,8 +113,6 @@ namespace kernel_selector { std::string toCLType(WeightsType wType) { switch (wType) { - case WeightsType::BINARY: - return GetTypeName(); case WeightsType::INT4: case WeightsType::INT8: return GetTypeName(); @@ -134,8 +132,6 @@ std::string toCLType(WeightsType wType) { std::string toCLType(Datatype dType) { switch (dType) { - case Datatype::BINARY: - return GetTypeName(); case Datatype::INT8: return GetTypeName(); case Datatype::UINT8: @@ -1435,7 +1431,6 @@ JitConstants MakeTypeJitConstants(Datatype dataType, const std::string& macroNam is_fp = false; break; case Datatype::UINT32: - case Datatype::BINARY: type = "uint"; max_val = "UINT_MAX"; min_val = "0"; @@ -1539,8 +1534,6 @@ JitConstants MakeTypeJitConstants(WeightsType weightsType, const std::string& ma return MakeTypeJitConstants(Datatype::INT4, macroName); case WeightsType::UINT4: return MakeTypeJitConstants(Datatype::UINT4, macroName); - case WeightsType::BINARY: - return MakeTypeJitConstants(Datatype::UINT32, macroName); case WeightsType::INT32: return MakeTypeJitConstants(Datatype::INT32, macroName); } diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_common.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_common.cpp index 193d589daba..6f8e9f621c6 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_common.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_common.cpp @@ -103,7 +103,6 @@ std::string toString(DataLayout l) { case kernel_selector::DataLayout::bs_f_bsv8__af8: return "BS_F_BSV8__AF8"; case kernel_selector::DataLayout::bs_f_bsv16__af8: return "BS_F_BSV16__AF8"; case kernel_selector::DataLayout::winograd_2x3_s1_data: return "WINOGRAD_2x3_S1_DATA"; - case kernel_selector::DataLayout::b_fs_yx_32fp: return "B_FS_YX_32FP"; case kernel_selector::DataLayout::bfzyx: return "BFZYX"; case kernel_selector::DataLayout::bzyxf: return "BZYXF"; case kernel_selector::DataLayout::fs_b_yx_fsv32: return "FS_B_YX_FSV32"; @@ -141,7 +140,6 @@ std::string toString(DataLayout l) { std::string toString(Datatype dType) { switch (dType) { - case Datatype::BINARY: return "BINARY"; case Datatype::UINT4: return "UINT4"; case Datatype::INT4: return "INT4"; case Datatype::INT8: return "INT8"; @@ -159,7 +157,6 @@ std::string toString(Datatype dType) { std::string toString(WeightsType wType) { switch (wType) { - case WeightsType::BINARY: return "BINARY"; case WeightsType::F16: return "F16"; case WeightsType::F32: return "F32"; case WeightsType::UINT4: return "UINT4"; @@ -186,8 +183,6 @@ std::string toString(KernelType kt) { case KernelType::ELTWISE: return "ELTWISE"; case KernelType::REORDER: return "REORDER"; case KernelType::SELECT: return "SELECT"; - case KernelType::BINARY_CONVOLUTION: - return "BINARY_CONVOLUTION"; case KernelType::NON_MAX_SUPPRESSION: return "NON_MAX_SUPPRESSION"; case KernelType::MATRIX_NMS: return "MATRIX_NMS"; @@ -361,7 +356,6 @@ std::string toString(WeightsLayout layout) { case WeightsLayout::os_is_yx_osv32_isv4: return "OS_IS_YX_OSV32_ISV4"; case WeightsLayout::os_is_zyx_osv32_isv4: return "OS_IS_ZYX_OSV32_ISV4"; case WeightsLayout::os_is_y_x8_osv8_isv4_swizzled_by_4: return "OS_IS_Y_X8_OSV8_ISV4_SWIZZLED_BY_4"; - case WeightsLayout::os_is_yx_osv32_isv32p: return "OS_IS_YX_OSV32_ISV32P"; case WeightsLayout::oizyx: return "OIZYX"; case WeightsLayout::iozyx: return "IOZYX"; case WeightsLayout::os_is_zyx_isv16_osv16: return "OS_IS_ZYX_ISV16_OSV16"; diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.cpp index d7b3002bb6f..f458b560b39 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.cpp @@ -80,9 +80,6 @@ void ParamsKey::EnableInputDataType(Datatype dt) { case Datatype::F32: key.inputType.val.F32 = 1; break; - case Datatype::BINARY: - key.inputType.val.binary = 1; - break; default: break; } @@ -125,9 +122,6 @@ void ParamsKey::EnableOutputDataType(Datatype dt) { case Datatype::F32: key.outputType.val.F32 = 1; break; - case Datatype::BINARY: - key.outputType.val.binary = 1; - break; default: break; } @@ -152,9 +146,6 @@ void ParamsKey::EnableInputWeightsType(WeightsType wt) { case WeightsType::UINT4: key.inputWeightsType.val.uint4 = 1; break; - case WeightsType::BINARY: - key.inputWeightsType.val.binary = 1; - break; case WeightsType::INT32: key.inputWeightsType.val.int32 = 1; default: @@ -181,9 +172,6 @@ void ParamsKey::EnableOutputWeightsType(WeightsType wt) { case WeightsType::UINT4: key.outputWeightsType.val.uint4 = 1; break; - case WeightsType::BINARY: - key.outputWeightsType.val.binary = 1; - break; case WeightsType::INT32: key.outputWeightsType.val.int32 = 1; default: diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.h b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.h index 8fb50c17166..c8ba581f51c 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.h +++ b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_params.h @@ -240,7 +240,6 @@ public: uint32_t cell : 1; } lstm_elt; struct quantize_t { - uint32_t packed_binary_output : 1; uint32_t scale_shift_opt : 1; } quantize; } dedicated; @@ -263,7 +262,6 @@ public: uint32_t int64 : 1; uint32_t F16 : 1; uint32_t F32 : 1; - uint32_t binary : 1; } val; uint32_t raw; } DataTypesKey; @@ -329,7 +327,6 @@ public: void EnableBilinearInterpolationPad() { key.restrict.val.dedicated.conv.bilinear_interpolation_pad = 1; } void EnableDeformableMask() { key.restrict.val.dedicated.conv.deformable_mask_enabled = 1; } - void EnableQuantizePackedBinaryOutput() { key.restrict.val.dedicated.quantize.packed_binary_output = 1; } void EnableQuantizeScaleShiftOpt() { key.restrict.val.dedicated.quantize.scale_shift_opt = 1; } void EnableWinogradReorder() { key.restrict.val.dedicated.reorder.winograd = 1; } @@ -606,8 +603,6 @@ struct dep_info { // - KernelBase::MakeFusedOpsDeclsJitConstants that creates arguments for kernel declaration and macro for all tensors used in // a fused op (requires FusedOpsConfiguration instance). // - fused_operation_desc contains a bunch of methods to generate variable/pointer names, type conversions, data loads -// If you need an example of custom code generation for fused ops, check BinaryConvolutionKernelGeneric::GetFusedPrimitivesJitConstants -// method in binary_convolution_kernel_generic.cpp. struct fused_operation_desc { std::shared_ptr op_params; int32_t dep_idx_start; diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_utils.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_utils.cpp index bb3eb06e32b..5530619fcf7 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_utils.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernel_selector_utils.cpp @@ -24,8 +24,6 @@ static WeightsType DataTypeToWeightsType(Datatype t) { return WeightsType::F16; case Datatype::F32: return WeightsType::F32; - case Datatype::BINARY: - return WeightsType::BINARY; case Datatype::INT32: return WeightsType::INT32; default: @@ -468,7 +466,6 @@ bool CheckInputsOutputNoPitchSameDims(const base_params& params) { {DataLayout::b_fs_yx_fsv8, {1, 8}}, {DataLayout::b_fs_zyx_fsv8, {1, 8}}, {DataLayout::fs_b_yx_fsv32, {1, 32}}, - {DataLayout::b_fs_yx_32fp, {1, 32}}, {DataLayout::bs_fs_yx_bsv32_fsv16, {32, 16}}, {DataLayout::bs_fs_zyx_bsv32_fsv16, {32, 16}}, {DataLayout::bs_fs_yx_bsv32_fsv32, {32, 32}}, diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1.cpp deleted file mode 100644 index 12d9593d9a8..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1.cpp +++ /dev/null @@ -1,237 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include -#include "binary_convolution_kernel_1x1.h" -#include -#include -#include - -namespace kernel_selector { - -static const int sub_group_size = 16; -static const int ic_pack_size = 32; -static const int xy_block_size = 16; - -ParamsKey BinaryConvolutionKernel1x1::GetSupportedKey() const { - ParamsKey k; - k.EnableInputDataType(Datatype::BINARY); - k.EnableInputWeightsType(WeightsType::BINARY); - k.EnableOutputDataType(Datatype::F16); - k.EnableOutputDataType(Datatype::F32); - k.EnableOutputDataType(Datatype::INT32); - k.EnableOutputDataType(Datatype::BINARY); - k.EnableInputLayout(DataLayout::b_fs_yx_32fp); - k.EnableOutputLayout(DataLayout::bfyx); - k.EnableOutputLayout(DataLayout::b_fs_yx_32fp); - k.EnableTensorOffset(); - k.EnableTensorPitches(); - k.EnableNonBiasTerm(); - k.EnableBatching(); - k.EnableDifferentTypes(); - return k; -} - -DeviceFeaturesKey BinaryConvolutionKernel1x1::get_required_device_features_key(const Params& params, const optional_params& /*options*/) const { - DeviceFeaturesKey k; - k.requires_subgroup_shuffle(); - k.requires_blocked_read_write(); - - return k; -} - -BinaryConvolutionKernelBase::DispatchData BinaryConvolutionKernel1x1::SetDefault(const binary_convolution_params& params, int) const { - DispatchData dispatchData = BinaryConvolutionKernelBase::SetDefault(params); - - const auto& out = params.outputs[0]; - - auto x = out.X().v; - auto y = out.Y().v; - auto f = out.Feature().v; - auto b = out.Batch().v; - - dispatchData.gws[0] = Align(x * y, sub_group_size); - dispatchData.gws[1] = CeilDiv(f, 2 * sub_group_size); // 1 WI calcs 32 OC - dispatchData.gws[2] = b; - - dispatchData.lws[0] = sub_group_size; - dispatchData.lws[1] = 1; - dispatchData.lws[2] = 1; - - return dispatchData; -} - -KernelsPriority BinaryConvolutionKernel1x1::GetKernelsPriority(const Params& /*params*/, const optional_params& /*options*/) const { - return FORCE_PRIORITY_1; -} - -bool BinaryConvolutionKernel1x1::Validate(const Params& p, const optional_params& o) const { - if (!BinaryConvolutionKernelBase::Validate(p, o) || !ConvolutionBinaryCheckInput(p, o)) - return false; - - const auto& params = static_cast(p); - - const auto& input = params.inputs[0]; - const auto& output = params.outputs[0]; - - const bool bOutputSizes = output.X().v != input.X().v || output.Y().v != input.Y().v; - const bool bFilterSize = params.filterSize.x != 1 || params.filterSize.y != 1; - const bool bStride = params.stride.x != 1 || params.stride.y != 1; - const bool bGroups = params.groups > 1; - - if (bOutputSizes || bFilterSize || bStride || bGroups) - return false; - - return true; -} - -JitConstants BinaryConvolutionKernel1x1::GetJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const { - auto jit = Parent::GetJitConstants(params, dispatchData); - - jit.AddConstant(MakeJitConstant("SUB_GROUP_SIZE", sub_group_size)); - jit.AddConstant(MakeJitConstant("INPUT0_FEATURE_NUM_PACKED", CeilDiv(params.inputs[0].Feature().v, ic_pack_size))); - jit.AddConstant(MakeJitConstant("OUTPUT_FEATURE_NUM_PACKED", CeilDiv(params.outputs[0].Feature().v, ic_pack_size))); - jit.AddConstant(MakeJitConstant("PADDED_INPUT", params.inputs[0].X().pad.Total() != 0)); - jit.AddConstant(MakeJitConstant("PADDED_OUTPUT", params.outputs[0].X().pad.Total() != 0)); - jit.AddConstant(MakeJitConstant("XY_BLOCK_SIZE", xy_block_size)); - if (params.inputs[0].Feature().v % ic_pack_size) { - jit.AddConstant(MakeJitConstant("LEFTOVERS_IC", params.inputs[0].Feature().v % ic_pack_size)); - jit.AddConstant(MakeJitConstant("FILTER_MASK", - (0xFFFFFFFF >> (ic_pack_size - params.inputs[0].Feature().v % ic_pack_size)))); - } - - if (params.outputs[0].GetDType() == Datatype::BINARY) { - jit.AddConstant(MakeJitConstant("BINARY_PACKED_OUTPUT", 1)); - } - - return jit; -} - -JitConstants BinaryConvolutionKernel1x1::GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& /*kd*/) const { - JitConstants jit = {}; - - auto input_dt = GetUnitType(params); - FusedOpsConfiguration conf = {"", {"b", "(f_block*16 + oc)", "y", "x"}, "res", input_dt, 1 }; - jit.Merge(MakeFusedOpsDeclsJitConstants(params, {conf})); - - size_t op_id = 0; - std::string input_decls = ""; - std::string eltwise_fused_ops = ""; - std::string prepare_data = ""; - for (auto& fused_dep : params.fused_ops) { - auto fused_dep_codegen = FusedOpsCodeGenerator(fused_dep); - auto get_aligned_load2 = [&](std::string ptr, std::string byte_offset) -> std::string { - if (fused_dep.tensors[0].GetDType() == Datatype::F32) - return "(_sub_group_block_read2((const __global uint*)(" + ptr + ") + (" + byte_offset + ")))"; - else - return "(_sub_group_block_read_us2((const __global ushort*)(" + ptr + ") + (" + byte_offset + - ")))"; - }; - - auto get_shuffle = [&](std::string var, std::string lid) -> std::string { - return "(_sub_group_shuffle(" + var + ", " + lid + "))"; - }; - - std::string data_type = fused_dep_codegen.GetInputTypeName(0, 1); - std::string vec_data_type = fused_dep_codegen.GetInputTypeName(0, 2); - std::string sc = "sc" + toCodeString(op_id); - std::string sh = "sh" + toCodeString(op_id); - std::string e_add = "e_add" + toCodeString(op_id); - std::string e_mul = "e_mul" + toCodeString(op_id); - - switch (fused_dep.GetType()) { - case KernelType::QUANTIZE: { - std::string var_name_in = fused_dep_codegen.GetInputVarName(0); - std::string var_name_out = fused_dep_codegen.GetInputVarName(3); - std::string cast_type_vec = (fused_dep.tensors[0].GetDType() == Datatype::F32) ? "as_float2" : "as_half2"; - std::string cast_type = (fused_dep.tensors[0].GetDType() == Datatype::F32) ? "as_float" : "as_half"; - - prepare_data += "\\\n\tint packed_res = 0;"; - if (fused_dep.tensors[0].Feature().v == params.outputs[0].Feature().v) { - prepare_data += "\\\n\t" + vec_data_type + " " + var_name_in + " = " + cast_type_vec + - get_aligned_load2(fused_dep_codegen.GetInputPtrName(0), "f_block*OC_BLOCK_SIZE") + ";"; - eltwise_fused_ops += "\\\n\t" + data_type + " thresh = (oc < 16) ? " + get_shuffle(var_name_in + ".s0", "oc") + - " : " + get_shuffle(var_name_in + ".s1", "oc") + ";"; - } else { - prepare_data += "\\\n\t" + data_type + " " + var_name_in + " = " + cast_type + - + "(" + fused_dep_codegen.GetInputPtrName(0) + "[0]);"; - eltwise_fused_ops += "\\\n\t" + data_type + " thresh = " + var_name_in + ";"; - } - - - if (fused_dep.tensors[2].Feature().v == params.outputs[0].Feature().v) { - // Per-channel output value - prepare_data += "\\\n\t" + vec_data_type + " " + var_name_out + " = " + cast_type_vec + - get_aligned_load2(fused_dep_codegen.GetInputPtrName(3), "f_block*OC_BLOCK_SIZE") + ";"; - eltwise_fused_ops +="\\\n\t" + data_type + " out_val = (oc < 16) ? " + get_shuffle(var_name_out + ".s0", "oc") + - " : " + get_shuffle(var_name_out + ".s1", "oc") + ";"; - } else { - // Per-tensor output value - prepare_data += "\\\n\t" + data_type + " " + var_name_out + " = " + cast_type + - + "(" + fused_dep_codegen.GetInputPtrName(3) + "[0]);"; - eltwise_fused_ops += "\\\n\t" + data_type + " out_val = " + var_name_out + ";"; - } - eltwise_fused_ops += "\\\n\tif (out_val == 1) "; - eltwise_fused_ops += "\\\n\t\tpacked_res |= (res > thresh) << oc;"; - eltwise_fused_ops += "\\\n\telse "; - eltwise_fused_ops += "\\\n\t\tpacked_res |= (res <= thresh) << oc;"; - - break; - } - - case KernelType::ACTIVATION: { - auto p = fused_dep.GetOpParams(); - base_activation_params activation = p->param; - if (activation.function != ActivationFunction::NONE) { - auto suffix = "_FUSED_OP" + toCodeString(op_id); - - jit.Merge(MakeActivationJitConstants(activation, fused_dep.output_tensor.GetDType(), suffix)); - eltwise_fused_ops += "\\\n\tres = ACTIVATION" + suffix + "((OUTPUT_TYPE)res, ACTIVATION_PARAMS" + suffix + ");"; - } - - break; - } - - case KernelType::ELTWISE: { - std::string cast_type = (fused_dep.tensors[0].GetDType() == Datatype::F32) ? "as_float2" : "as_half2"; - std::string var_name = fused_dep_codegen.GetInputVarName(0); - prepare_data += "\\\n\t" + vec_data_type + " " + var_name + " = " + cast_type + - get_aligned_load2(fused_dep_codegen.GetInputPtrName(0), "f_block*OC_BLOCK_SIZE") + ";"; - - auto eltwise_p = std::dynamic_pointer_cast(fused_dep.op_params); - - if (eltwise_p->mode == EltwiseMode::ADD) { - eltwise_fused_ops += "\\\n\t" + data_type + " " + e_add + " = (oc < 16) ? " + - get_shuffle(var_name + ".s0", "oc") + " : " + get_shuffle(var_name + ".s1", "oc") + ";"; - eltwise_fused_ops += "\\\n\tres = res+" + e_add + ";"; - } else if (eltwise_p->mode == EltwiseMode::MUL) { - eltwise_fused_ops += "\\\n\t" + data_type + " " + e_mul + " = (oc < 16) ? " + - get_shuffle(var_name + ".s0", "oc") + " : " + get_shuffle(var_name + ".s1", "oc") + ";"; - eltwise_fused_ops += "\\\n\tres = res*" + e_mul + ";"; - } else { - throw std::invalid_argument("Not supported eltwise fusing op in binary_convolution_1x1 kernel: " + params.layerID); - } - - break; - } - - default: - throw std::invalid_argument("Invalid fused op in binary_convolution_1x1 kernel: " + params.layerID); - } - - op_id++; - } - jit.AddConstant(MakeJitConstant("DO_ELTWISE_FUSED_OPS", eltwise_fused_ops)); - jit.AddConstant(MakeJitConstant("FUSED_OPS_PREPARE_DATA", prepare_data)); - jit.AddConstant(MakeJitConstant("CUSTOM_FUSED_OPS", true)); - - return jit; -} - -KernelsData BinaryConvolutionKernel1x1::GetKernelsData(const Params& params, const optional_params& options) const { - return GetTunedKernelsDataByIndex(params, options); -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1.h deleted file mode 100644 index 03e30737e6b..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1.h +++ /dev/null @@ -1,34 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "binary_convolution_kernel_base.h" -#include - -namespace kernel_selector { - -class BinaryConvolutionKernel1x1 : public BinaryConvolutionKernelBase { -public: - using Parent = BinaryConvolutionKernelBase; - - BinaryConvolutionKernel1x1() : BinaryConvolutionKernelBase("binary_convolution_gpu_1x1") {} - virtual ~BinaryConvolutionKernel1x1() {} - - KernelsData GetKernelsData(const Params& params, const optional_params& options) const override; - KernelsPriority GetKernelsPriority(const Params& params, const optional_params& options) const override; - ParamsKey GetSupportedKey() const override; - DeviceFeaturesKey get_required_device_features_key(const Params& params, const optional_params& /*options*/) const override; - -protected: - WeightsLayout GetPreferredWeightLayout(const binary_convolution_params &) const override { - return WeightsLayout::os_is_yx_osv32_isv32p; - } - JitConstants GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const override; - bool Validate(const Params& p, const optional_params& o) const override; - DispatchData SetDefault(const binary_convolution_params& arg, int autoTuneIndex = -1) const override; - JitConstants GetJitConstants(const binary_convolution_params& params, const DispatchData& dispatchData) const override; -}; -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1_b_fs_yx_fsv16.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1_b_fs_yx_fsv16.cpp deleted file mode 100644 index 282e97f5b4a..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1_b_fs_yx_fsv16.cpp +++ /dev/null @@ -1,200 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include -#include "binary_convolution_kernel_1x1_b_fs_yx_fsv16.h" -#include -#include -#include - -namespace kernel_selector { - -static const int sub_group_size = 16; -static const int ic_pack_size = 32; -static const int xy_block_size = 16; - -ParamsKey BinaryConvolutionKernel1x1_b_fs_yx_fsv16::GetSupportedKey() const { - ParamsKey k; - k.EnableInputDataType(Datatype::BINARY); - k.EnableInputWeightsType(WeightsType::BINARY); - k.EnableOutputDataType(Datatype::F16); - k.EnableOutputDataType(Datatype::F32); - k.EnableOutputDataType(Datatype::UINT8); - k.EnableOutputDataType(Datatype::INT8); - k.EnableOutputDataType(Datatype::INT32); - k.EnableInputLayout(DataLayout::b_fs_yx_32fp); - k.EnableOutputLayout(DataLayout::b_fs_yx_fsv16); - k.EnableTensorOffset(); - k.EnableTensorPitches(); - k.EnableNonBiasTerm(); - k.EnableBatching(); - k.EnableDifferentTypes(); - return k; -} - -DeviceFeaturesKey BinaryConvolutionKernel1x1_b_fs_yx_fsv16::get_required_device_features_key(const Params& params, const optional_params& /*options*/) const { - DeviceFeaturesKey k; - k.requires_subgroup_shuffle(); - k.requires_blocked_read_write(); - - return k; -} - -BinaryConvolutionKernelBase::DispatchData BinaryConvolutionKernel1x1_b_fs_yx_fsv16::SetDefault( - const binary_convolution_params& params, - int) const { - DispatchData dispatchData = BinaryConvolutionKernelBase::SetDefault(params); - - const auto& out = params.outputs[0]; - - auto x = out.X().v; - auto y = out.Y().v; - auto f = out.Feature().v; - auto b = out.Batch().v; - - dispatchData.gws[0] = Align(x * y, sub_group_size); - dispatchData.gws[1] = CeilDiv(f, sub_group_size); // 1 WI calcs 16 OC - dispatchData.gws[2] = b; - - dispatchData.lws = { static_cast(sub_group_size), 1, 1 }; - - return dispatchData; -} - -KernelsPriority BinaryConvolutionKernel1x1_b_fs_yx_fsv16::GetKernelsPriority(const Params& /*params*/, const optional_params& /*options*/) const { - return FORCE_PRIORITY_1; -} - -bool BinaryConvolutionKernel1x1_b_fs_yx_fsv16::Validate(const Params& p, const optional_params& o) const { - if (!BinaryConvolutionKernelBase::Validate(p, o) || !ConvolutionBinaryCheckInput(p, o)) - return false; - - const auto& params = static_cast(p); - - const auto& input = params.inputs[0]; - const auto& output = params.outputs[0]; - - const bool bOutputSizes = output.X().v != input.X().v || output.Y().v != input.Y().v; - const bool bFilterSize = params.filterSize.x != 1 || params.filterSize.y != 1; - const bool bStride = params.stride.x != 1 || params.stride.y != 1; - const bool bGroups = params.groups > 1; - - if (bOutputSizes || bFilterSize || bStride || bGroups) - return false; - - return true; -} - -JitConstants BinaryConvolutionKernel1x1_b_fs_yx_fsv16::GetJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const { - auto jit = Parent::GetJitConstants(params, dispatchData); - - jit.AddConstant(MakeJitConstant("SUB_GROUP_SIZE", sub_group_size)); - jit.AddConstant(MakeJitConstant("INPUT0_FEATURE_NUM_PACKED", CeilDiv(params.inputs[0].Feature().v, ic_pack_size))); - jit.AddConstant(MakeJitConstant("OUTPUT_FEATURE_NUM_PACKED", CeilDiv(params.outputs[0].Feature().v, ic_pack_size))); - jit.AddConstant(MakeJitConstant("PADDED_INPUT", params.inputs[0].X().pad.Total() != 0)); - jit.AddConstant(MakeJitConstant("PADDED_OUTPUT", params.outputs[0].X().pad.Total() != 0)); - jit.AddConstant(MakeJitConstant("XY_BLOCK_SIZE", xy_block_size)); - if (params.inputs[0].Feature().v % ic_pack_size) { - jit.AddConstant(MakeJitConstant("LEFTOVERS_IC", params.inputs[0].Feature().v % ic_pack_size)); - jit.AddConstant(MakeJitConstant("FILTER_MASK", - (0xFFFFFFFF >> (ic_pack_size - params.inputs[0].Feature().v % ic_pack_size)))); - } - - if (params.outputs[0].Feature().v % 32 != 0) { - jit.AddConstant(MakeJitConstant("LEFTOVERS_OC", true)); - } - - if (params.outputs[0].GetDType() == Datatype::BINARY) { - jit.AddConstant(MakeJitConstant("BINARY_PACKED_OUTPUT", 1)); - } - - return jit; -} - -JitConstants BinaryConvolutionKernel1x1_b_fs_yx_fsv16::GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& /*kd*/) const { - JitConstants jit = {}; - - auto input_dt = GetUnitType(params); - FusedOpsConfiguration conf = {"", {"b", "(f_block*16 + oc)", "y", "x"}, "res", input_dt, 1 }; - jit.Merge(MakeFusedOpsDeclsJitConstants(params, {conf})); - - size_t op_id = 0; - std::string input_decls = ""; - std::string eltwise_fused_ops = ""; - std::string prepare_data = ""; - for (auto& fused_dep : params.fused_ops) { - auto fused_dep_codegen = FusedOpsCodeGenerator(fused_dep); - - auto get_aligned_load = [&](std::string ptr, std::string byte_offset) -> std::string { - if (fused_dep.tensors[0].GetDType() == Datatype::F32) - return "(_sub_group_block_read((const __global uint*)(" + ptr + ") + (" + byte_offset + ")))"; - else - return "(_sub_group_block_read_us((const __global ushort*)(" + ptr + ") + (" + byte_offset + - ")))"; - }; - - auto get_shuffle = [&](std::string var, std::string lid) -> std::string { - return "(_sub_group_shuffle(" + var + ", " + lid + "))"; - }; - - std::string data_type = fused_dep_codegen.GetInputTypeName(0, 1); - std::string vec_data_type = fused_dep_codegen.GetInputTypeName(0, 1); - std::string sc = "sc" + toCodeString(op_id); - std::string e_add = "e_add" + toCodeString(op_id); - std::string e_mul = "e_mul" + toCodeString(op_id); - - switch (fused_dep.GetType()) { - case KernelType::ACTIVATION: { - auto p = fused_dep.GetOpParams(); - base_activation_params activation = p->param; - if (activation.function != ActivationFunction::NONE) { - auto suffix = "_FUSED_OP" + toCodeString(op_id); - - jit.Merge(MakeActivationJitConstants(activation, fused_dep.output_tensor.GetDType(), suffix)); - eltwise_fused_ops += "\\\n\tres = ACTIVATION" + suffix + "((OUTPUT_TYPE)res, ACTIVATION_PARAMS" + suffix + ");"; - } - - break; - } - - case KernelType::ELTWISE: { - std::string cast_type = (fused_dep.tensors[0].GetDType() == Datatype::F32) ? "as_float" : "as_half"; - std::string var_name = fused_dep_codegen.GetInputVarName(0); - prepare_data += "\\\n\t" + vec_data_type + " " + var_name + " = " + cast_type + - get_aligned_load(fused_dep_codegen.GetInputPtrName(0), "f_block*OC_BLOCK_SIZE") + ";"; - - auto eltwise_p = std::dynamic_pointer_cast(fused_dep.op_params); - - if (eltwise_p->mode == EltwiseMode::ADD) { - eltwise_fused_ops += "\\\n\t" + data_type + " " + e_add + " = " + get_shuffle(var_name, "oc") + ";"; - eltwise_fused_ops += "\\\n\tres = res+" + var_name + ";"; - } else if (eltwise_p->mode == EltwiseMode::MUL) { - eltwise_fused_ops += "\\\n\t" + data_type + " " + e_mul + " = " + get_shuffle(var_name, "oc") + ";"; - eltwise_fused_ops += "\\\n\tres = res*" + var_name + ";"; - } else { - throw std::invalid_argument("Not supported eltwise fusing op in binary_convolution_1x1_fsv16 kernel: " + params.layerID); - } - - break; - } - - default: - throw std::invalid_argument("Invalid fused op in binary_convolution_1x1_fsv16 kernel: " + params.layerID); - } - - op_id++; - } - jit.AddConstant(MakeJitConstant("DO_ELTWISE_FUSED_OPS", eltwise_fused_ops)); - jit.AddConstant(MakeJitConstant("FUSED_OPS_PREPARE_DATA", prepare_data)); - jit.AddConstant(MakeJitConstant("CUSTOM_FUSED_OPS", true)); - - return jit; -} - -KernelsData BinaryConvolutionKernel1x1_b_fs_yx_fsv16::GetKernelsData(const Params& params, const optional_params& options) const { - return GetTunedKernelsDataByIndex(params, options); -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1_b_fs_yx_fsv16.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1_b_fs_yx_fsv16.h deleted file mode 100644 index d7eb023f5b1..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_1x1_b_fs_yx_fsv16.h +++ /dev/null @@ -1,34 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "binary_convolution_kernel_base.h" -#include - -namespace kernel_selector { - -class BinaryConvolutionKernel1x1_b_fs_yx_fsv16 : public BinaryConvolutionKernelBase { -public: - using Parent = BinaryConvolutionKernelBase; - - BinaryConvolutionKernel1x1_b_fs_yx_fsv16() : BinaryConvolutionKernelBase("binary_convolution_gpu_1x1_b_fs_yx_fsv16") {} - virtual ~BinaryConvolutionKernel1x1_b_fs_yx_fsv16() {} - - KernelsData GetKernelsData(const Params& params, const optional_params& options) const override; - KernelsPriority GetKernelsPriority(const Params& params, const optional_params& options) const override; - ParamsKey GetSupportedKey() const override; - DeviceFeaturesKey get_required_device_features_key(const Params& params, const optional_params& /*options*/) const override; - -protected: - WeightsLayout GetPreferredWeightLayout(const binary_convolution_params &) const override { - return WeightsLayout::os_is_yx_osv32_isv32p; - } - JitConstants GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const override; - bool Validate(const Params& p, const optional_params& o) const override; - DispatchData SetDefault(const binary_convolution_params& arg, int autoTuneIndex = -1) const override; - JitConstants GetJitConstants(const binary_convolution_params& params, const DispatchData& dispatchData) const override; -}; -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_base.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_base.cpp deleted file mode 100644 index acc55df224f..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_base.cpp +++ /dev/null @@ -1,272 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include -#include "binary_convolution_kernel_base.h" -#include "kernel_selector_utils.h" -#include "common_tools.h" -#include -#include -#include - -namespace kernel_selector { -bool BinaryConvolutionKernelBase::Validate(const Params& p, const optional_params& o) const { - if (p.GetType() != KernelType::BINARY_CONVOLUTION || o.GetType() != KernelType::BINARY_CONVOLUTION) { - return false; - } - - const binary_convolution_params& params = static_cast(p); - const binary_convolution_optional_params& optParams = static_cast(o); - - bool bSupportedWeightsLayout = params.weights.GetLayout() == GetPreferredWeightLayout(params); - - const bool bWeightsOK = bSupportedWeightsLayout || optParams.allowStaticInputReordering; - - if (!bWeightsOK) { - return false; - } - - return true; -} - -JitConstants BinaryConvolutionKernelBase::GetJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const { - JitConstants jit = WeightBiasKernelBase::GetJitConstants(params); - jit.Merge(GetFusedPrimitivesJitConstants(params, dispatchData)); - - jit.AddConstants({ - MakeJitConstant("STRIDE", params.stride), - MakeJitConstant("PADDING", params.padding), - MakeJitConstant("DILATION", params.dilation), - }); - - jit.Merge(MakeTypeJitConstants(params.out_dt, "CONV_RESULT")); - - return jit; -} - -JitConstants BinaryConvolutionKernelBase::GetFusedPrimitivesJitConstants(const binary_convolution_params& /*params*/, - const DispatchData& /*kd*/) const { - return {}; -} - -bool BinaryConvolutionKernelBase::CheckWorkGroups(const BinaryConvolutionKernelBase::DispatchData& dispatchData) { - if (dispatchData.gws.size() != 3 || dispatchData.lws.size() != 3) - return false; - - for (size_t i = 0; i < dispatchData.gws.size(); i++) { - if (dispatchData.gws[i] == 0 || dispatchData.lws[i] == 0) - return false; - if ((dispatchData.gws[i] % dispatchData.lws[i]) != 0) - return false; - } - - return true; -} - -BinaryConvolutionKernelBase::DispatchData BinaryConvolutionKernelBase::SetDefault(const binary_convolution_params& params, - int) const { - DispatchData dispatchData; - auto in_layout = params.inputs[0].GetLayout(); - auto out_layout = params.outputs[0].GetLayout(); - std::vector> dims_by_gws; - - const auto& out = params.outputs[0]; - std::vector global; - if (out_layout == DataLayout::bfyx || out_layout == DataLayout::byxf) { - global = {out.X().v, out.Y().v, out.Feature().v * out.Batch().v}; - dims_by_gws = {{Tensor::DataChannelName::X}, - {Tensor::DataChannelName::Y}, - {Tensor::DataChannelName::FEATURE, Tensor::DataChannelName::BATCH}}; - } else { - global = {out.Feature().v * out.Batch().v, out.X().v, out.Y().v}; - dims_by_gws = {{Tensor::DataChannelName::FEATURE, Tensor::DataChannelName::BATCH}, - {Tensor::DataChannelName::X}, - {Tensor::DataChannelName::Y}}; - } - - auto local = GetOptimalLocalWorkGroupSizes(global, params.engineInfo, in_layout, out_layout, dims_by_gws); - - dispatchData.gws = global; - dispatchData.lws = local; - - dispatchData.cldnnStyle.blockWidth = 1; - dispatchData.cldnnStyle.blockHeight = 1; - dispatchData.cldnnStyle.prefetch = 0; - dispatchData.cldnnStyle.inputBlockArraySize = 0; - dispatchData.cldnnStyle.inputBlockWidth = 0; - - dispatchData.gemmStyle.globalWorkSizeDX = 1; - dispatchData.gemmStyle.globalWorkSizeDY = 1; - dispatchData.gemmStyle.globalWorkSizeDZ = 1; - dispatchData.gemmStyle.subBlockDimK = 1; - dispatchData.gemmStyle.subBlockDimM = 0; - dispatchData.gemmStyle.subBlockDimN = 0; - return dispatchData; -} - -KernelsData BinaryConvolutionKernelBase::GetCommonKernelsData(const Params& params, - const optional_params& options, - const std::string exeMode, - int autoTuneIndex) const { - if (!Validate(params, options)) { - return {}; - } - - KernelData kd = KernelData::Default(params); - binary_convolution_params& newParams = *static_cast(kd.params.get()); - - if (NeedPaddedInput()) { - kd.reorderInput = ConvolutionBinaryUpdateInputParams(newParams); - } - DispatchData dispatchData = SetDefault(newParams, autoTuneIndex); - - if (!CheckWorkGroups(dispatchData)) { - // Internal Error - wrong calculation of global/local work group sizes - return {}; - } - - bool succeed = UpdateWeightsParams(newParams, - options, - GetPreferredWeightLayout(newParams), - kd.weightsReorderParams, - GetSupportedKey()); - - if (!succeed) { - return {}; - } - - auto finalKernelName = GetKernelName(newParams); - auto cldnnJit = GetJitConstants(newParams, dispatchData); - auto entryPoint = GetEntryPoint(finalKernelName, newParams.layerID, params, options); - auto jit = CreateJit(finalKernelName, cldnnJit, entryPoint); - - auto& kernel = kd.kernels[0]; - uint32_t fused_deps_total = 0; - for (auto& fused_dep : newParams.fused_ops) { - for (int i = 0; i < static_cast(fused_dep.dep_size); i++) { - kernel.params.arguments.push_back({ArgumentDescriptor::Types::INPUT_OF_FUSED_PRIMITIVE, fused_deps_total}); - fused_deps_total++; - } - } - - FillCLKernelData(kernel, - dispatchData, - params.engineInfo, - finalKernelName, - jit, - entryPoint, - exeMode, - true, - !newParams.bias.empty(), - 1, - fused_deps_total); - - kd.autoTuneIndex = autoTuneIndex; - - return {kd}; -} - -bool CheckConvolutionBinaryPaddedInputDesc(const binary_convolution_params& params, const DataTensor& reqDesc) { - assert(params.inputs.size() == 1); - - bool properPadding = reqDesc.X().pad.before <= params.inputs[0].X().pad.before && - reqDesc.Y().pad.before <= params.inputs[0].Y().pad.before && - reqDesc.Feature().pad.before <= params.inputs[0].Feature().pad.before && - reqDesc.Batch().pad.before <= params.inputs[0].Batch().pad.before; - - properPadding &= reqDesc.X().pad.after <= params.inputs[0].X().pad.after && - reqDesc.Y().pad.after <= params.inputs[0].Y().pad.after && - reqDesc.Feature().pad.after <= params.inputs[0].Feature().pad.after && - reqDesc.Batch().pad.after <= params.inputs[0].Batch().pad.after; - - return properPadding; -} - -static DataTensor GetConvolutionBFYXPaddedTensor(const binary_convolution_params& cp) { - assert(cp.inputs.size() == 1); - assert(cp.inputs[0].GetDims().size() == 4U); - - DataTensor t = cp.inputs[0]; - std::vector pad{{0, 0}, {0, 0}, {0, 0}, {0, 0}}; - - pad[0].before = cp.padding.x; - pad[1].before = cp.padding.y; - - const auto inputLimitX = (cp.outputs[0].X().v - 1) * cp.stride.x + (cp.filterSize.x - 1) * cp.dilation.x + 1; - const auto inputLimitY = (cp.outputs[0].Y().v - 1) * cp.stride.y + (cp.filterSize.y - 1) * cp.dilation.y + 1; - - pad[0].after = (size_t)std::max(static_cast(inputLimitX) - static_cast(t.X().v) - static_cast(pad[0].before), static_cast(0)); - pad[1].after = (size_t)std::max(static_cast(inputLimitY) - static_cast(t.Y().v) - static_cast(pad[1].before), static_cast(0)); - - Tensor::NDims dims(4); - const Tensor::NDims& orgDims = cp.inputs[0].GetDims(); - size_t pitch = 1; - for (size_t i = 0; i < dims.size(); i++) { - dims[i].pad = pad[i]; - dims[i].v = orgDims[i].v; - dims[i].pitch = pitch; - pitch *= dims[i].LogicalDimPadded(); - } - - return {dims, t.GetDType(), t.GetLayout()}; -} - -bool ConvolutionBinaryCheckInput(const Params& p, const optional_params& o) { - const binary_convolution_params& params = static_cast(p); - - if (params.padding.x == 0 && params.padding.y == 0) { - const auto req_input = GetConvolutionBFYXPaddedTensor(params); - const bool bProperInputDesc = CheckConvolutionBinaryPaddedInputDesc(params, req_input); - - return bProperInputDesc; - } - - return true; -} - -bool ConvolutionBinaryUpdateInputParams(binary_convolution_params& params) { - const auto req_input = GetConvolutionBFYXPaddedTensor(params); - const bool bProperInputDesc = CheckConvolutionBinaryPaddedInputDesc(params, req_input); - - if (!bProperInputDesc) { - params.inputs[0] = req_input; - return true; - } - - return false; -} - -std::string BinaryConvolutionKernelBase::GetAutoTuneOptions(int autoTuneIndex) const { - if ((autoTuneIndex >= 0) && (autoTuneIndex < static_cast(autoTuneOptions.size()))) { - return autoTuneOptions[autoTuneIndex]; - } - - return EXE_MODE_DEFAULT; -} - -KernelsData BinaryConvolutionKernelBase::GetTunedKernelsDataByIndex(const Params& params, - const optional_params& options, - const int autoTuneIndex) const { - return GetCommonKernelsData(params, options, GetAutoTuneOptions(autoTuneIndex), autoTuneIndex); -} - -KernelsData BinaryConvolutionKernelBase::GetKernelsDataForAutoTune(const Params& params, - const optional_params& options) const { - if (!Validate(params, options)) { - return {}; - } - - KernelsData res = {}; - - for (size_t i = 0; i < autoTuneOptions.size(); i++) { - KernelsData kd = GetTunedKernelsDataByIndex(params, options, static_cast(i)); - if (!kd.empty()) { - res.emplace_back(kd[0]); - } - } - - return res; -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_base.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_base.h deleted file mode 100644 index 686b81d7e33..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_base.h +++ /dev/null @@ -1,73 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "weight_bias_kernel_base.h" -#include "binary_convolution_params.h" -#include -#include - -namespace kernel_selector { -//////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// -// BinaryConvolutionKernelBase -//////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// -class BinaryConvolutionKernelBase : public WeightBiasKernelBase { -public: - using WeightBiasKernelBase::WeightBiasKernelBase; - virtual ~BinaryConvolutionKernelBase() {} - - struct DispatchData : public CommonDispatchData { - struct CLDNNStyle { - size_t blockWidth, blockHeight; // used for kernels processing blocks - size_t prefetch; - size_t inputBlockArraySize; // Number of elements in array of UNIT_TYPE that must be specified in kernel to - // store/cache input block. - size_t inputBlockWidth; // Number of elements in X dimension stored/cached in input block. - }; - - struct GEMMStyle { - size_t subBlockDimM; - size_t subBlockDimK; - size_t subBlockDimN; - size_t globalWorkSizeDX; - size_t globalWorkSizeDY; - size_t globalWorkSizeDZ; - }; - - union { - CLDNNStyle cldnnStyle; - GEMMStyle gemmStyle; - }; - }; - - std::string GetAutoTuneOptions(int autoTuneIndex) const; - std::vector autoTuneOptions = {EXE_MODE_DEFAULT, EXE_MODE_NO_PRERA_SCH, EXE_MODE_AGE_BASED}; - KernelsData GetKernelsDataForAutoTune(const Params& params, const optional_params& options) const override; - KernelsData GetTunedKernelsDataByIndex(const Params& params, - const optional_params& options, - int autoTuneIndex = -1) const override; - -protected: - virtual WeightsLayout GetPreferredWeightLayout(const binary_convolution_params &) const = 0; - virtual std::string GetKernelName(const binary_convolution_params&) const { return kernelName; } - virtual bool NeedPaddedInput() const { return false; } - bool Validate(const Params& p, const optional_params& o) const override; - using WeightBiasKernelBase::GetJitConstants; - virtual JitConstants GetJitConstants(const binary_convolution_params& params, const DispatchData& dispatchData) const; - virtual JitConstants GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const; - virtual DispatchData SetDefault(const binary_convolution_params& params, int autoTuneIndex = -1) const; - static bool CheckWorkGroups(const DispatchData&); - KernelsData GetCommonKernelsData(const Params& params, - const optional_params& options, - const std::string exeMode = EXE_MODE_DEFAULT, - int autoTuneIndex = -1) const; -}; - -bool ConvolutionBinaryCheckInput(const Params& p, const optional_params& o); -bool CheckConvolutionBinaryPaddedInputDesc(const binary_convolution_params& params, const DataTensor& reqDesc); -bool ConvolutionBinaryUpdateInputParams(binary_convolution_params& params); - -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_generic.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_generic.cpp deleted file mode 100644 index 9acd73f38b8..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_generic.cpp +++ /dev/null @@ -1,251 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include -#include "binary_convolution_kernel_generic.h" -#include -#include -#include - -namespace kernel_selector { - -static const int sub_group_size = 16; -static const int ic_pack_size = 32; -static const int x_block_size = 16; - -ParamsKey BinaryConvolutionKernelGeneric::GetSupportedKey() const { - ParamsKey k; - k.EnableInputDataType(Datatype::BINARY); - k.EnableInputWeightsType(WeightsType::BINARY); - k.EnableOutputDataType(Datatype::F16); - k.EnableOutputDataType(Datatype::F32); - k.EnableOutputDataType(Datatype::INT32); - k.EnableOutputDataType(Datatype::BINARY); - k.EnableInputLayout(DataLayout::b_fs_yx_32fp); - k.EnableOutputLayout(DataLayout::bfyx); - k.EnableOutputLayout(DataLayout::b_fs_yx_32fp); - k.EnableTensorOffset(); - k.EnableTensorPitches(); - k.EnableNonBiasTerm(); - k.EnableBatching(); - k.EnableDifferentTypes(); - return k; -} - -DeviceFeaturesKey BinaryConvolutionKernelGeneric::get_required_device_features_key(const Params& params, const optional_params& /*options*/) const { - DeviceFeaturesKey k; - k.requires_subgroup_shuffle(); - k.requires_blocked_read_write(); - - return k; -} - -BinaryConvolutionKernelBase::DispatchData BinaryConvolutionKernelGeneric::SetDefault(const binary_convolution_params& params, - int) const { - DispatchData dispatchData = BinaryConvolutionKernelBase::SetDefault(params); - - const auto& out = params.outputs[0]; - - auto x = out.X().v; - auto y = out.Y().v; - auto f = out.Feature().v; - auto b = out.Batch().v; - - dispatchData.gws[0] = Align(x, sub_group_size) * y; - dispatchData.gws[1] = CeilDiv(f, 2 * sub_group_size); // 1 WI calc 2 OC x 16 X - dispatchData.gws[2] = b; - - dispatchData.lws[0] = sub_group_size; - dispatchData.lws[1] = 1; - dispatchData.lws[2] = 1; - - return dispatchData; -} - -KernelsPriority BinaryConvolutionKernelGeneric::GetKernelsPriority(const Params& /*params*/, const optional_params& /*options*/) const { - return FORCE_PRIORITY_2; -} - -bool BinaryConvolutionKernelGeneric::Validate(const Params& p, const optional_params& o) const { - if (!BinaryConvolutionKernelBase::Validate(p, o) || !ConvolutionBinaryCheckInput(p, o)) - return false; - - const auto& params = static_cast(p); - - if (params.groups > 1) - return false; - - return true; -} - -JitConstants BinaryConvolutionKernelGeneric::GetJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const { - auto jit = Parent::GetJitConstants(params, dispatchData); - - auto input = params.inputs[0]; - auto output = params.outputs[0]; - size_t input_line_size = params.stride.x * (x_block_size - 1) + params.weights.X().v; - - int pad_physical_val = params.pad_value == -1.0f ? 0x00000000 : 0xFFFFFFFF; - jit.AddConstant(MakeJitConstant("SUB_GROUP_SIZE", sub_group_size)); - jit.AddConstant(MakeJitConstant("INPUT0_FEATURE_NUM_PACKED", CeilDiv(params.inputs[0].Feature().v, ic_pack_size))); - jit.AddConstant(MakeJitConstant("OUTPUT_FEATURE_NUM_PACKED", CeilDiv(params.outputs[0].Feature().v, ic_pack_size))); - jit.AddConstant(MakeJitConstant("PAD_VALUE", pad_physical_val)); - jit.AddConstant(MakeJitConstant("OUTPUT_X_BLOCK_SIZE", x_block_size)); - jit.AddConstant(MakeJitConstant("INPUT_ELEMENTS_PER_WI", CeilDiv(input_line_size, sub_group_size))); - jit.AddConstant(MakeJitConstant("X_BLOCKS", CeilDiv(output.X().v, x_block_size))); - jit.AddConstant(MakeJitConstant("EXCLUDE_PAD", params.pad_value == 0.0f)); - if (params.inputs[0].Feature().v % ic_pack_size) { - jit.AddConstant(MakeJitConstant("LEFTOVERS_IC", params.inputs[0].Feature().v % ic_pack_size)); - jit.AddConstant(MakeJitConstant("FILTER_MASK", - (0xFFFFFFFF >> (ic_pack_size - params.inputs[0].Feature().v % ic_pack_size)))); - } - - if (params.outputs[0].GetDType() == Datatype::BINARY) { - jit.AddConstant(MakeJitConstant("BINARY_PACKED_OUTPUT", 1)); - } - - return jit; -} - -JitConstants BinaryConvolutionKernelGeneric::GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& /*kd*/) const { - JitConstants jit = {}; - - auto input_dt = GetUnitType(params); - FusedOpsConfiguration conf = {"", {"b", "(f_block*16 + i)", "y", "x"}, "res", input_dt, 1 }; - jit.Merge(MakeFusedOpsDeclsJitConstants(params, {conf})); - - size_t op_id = 0; - std::string eltwise_fused_ops = ""; - std::string channel_pack_fused_ops = ""; - std::string prepare_data = ""; - for (auto& fused_dep : params.fused_ops) { - auto fused_dep_codegen = FusedOpsCodeGenerator(fused_dep); - auto get_aligned_load2 = [&](std::string ptr, std::string byte_offset) -> std::string { - if (fused_dep.tensors[0].GetDType() == Datatype::F32) - return "(_sub_group_block_read2((const __global uint*)(" + ptr + ") + (" + byte_offset + ")))"; - else - return "(_sub_group_block_read_us2((const __global ushort*)(" + ptr + ") + (" + byte_offset + - ")))"; - }; - std::string data_type = fused_dep_codegen.GetInputTypeName(0, 1); - std::string vec_data_type = fused_dep_codegen.GetInputTypeName(0, 2); - std::string sc = "sc" + toCodeString(op_id); - std::string sh = "sh" + toCodeString(op_id); - std::string e_add = "e_add" + toCodeString(op_id); - std::string e_mul = "e_mul" + toCodeString(op_id); - - switch (fused_dep.GetType()) { - case KernelType::QUANTIZE: { - std::string var_name_in = fused_dep_codegen.GetInputVarName(0); - std::string var_name_out = fused_dep_codegen.GetInputVarName(3); - std::string cast_type_vec = (fused_dep.tensors[0].GetDType() == Datatype::F32) ? "as_float2" : "as_half2"; - std::string cast_type = (fused_dep.tensors[0].GetDType() == Datatype::F32) ? "as_float" : "as_half"; - - if (fused_dep.tensors[0].Feature().v == params.outputs[0].Feature().v) { - prepare_data += vec_data_type + " " + var_name_in + " = " + cast_type_vec + - get_aligned_load2(fused_dep_codegen.GetInputPtrName(0), "f_block*OC_BLOCK_SIZE") + ";"; - } else { - prepare_data += data_type + " " + var_name_in + " = " + cast_type + - + "(" + fused_dep_codegen.GetInputPtrName(0) + "[0]);"; - } - - if (fused_dep.tensors[2].Feature().v == params.outputs[0].Feature().v) { - prepare_data += vec_data_type + " " + var_name_out + " = " + cast_type_vec + - get_aligned_load2(fused_dep_codegen.GetInputPtrName(3), "f_block*OC_BLOCK_SIZE") + ";"; - } else { - prepare_data += data_type + " " + var_name_out + " = " + cast_type + - "(" + fused_dep_codegen.GetInputPtrName(3)+"[0]);"; - } - - std::string var_in_s0 = fused_dep.tensors[0].Feature().v == params.outputs[0].Feature().v ? var_name_in + ".s0" : var_name_in; - std::string var_in_s1 = fused_dep.tensors[0].Feature().v == params.outputs[0].Feature().v ? var_name_in + ".s1" : var_name_in; - - std::string var_out_s0 = fused_dep.tensors[3].Feature().v == params.outputs[0].Feature().v ? var_name_out + ".s0" : var_name_out; - std::string var_out_s1 = fused_dep.tensors[3].Feature().v == params.outputs[0].Feature().v ? var_name_out + ".s1" : var_name_out; - - channel_pack_fused_ops += "\\\n\tfor (int i = 0; i < 16; i++) {"; - channel_pack_fused_ops += "\\\n\tint ch0, ch1;"; - if (fused_dep.tensors[2].Feature().v == params.outputs[0].Feature().v) { - channel_pack_fused_ops += "\\\n\tif ("+ var_out_s0 + " == UNIT_VAL_ONE) "; - channel_pack_fused_ops += "\\\n\t\tch0 = dst[0*SUB_GROUP_SIZE + i] > " + var_in_s0 + " ? (1 << lid) : 0;"; - channel_pack_fused_ops += "\\\n\telse "; - channel_pack_fused_ops += "\\\n\t\tch0 = dst[0*SUB_GROUP_SIZE + i] <= " + var_in_s0 + " ? (1 << lid) : 0;"; - channel_pack_fused_ops += "\\\n\tif ("+ var_out_s1 + " == UNIT_VAL_ONE) "; - channel_pack_fused_ops += "\\\n\t\tch1 = dst[1*SUB_GROUP_SIZE + i] > " + var_in_s1 + " ? " - "(1 << (SUB_GROUP_SIZE + lid)) : 0;"; - channel_pack_fused_ops += "\\\n\telse "; - channel_pack_fused_ops += "\\\n\t\tch1 = dst[1*SUB_GROUP_SIZE + i] <= " + var_in_s1 + " ? " - "(1 << (SUB_GROUP_SIZE + lid)) : 0;"; - } else { - channel_pack_fused_ops += "\\\n\tif ("+ var_out_s0 + " == UNIT_VAL_ONE) {"; - channel_pack_fused_ops += "\\\n\t\tch0 = dst[0*SUB_GROUP_SIZE + i] > " + var_in_s0 + " ? (1 << lid) : 0;"; - channel_pack_fused_ops += "\\\n\t\tch1 = dst[1*SUB_GROUP_SIZE + i] > " + var_in_s1 + " ? " - "(1 << (SUB_GROUP_SIZE + lid)) : 0;"; - channel_pack_fused_ops += "\\\n\t} else {"; - channel_pack_fused_ops += "\\\n\t\tch0 = dst[0*SUB_GROUP_SIZE + i] <= " + var_in_s0 + " ? (1 << lid) : 0;"; - channel_pack_fused_ops += "\\\n\t\tch1 = dst[1*SUB_GROUP_SIZE + i] <= " + var_in_s1 + " ? " - "(1 << (SUB_GROUP_SIZE + lid)) : 0;"; - channel_pack_fused_ops += "\\\n\t}"; - } - channel_pack_fused_ops += "\\\n\tint packed = ch0 + ch1;"; - channel_pack_fused_ops += "\\\n\tpacked_out[i] = sub_group_reduce_add(packed);"; - channel_pack_fused_ops += "\\\n\t}"; - - break; - } - - case KernelType::ACTIVATION: { - auto p = fused_dep.GetOpParams(); - base_activation_params activation = p->param; - if (activation.function != ActivationFunction::NONE) { - auto suffix = "_FUSED_OP" + toCodeString(op_id); - - jit.Merge(MakeActivationJitConstants(activation, fused_dep.output_tensor.GetDType(), suffix)); - eltwise_fused_ops += "\\\n\tres = ACTIVATION" + suffix + "((OUTPUT_TYPE)res, ACTIVATION_PARAMS" + suffix + ");"; - } - - break; - } - - case KernelType::ELTWISE: { - std::string cast_type = (fused_dep.tensors[0].GetDType() == Datatype::F32) ? "as_float2" : "as_half2"; - std::string var_name = fused_dep_codegen.GetInputVarName(0); - prepare_data += vec_data_type + " " + var_name + " = " + cast_type + - get_aligned_load2(fused_dep_codegen.GetInputPtrName(0), "f_block*OC_BLOCK_SIZE") + ";"; - - auto eltwise_p = std::dynamic_pointer_cast(fused_dep.op_params); - - if (eltwise_p->mode == EltwiseMode::ADD) { - eltwise_fused_ops += data_type + " " + e_add + " = (i < 16) ? " + var_name + ".s0" + " : " + var_name + ".s1;"; - eltwise_fused_ops += "res = res+" + e_add +";"; - } else if (eltwise_p->mode == EltwiseMode::MUL) { - eltwise_fused_ops += data_type + " " + e_mul + " = (i < 16) ? " + var_name + ".s0" + " : " + var_name + ".s1;"; - eltwise_fused_ops += "res = res*" + e_mul +";"; - } else { - throw std::invalid_argument("Not supported eltwise fusing op in binary_convolution_generic kernel: " + params.layerID); - } - - break; - } - - default: - throw std::invalid_argument("Invalid fused op in binary_convolution_generic kernel: " + params.layerID); - } - - op_id++; - } - jit.AddConstant(MakeJitConstant("DO_ELTWISE_FUSED_OPS", eltwise_fused_ops)); - jit.AddConstant(MakeJitConstant("DO_CHANNEL_PACK_OPS", channel_pack_fused_ops)); - jit.AddConstant(MakeJitConstant("FUSED_OPS_PREPARE_DATA", prepare_data)); - jit.AddConstant(MakeJitConstant("CUSTOM_FUSED_OPS", true)); - - return jit; -} - -KernelsData BinaryConvolutionKernelGeneric::GetKernelsData(const Params& params, const optional_params& options) const { - return GetTunedKernelsDataByIndex(params, options); -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_generic.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_generic.h deleted file mode 100644 index 82653b2d779..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_generic.h +++ /dev/null @@ -1,34 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "binary_convolution_kernel_base.h" -#include - -namespace kernel_selector { - -class BinaryConvolutionKernelGeneric : public BinaryConvolutionKernelBase { -public: - using Parent = BinaryConvolutionKernelBase; - - BinaryConvolutionKernelGeneric() : BinaryConvolutionKernelBase("binary_convolution_gpu_generic") {} - virtual ~BinaryConvolutionKernelGeneric() {} - - KernelsData GetKernelsData(const Params& params, const optional_params& options) const override; - KernelsPriority GetKernelsPriority(const Params& params, const optional_params& options) const override; - ParamsKey GetSupportedKey() const override; - DeviceFeaturesKey get_required_device_features_key(const Params& params, const optional_params& /*options*/) const override; - -protected: - WeightsLayout GetPreferredWeightLayout(const binary_convolution_params &) const override { - return WeightsLayout::os_is_yx_osv32_isv32p; - } - JitConstants GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const override; - bool Validate(const Params& p, const optional_params& o) const override; - DispatchData SetDefault(const binary_convolution_params& arg, int autoTuneIndex = -1) const override; - JitConstants GetJitConstants(const binary_convolution_params& params, const DispatchData& dispatchData) const override; -}; -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_ref.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_ref.cpp deleted file mode 100644 index 87a61d64c6e..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_ref.cpp +++ /dev/null @@ -1,98 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "binary_convolution_kernel_ref.h" -#include - -namespace kernel_selector { - -ParamsKey BinaryConvolutionKernelRef::GetSupportedKey() const { - ParamsKey k; - k.EnableInputDataType(Datatype::BINARY); - k.EnableInputWeightsType(WeightsType::BINARY); - k.EnableOutputDataType(Datatype::F16); - k.EnableOutputDataType(Datatype::F32); - k.EnableOutputDataType(Datatype::INT32); - k.EnableOutputDataType(Datatype::BINARY); - k.EnableInputLayout(DataLayout::b_fs_yx_32fp); - k.EnableOutputLayout(DataLayout::bfyx); - k.EnableOutputLayout(DataLayout::b_fs_yx_32fp); - k.EnableTensorOffset(); - k.EnableTensorPitches(); - k.EnableDilation(); - k.EnableNonBiasTerm(); - k.EnableBatching(); - return k; -} - -BinaryConvolutionKernelBase::DispatchData BinaryConvolutionKernelRef::SetDefault(const binary_convolution_params& params, - int) const { - DispatchData dispatchData = BinaryConvolutionKernelBase::SetDefault(params); - - const auto& out = params.outputs[0]; - - auto b = out.Batch().v; - auto f = out.Feature().v; - auto y = out.Y().v; - auto x = out.X().v; - - dispatchData.gws[0] = b; - dispatchData.gws[1] = f; - dispatchData.gws[2] = x * y; - - dispatchData.lws[0] = 1; - dispatchData.lws[1] = 1; - dispatchData.lws[2] = 1; - - return dispatchData; -} - -KernelsPriority BinaryConvolutionKernelRef::GetKernelsPriority(const Params& /*params*/, const optional_params& /*options*/) const { - return DONT_USE_IF_HAVE_SOMETHING_ELSE; -} - -JitConstants BinaryConvolutionKernelRef::GetJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const { - auto jit = Parent::GetJitConstants(params, dispatchData); - - int pad_physical_val = params.pad_value == -1.0f ? 0x00000000 : 0xFFFFFFFF; - int leftovers_mask = (0xFFFFFFFF >> (32 - params.inputs[0].Feature().v % 32)); - jit.AddConstant(MakeJitConstant("INPUT0_FEATURE_NUM_PACKED", CeilDiv(params.inputs[0].Feature().v, 32))); - jit.AddConstant(MakeJitConstant("FEATURE_PACK_SIZE", 32)); - jit.AddConstant(MakeJitConstant("OFM_BLOCK_SIZE", 32)); - jit.AddConstant(MakeJitConstant("EXCLUDE_PAD", params.pad_value == 0.0f)); - jit.AddConstant(MakeJitConstant("PAD_VALUE", pad_physical_val)); - jit.AddConstant(MakeJitConstant("LEFTOVERS", params.inputs[0].Feature().v % 32 != 0)); - jit.AddConstant(MakeJitConstant("LEFTOVERS_MASK", leftovers_mask)); - - return jit; -} - -KernelsData BinaryConvolutionKernelRef::GetKernelsData(const Params& params, const optional_params& options) const { - return GetTunedKernelsDataByIndex(params, options); -} - -bool BinaryConvolutionKernelRef::Validate(const Params& p, const optional_params& o) const { - if (!BinaryConvolutionKernelBase::Validate(p, o) || !ConvolutionBinaryCheckInput(p, o)) - return false; - - const auto& params = static_cast(p); - - if (!params.fused_ops.empty()) - return false; - - return true; -} - -JitConstants BinaryConvolutionKernelRef::GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& /*kd*/) const { - JitConstants jit = {}; - - auto input_dt = GetUnitType(params); - FusedOpsConfiguration conf = {"", {"b", "f", "y", "x"}, "res", input_dt, 1 }; - jit.Merge(MakeFusedOpsJitConstants(params, {conf})); - - return jit; -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_ref.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_ref.h deleted file mode 100644 index f993acd1810..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_ref.h +++ /dev/null @@ -1,33 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "binary_convolution_kernel_base.h" -#include - -namespace kernel_selector { - -class BinaryConvolutionKernelRef : public BinaryConvolutionKernelBase { -public: - using Parent = BinaryConvolutionKernelBase; - - BinaryConvolutionKernelRef() : BinaryConvolutionKernelBase("binary_convolution_gpu_ref") {} - virtual ~BinaryConvolutionKernelRef() {} - - KernelsData GetKernelsData(const Params& params, const optional_params& options) const override; - KernelsPriority GetKernelsPriority(const Params& params, const optional_params& options) const override; - ParamsKey GetSupportedKey() const override; - -protected: - WeightsLayout GetPreferredWeightLayout(const binary_convolution_params &) const override { - return WeightsLayout::os_is_yx_osv32_isv32p; - } - JitConstants GetFusedPrimitivesJitConstants(const binary_convolution_params& params, - const DispatchData& dispatchData) const override; - bool Validate(const Params& p, const optional_params& o) const override; - DispatchData SetDefault(const binary_convolution_params& arg, int autoTuneIndex = -1) const override; - JitConstants GetJitConstants(const binary_convolution_params& params, const DispatchData& dispatchData) const override; -}; -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_selector.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_selector.cpp deleted file mode 100644 index 5bb0a615002..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_selector.cpp +++ /dev/null @@ -1,23 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "binary_convolution_kernel_selector.h" -#include "binary_convolution_kernel_ref.h" -#include "binary_convolution_kernel_generic.h" -#include "binary_convolution_kernel_1x1.h" -#include "binary_convolution_kernel_1x1_b_fs_yx_fsv16.h" - -namespace kernel_selector { -binary_convolution_kernel_selector::binary_convolution_kernel_selector() { - Attach(); - Attach(); - Attach(); - Attach(); -} - -KernelsData binary_convolution_kernel_selector::GetBestKernels(const Params& params, - const optional_params& options) const { - return GetNaiveBestKernel(params, options, KernelType::BINARY_CONVOLUTION); -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_selector.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_selector.h deleted file mode 100644 index 65ed563de5b..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_kernel_selector.h +++ /dev/null @@ -1,23 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "kernel_selector.h" - -namespace kernel_selector { -class binary_convolution_kernel_selector : public kernel_selector_base { -public: - static binary_convolution_kernel_selector& Instance() { - static binary_convolution_kernel_selector instance_; - return instance_; - } - - binary_convolution_kernel_selector(); - - virtual ~binary_convolution_kernel_selector() {} - - KernelsData GetBestKernels(const Params& params, const optional_params& options) const override; -}; -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_params.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_params.cpp deleted file mode 100644 index 070d02e7118..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_params.cpp +++ /dev/null @@ -1,52 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "binary_convolution_params.h" -#include -#include - -namespace kernel_selector { -std::string binary_convolution_params::to_string() const { - std::stringstream s; - - s << base_params::to_string() << "_"; - s << filterSize.x << "_" << filterSize.y << "_"; - s << stride.x << "_" << stride.y << "_"; - s << dilation.x << "_" << dilation.y << "_"; - s << padding.x << "_" << padding.y << "_"; - s << 1; - s << groups; - - return s.str(); -} - -std::string binary_convolution_params::to_cache_string_v2() const { - std::stringstream s; - - s << weight_bias_params::to_cache_string_v2() << ";"; - s << filterSize.x << "_" << filterSize.y << "_" << filterSize.z << ";"; - s << stride.x << "_" << stride.y << "_" << stride.z << ";"; - s << dilation.x << "_" << dilation.y << "_" << dilation.z << ";"; - s << padding.x << "_" << padding.y << "_" << padding.z << ";"; - s << 1 << ";"; - s << groups; - - return s.str(); -} - -ParamsKey binary_convolution_params::GetParamsKey() const { - ParamsKey k = weight_bias_params::GetParamsKey(); - - if (dilation.x != 1 || - dilation.y != 1) { - k.EnableDilation(); - } - - if (groups > 1) { - k.EnableGroupedConvolution(); - } - - return k; -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_params.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_params.h deleted file mode 100644 index 9f9d9b38f39..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/binary_convolution/binary_convolution_params.h +++ /dev/null @@ -1,39 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "weight_bias_params.h" -#include -#include - -namespace kernel_selector { - -//////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// -// binary_convolution_params -//////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// -struct binary_convolution_params : public weight_bias_params { - binary_convolution_params() : weight_bias_params(KernelType::BINARY_CONVOLUTION) {} - - uSize filterSize; - uSize stride; - uSize dilation; - uSize padding; - Datatype out_dt = Datatype::UNSUPPORTED; - float pad_value = 0.0f; - uint32_t groups = 1; - - std::string to_string() const override; - std::string to_cache_string_v2() const override; - ParamsKey GetParamsKey() const override; -}; - -//////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// -// convolution_optional_params -//////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// -struct binary_convolution_optional_params : weight_bias_optional_params { - binary_convolution_optional_params() : weight_bias_optional_params(KernelType::BINARY_CONVOLUTION) {} -}; - -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_base.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_base.cpp index 69479939834..da10f037941 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_base.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_base.cpp @@ -14,34 +14,12 @@ bool QuantizeKernelBase::Validate(const Params& p, const optional_params&) const if (params.inputs.size() != 5) return false; - // Binary packed output is possible only with bfyx input and b_fs_yx_32fp output - if (params.outputs[0].GetDType() == Datatype::BINARY && - (params.outputs[0].GetLayout() != DataLayout::b_fs_yx_32fp || params.inputs[0].GetLayout() != DataLayout::bfyx)) - return false; - return true; } JitConstants QuantizeKernelBase::GetJitConstants(const quantize_params& params, const CommonDispatchData& dispatchData) const { JitConstants jit = MakeBaseParamsJitConstants(params); - if (params.packed_binary_output) { - jit.AddConstant(MakeJitConstant("PACKED_BINARY_OUTPUT", params.packed_binary_output)); - jit.AddConstant(MakeJitConstant("OUTPUT_FEATURE_NUM_PACKED", CeilDiv(params.outputs[0].Feature().v, 32))); - jit.AddConstant(MakeJitConstant("OC_BLOCK_SIZE", 32)); - if ((params.inputs[3].LogicalSize() == 1 && params.inputs[4].LogicalSize() == 1) || - (params.inputs[3].LogicalSize() == params.inputs[3].Batch().v && - params.inputs[4].LogicalSize() == params.inputs[4].Batch().v)) { - jit.AddConstant(MakeJitConstant("SINGLE_OUT_VAL", 1)); - - } else if (params.inputs[3].LogicalSize() == params.outputs[0].Feature().v && - params.inputs[4].LogicalSize() == params.outputs[0].Feature().v) { - jit.AddConstant(MakeJitConstant("PER_CHANNEL_OUT_VAL", 1)); - } else { - throw std::runtime_error("Unsupported const blob shape in node " + params.layerID); - } - } - jit.AddConstant(MakeJitConstant("LEVELS", static_cast(params.levels))); jit.AddConstant(MakeJitConstant("LWS_0", dispatchData.lws[0])); diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_params.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_params.h index 692acc7b9d4..d781ac397d0 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_params.h +++ b/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_params.h @@ -14,7 +14,6 @@ struct quantize_params : public base_params { quantize_params() : base_params(KernelType::QUANTIZE) , levels(0) - , packed_binary_output(false) , scale_shift_opt(false) , has_post_scale(true) , has_post_shift(true) @@ -38,7 +37,6 @@ struct quantize_params : public base_params { , out_shift(0.0f) { } int levels; - bool packed_binary_output; bool scale_shift_opt; bool has_post_scale; bool has_post_shift; @@ -65,8 +63,6 @@ struct quantize_params : public base_params { ParamsKey GetParamsKey() const override { auto k = base_params::GetParamsKey(); - if (packed_binary_output) - k.EnableQuantizePackedBinaryOutput(); if (scale_shift_opt) k.EnableQuantizeScaleShiftOpt(); return k; diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_ref.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_ref.cpp index 131cbb5f9e7..8b5c16c9518 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_ref.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernels/quantize/quantize_kernel_ref.cpp @@ -17,14 +17,12 @@ ParamsKey QuantizeKernelRef::GetSupportedKey() const { k.EnableOutputDataType(Datatype::F32); k.EnableOutputDataType(Datatype::UINT8); k.EnableOutputDataType(Datatype::INT8); - k.EnableOutputDataType(Datatype::BINARY); k.EnableAllInputLayout(); k.EnableAllOutputLayout(); k.EnableTensorOffset(); k.EnableTensorPitches(); k.EnableBatching(); k.EnableDifferentTypes(); - k.EnableQuantizePackedBinaryOutput(); k.EnableDynamicShapesSupport(); return k; } @@ -34,7 +32,7 @@ CommonDispatchData QuantizeKernelRef::SetDefault(const quantize_params& params) auto output = params.outputs[0]; - if (output.GetLayout() == DataLayout::b_fs_yx_fsv16 && !params.packed_binary_output) { + if (output.GetLayout() == DataLayout::b_fs_yx_fsv16) { dispatchData.gws[0] = output.Batch().v; dispatchData.gws[1] = Align(output.Feature().v, sub_group_size); dispatchData.gws[2] = output.Y().v * output.X().v * output.Z().v; @@ -44,7 +42,7 @@ CommonDispatchData QuantizeKernelRef::SetDefault(const quantize_params& params) dispatchData.lws[2] = 1; } else { dispatchData.gws[0] = output.Batch().v; - dispatchData.gws[1] = params.packed_binary_output ? CeilDiv(output.Feature().v, 32) : output.Feature().v; + dispatchData.gws[1] = output.Feature().v; dispatchData.gws[2] = Align(output.X().v * output.Y().v * output.Z().v * output.W().v * output.U().v * output.V().v, 16); dispatchData.lws[0] = 1; @@ -57,7 +55,7 @@ CommonDispatchData QuantizeKernelRef::SetDefault(const quantize_params& params) JitConstants QuantizeKernelRef::GetJitConstants(const quantize_params& params, const CommonDispatchData& dispatchData) const { JitConstants jit = Parent::GetJitConstants(params, dispatchData); - if (params.outputs[0].GetLayout() == DataLayout::b_fs_yx_fsv16 && !params.packed_binary_output) { + if (params.outputs[0].GetLayout() == DataLayout::b_fs_yx_fsv16) { jit.AddConstant(MakeJitConstant("SUB_GROUP_SIZE", sub_group_size)); } return jit; @@ -68,15 +66,6 @@ bool QuantizeKernelRef::Validate(const Params& p, const optional_params&) const if (params.inputs.size() != 5) return false; - // Binary packed output is possible only with b_fs_yx_32fp output layout and some input layouts - if (params.outputs[0].GetDType() == Datatype::BINARY && - (params.outputs[0].GetLayout() != DataLayout::b_fs_yx_32fp || - (params.inputs[0].GetLayout() != DataLayout::bfyx && - params.inputs[0].GetLayout() != DataLayout::bfzyx && - params.inputs[0].GetLayout() != DataLayout::b_fs_zyx_fsv16 && - params.inputs[0].GetLayout() != DataLayout::b_fs_yx_fsv16 && - params.inputs[0].GetLayout() != DataLayout::fs_b_yx_fsv32))) - return false; return true; } diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_base.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_base.cpp index 7fecbc14345..7aea6878e13 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_base.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_base.cpp @@ -22,7 +22,6 @@ inline uint32_t SubGroupSize(WeightsLayout l) { case WeightsLayout::os_i_osv16__ai8: case WeightsLayout::i_yxs_os_yxsv2_osv16: case WeightsLayout::iy_xs_os_xsv2_osv16__ao32: - case WeightsLayout::os_is_yx_osv32_isv32p: case WeightsLayout::os_is_yx_isv16_osv16: case WeightsLayout::os_is_zyx_isv16_osv16: case WeightsLayout::is_os_zyx_isv16_osv16: diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_binary.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_binary.cpp deleted file mode 100644 index 41e922d8efe..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_binary.cpp +++ /dev/null @@ -1,90 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "reorder_kernel_binary.h" -#include "kernel_selector_utils.h" -#include - -namespace kernel_selector { -ParamsKey ReorderKernelBinary::GetSupportedKey() const { - ParamsKey k; - k.EnableInputDataType(Datatype::F16); - k.EnableInputDataType(Datatype::F32); - k.EnableInputDataType(Datatype::BINARY); - k.EnableOutputDataType(Datatype::BINARY); - k.EnableOutputDataType(Datatype::F32); - k.EnableOutputDataType(Datatype::F16); - k.EnableDifferentTypes(); - k.EnableInputLayout(DataLayout::bfyx); - k.EnableInputLayout(DataLayout::b_fs_yx_32fp); - k.EnableOutputLayout(DataLayout::b_fs_yx_32fp); - k.EnableOutputLayout(DataLayout::bfyx); - k.EnableTensorOffset(); - k.EnableTensorPitches(); - k.EnableBatching(); - return k; -} - -JitConstants ReorderKernelBinary::GetJitConstants(const reorder_params& params) const { - auto jit = ReorderKernelBase::GetJitConstants(params); - KernelData kd = KernelData::Default(params); - reorder_params& newParams = *static_cast(kd.params.get()); - - const auto& input = newParams.inputs[0]; - jit.AddConstant(MakeJitConstant("ELEMENTS_COUNT", input.LogicalSize())); - jit.AddConstant(MakeJitConstant("IFM_PACK_SIZE", 32)); - - if (input.GetDType() == Datatype::BINARY) { - jit.AddConstant(MakeJitConstant("BINARY_INPUT", 1)); - jit.AddConstant(MakeJitConstant("INPUT_PACKED_FEATURES_NUM", CeilDiv(input.Feature().v, 16))); - } - - if (params.outputs[0].GetDType() == Datatype::BINARY) { - jit.AddConstant(MakeJitConstant("BINARY_OUTPUT", 1)); - jit.AddConstant(MakeJitConstant("OUTPUT_PACKED_FEATURES_NUM", CeilDiv(params.outputs[0].Feature().v, 32))); - } - - return jit; -} - -ReorderKernelBinary::DispatchData ReorderKernelBinary::SetDefault(const reorder_params& params) const { - DispatchData dispatchData; - auto in_layout = params.inputs[0].GetLayout(); - auto out_layout = params.outputs[0].GetLayout(); - std::vector> dims_by_gws = {{ Tensor::DataChannelName::BATCH }, - { Tensor::DataChannelName::FEATURE }, - { Tensor::DataChannelName::X, Tensor::DataChannelName::Y }}; - - const auto& input = params.inputs[0]; - - dispatchData.gws = { input.Batch().v, CeilDiv(input.Feature().v, 32), input.Y().v * input.X().v }; - dispatchData.lws = GetOptimalLocalWorkGroupSizes(dispatchData.gws, params.engineInfo, in_layout, out_layout, dims_by_gws); - - return dispatchData; -} - -KernelsData ReorderKernelBinary::GetKernelsData(const Params& params, const optional_params& options) const { - assert(params.GetType() == KernelType::REORDER); - - const reorder_params& orgParams = static_cast(params); - - if (orgParams.inputs[0].GetDType() != Datatype::BINARY && - orgParams.outputs[0].GetDType() != Datatype::BINARY) - return {}; - - if (orgParams.inputs[0].GetDType() == Datatype::BINARY && - orgParams.inputs[0].GetLayout() != DataLayout::b_fs_yx_32fp) - return {}; - - if (orgParams.outputs[0].GetDType() == Datatype::BINARY && - orgParams.outputs[0].GetLayout() != DataLayout::b_fs_yx_32fp) - return {}; - - return GetCommonKernelsData(orgParams, options); -} - -KernelsPriority ReorderKernelBinary::GetKernelsPriority(const Params& /*params*/, const optional_params& /*options*/) const { - return FORCE_PRIORITY_6; -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_binary.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_binary.h deleted file mode 100644 index 511c161196f..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_binary.h +++ /dev/null @@ -1,22 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "reorder_kernel_base.h" - -namespace kernel_selector { -class ReorderKernelBinary : public ReorderKernelBase { -public: - ReorderKernelBinary() : ReorderKernelBase("reorder_data_binary") {} - - KernelsData GetKernelsData(const Params& params, const optional_params& options) const override; - KernelsPriority GetKernelsPriority(const Params& params, const optional_params& options) const override; - JitConstants GetJitConstants(const reorder_params& params) const override; - DispatchData SetDefault(const reorder_params& arg) const override; - -protected: - ParamsKey GetSupportedKey() const override; -}; -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_selector.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_selector.cpp index 80dffe7bcd0..0a56ae7b803 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_selector.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_kernel_selector.cpp @@ -8,7 +8,6 @@ #include "reorder_from_winograd_2x3_kernel.h" #include "reorder_to_winograd_2x3_kernel.h" #include "reorder_kernel_to_yxfb_batched.h" -#include "reorder_kernel_binary.h" #include "reorder_biplanar_nv12.h" #include "reorder_kernel_fs_b_yx_fsv32_to_bfyx.h" #include "reorder_kernel_bfyx_to_blocked_format.h" @@ -18,7 +17,6 @@ namespace kernel_selector { reorder_kernel_selector::reorder_kernel_selector() { Attach(); - Attach(); Attach(); Attach(); Attach(); diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_binary_kernel.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_binary_kernel.cpp deleted file mode 100644 index 5fa9a2f56ad..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_binary_kernel.cpp +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "reorder_weights_binary_kernel.h" -#include "kernel_selector_utils.h" -#include - -namespace kernel_selector { -ParamsKey ReorderWeightsBinaryKernel::GetSupportedKey() const { - ParamsKey k; - k.EnableInputWeightsType(WeightsType::BINARY); - k.EnableOutputWeightsType(WeightsType::BINARY); - k.EnableInputWeightsLayout(WeightsLayout::oiyx); - k.EnableOutputWeightsLayout(WeightsLayout::os_is_yx_osv32_isv32p); - k.EnableDifferentTypes(); - k.EnableTensorOffset(); - k.EnableTensorPitches(); - return k; -} - -ReorderWeightsBinaryKernel::DispatchData ReorderWeightsBinaryKernel::SetDefault( - const reorder_weights_params& params) const { - const auto& out = params.output; - - DispatchData dispatchData; - - dispatchData.gws = { out.OFM().v, CeilDiv(out.IFM().v, 32), out.X().v * out.Y().v }; - dispatchData.lws = GetOptimalLocalWorkGroupSizes(dispatchData.gws, params.engineInfo); - - return dispatchData; -} - -KernelsData ReorderWeightsBinaryKernel::GetKernelsData(const Params& params, const optional_params& options) const { - const reorder_weights_params& orgParams = static_cast(params); - return GetCommonKernelsData(orgParams, options); -} - -KernelsPriority ReorderWeightsBinaryKernel::GetKernelsPriority(const Params& /*params*/, const optional_params& /*options*/) const { - return FORCE_PRIORITY_4; -} -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_binary_kernel.h b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_binary_kernel.h deleted file mode 100644 index 766ad443fca..00000000000 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_binary_kernel.h +++ /dev/null @@ -1,21 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#pragma once - -#include "reorder_kernel_base.h" - -namespace kernel_selector { -class ReorderWeightsBinaryKernel : public ReorderKernelBase { -public: - ReorderWeightsBinaryKernel() : ReorderKernelBase("reorder_weights_binary") {} - - KernelsData GetKernelsData(const Params& params, const optional_params& options) const override; - KernelsPriority GetKernelsPriority(const Params& params, const optional_params& options) const override; - DispatchData SetDefault(const reorder_weights_params& arg) const override; - -protected: - ParamsKey GetSupportedKey() const override; -}; -} // namespace kernel_selector diff --git a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_kernel_selector.cpp b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_kernel_selector.cpp index fbe010f111b..c43fd014212 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_kernel_selector.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/kernels/reorder/reorder_weights_kernel_selector.cpp @@ -8,7 +8,6 @@ #include "reorder_weights_winograd_6x3_kernel.h" #include "reorder_weights_image_fyx_b_kernel.h" #include "reorder_weights_image_winograd_6x3_kernel.h" -#include "reorder_weights_binary_kernel.h" #include "reorder_weights_opt.h" #include "reorder_weights_int4.h" @@ -20,7 +19,6 @@ ReorderWeightsKernelSelector::ReorderWeightsKernelSelector() { Attach(); Attach(); Attach(); - Attach(); Attach(); Attach(); } diff --git a/src/plugins/intel_gpu/src/kernel_selector/tensor_type.cpp b/src/plugins/intel_gpu/src/kernel_selector/tensor_type.cpp index 3fcd03bdece..e4314d9cc11 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/tensor_type.cpp +++ b/src/plugins/intel_gpu/src/kernel_selector/tensor_type.cpp @@ -61,7 +61,6 @@ DataTensor::DataChannelArray DataTensor::dataChannelArray {{ { DataLayout::bfzyx, { 0, 1, 2, -1, -1, -1, 3, 4 } }, { DataLayout::bzyxf, { 1, 2, 3, -1, -1, -1, 0, 4 } }, { DataLayout::fs_b_yx_fsv32, { 0, 1, -1, -1, -1, -1, 3, 2 } }, - { DataLayout::b_fs_yx_32fp, { 0, 1, -1, -1, -1, -1, 2, 3 } }, { DataLayout::bfwzyx, { 0, 1, 2, 3, -1, -1, 4, 5 } }, { DataLayout::bfuwzyx, { 0, 1, 2, 3, 4, -1, 5, 6 } }, { DataLayout::bfvuwzyx, { 0, 1, 2, 3, 4, 5, 6, 7 } }, @@ -151,7 +150,6 @@ WeightsTensor::WeightsChannelArray WeightsTensor::weightsChannelArray {{ { WeightsLayout::os_is_zyx_osv32_isv4, { 0, 1, 2, 3, 4, -1 } }, { WeightsLayout::oizyx, { 0, 1, 2, 3, 4, -1 } }, { WeightsLayout::iozyx, { 0, 1, 2, 4, 3, -1 } }, - { WeightsLayout::os_is_yx_osv32_isv32p, { 0, 1, -1, 2, 3, -1 } }, { WeightsLayout::os_is_zyx_isv16_osv16, { 0, 1, 2, 3, 4, -1 } }, { WeightsLayout::os_is_yx_isv16_osv16, { 0, 1, -1, 2, 3, -1 } }, { WeightsLayout::is_os_yx_osv8_isv4, { 0, 1, -1, 3, 2, -1 } }, @@ -261,10 +259,6 @@ NDims DataTensor::GetSimpleDims(const std::vector& d, DataLayout l) { assert(newDims.size() == 5); newDims[3] = RoundUp(newDims[3], 32); break; - case b_fs_yx_32fp: - assert(newDims.size() == 4); - newDims[3] = RoundUp(newDims[3], 32); - break; case fs_b_yx_fsv32: assert(newDims.size() == 4); newDims[3] = RoundUp(newDims[3], 32); @@ -788,11 +782,6 @@ NDims WeightsTensor::GetSimpleDims(const std::vector& d, WeightsLayout l newDims[3] = RoundUp(newDims[3], 4); newDims[4] = RoundUp(newDims[4], 32); break; - case os_is_yx_osv32_isv32p: - assert(newDims.size() == 4); - newDims[2] = RoundUp(newDims[2], 32); // ic - newDims[3] = RoundUp(newDims[3], 32); // oc - break; case os_is_yx_isv16_osv16: assert(newDims.size() == 4); newDims[2] = RoundUp(newDims[2], 16); diff --git a/src/plugins/intel_gpu/src/kernel_selector/tensor_type.h b/src/plugins/intel_gpu/src/kernel_selector/tensor_type.h index 205b3198a7a..23a56729f1e 100644 --- a/src/plugins/intel_gpu/src/kernel_selector/tensor_type.h +++ b/src/plugins/intel_gpu/src/kernel_selector/tensor_type.h @@ -72,7 +72,6 @@ enum DataLayout { bfzyx, // batch+feature+3D spatial bzyxf, fs_b_yx_fsv32, // for FP16 kernels, 32 features to avoid partial writes - b_fs_yx_32fp, // bfyx with blocks of 16 packed binary input channels bfwzyx, // batch, feature, 4D spatial bfuwzyx, // batch, feature, 5D spatial bfvuwzyx, // batch, feature, 6D spatial @@ -191,7 +190,6 @@ enum WeightsLayout { os_is_yx_osv4_isv16, oizyx, iozyx, - os_is_yx_osv32_isv32p, // 2 blocks: 32 packed binary in channels and 32 output channels os_is_osv32_isv32_swizzled_by_4, // for weights for 1x1 IMAD convolution os_i_yxs_osv4_yxsv4, // for weights for depthwise IMAD convolution os_y_is_x_osv8_isv2, diff --git a/src/plugins/intel_gpu/src/plugin/graph.cpp b/src/plugins/intel_gpu/src/plugin/graph.cpp index e152e0367fe..0eeeb3555a0 100644 --- a/src/plugins/intel_gpu/src/plugin/graph.cpp +++ b/src/plugins/intel_gpu/src/plugin/graph.cpp @@ -158,7 +158,6 @@ std::shared_ptr Graph::get_runtime_model(std::vectorget_bilinear_interpolation_pad()); } -static void CreateBinaryConvolutionOp(ProgramBuilder& p, const std::shared_ptr& op) { - validate_inputs_count(op, {2}); - auto inputs = p.GetInputInfo(op); - std::string layerName = layer_type_name_ID(op); - - auto outDims = op->get_output_shape(0); - - std::vector weights = {inputs[1].pid}; - cldnn::data_types calc_precision = cldnn::element_type_to_data_type(op->get_output_element_type(0)); - - auto strides = op->get_strides(); - auto pads_begin = op->get_pads_begin(); - auto dilations = op->get_dilations(); - - // Extend 1d vectors to 2d as 1d can't be handled properly by the graph optimizer for now - strides.resize(std::max(2, strides.size()), 1); - pads_begin.resize(std::max(2, pads_begin.size()), 0); - dilations.resize(std::max(2, dilations.size()), 1); - - auto convPrim = cldnn::binary_convolution(layerName, - inputs[0], - weights, - strides, - pads_begin, - dilations, - tensor_from_dims(outDims), - 1, - op->get_pad_value(), - calc_precision); - - p.add_primitive(*op, convPrim); -} - REGISTER_FACTORY_IMPL(v1, GroupConvolution); REGISTER_FACTORY_IMPL(v1, Convolution); REGISTER_FACTORY_IMPL(v1, ConvolutionBackpropData); REGISTER_FACTORY_IMPL(v1, GroupConvolutionBackpropData); REGISTER_FACTORY_IMPL(v1, DeformableConvolution); REGISTER_FACTORY_IMPL(v8, DeformableConvolution); -REGISTER_FACTORY_IMPL(v1, BinaryConvolution); } // namespace intel_gpu } // namespace ov diff --git a/src/plugins/intel_gpu/src/plugin/transformations/binary_conv_to_conv.cpp b/src/plugins/intel_gpu/src/plugin/transformations/binary_conv_to_conv.cpp new file mode 100644 index 00000000000..40c4dfa781e --- /dev/null +++ b/src/plugins/intel_gpu/src/plugin/transformations/binary_conv_to_conv.cpp @@ -0,0 +1,120 @@ +// Copyright (C) 2023 Intel Corporation +// SPDX-License-Identifier: Apache-2.0 +// + +#include "binary_conv_to_conv.hpp" +#include + +#include "openvino/core/coordinate_diff.hpp" +#include "openvino/core/type/element_type.hpp" +#include "openvino/core/type/float16.hpp" +#include "openvino/op/constant.hpp" +#include "openvino/op/binary_convolution.hpp" +#include "openvino/op/convolution.hpp" +#include "openvino/op/fake_quantize.hpp" +#include "openvino/op/pad.hpp" +#include "openvino/op/util/attr_types.hpp" +#include "openvino/pass/pattern/op/pattern.hpp" +#include "openvino/pass/pattern/op/wrap_type.hpp" +#include "openvino/pass/pattern/op/or.hpp" +#include "transformations/utils/utils.hpp" + +namespace ov { +namespace intel_gpu { + +namespace { +template +void convert_packed_bin_to_fp(const uint8_t* src_ptr, DST_T* dst_ptr, size_t size) { + for (size_t i = 0; i < size; i++) { + auto val = (src_ptr[i / 8] >> (i % 8)) & 0x01; + dst_ptr[i] = static_cast(val == 0 ? -1.0f : 1.0f); + } +} +} // namespace + +ConvertBinaryConvolutionToConvolution::ConvertBinaryConvolutionToConvolution() { + using namespace ov::pass::pattern; + + auto binary_fq = [](const Output& node) { + auto fq = std::dynamic_pointer_cast(node.get_node_shared_ptr()); + if (!fq) + return false; + + return fq->get_levels() == 2; + }; + + auto activations_input_m = any_input(); + auto in_lo_m = wrap_type(); + auto in_hi_m = wrap_type(); + auto out_lo_m = wrap_type(); + auto out_hi_m = wrap_type(); + auto fq_m = wrap_type({activations_input_m, in_lo_m, in_hi_m, out_lo_m, out_hi_m}, binary_fq); + auto weights_input_m = wrap_type(type_matches(ov::element::u1)); + auto binary_conv_m = wrap_type({fq_m, weights_input_m}); + + + ov::matcher_pass_callback callback = [=](ov::pass::pattern::Matcher& m) { + const auto& pattern_map = m.get_pattern_value_map(); + + auto binary_conv = std::dynamic_pointer_cast(pattern_map.at(binary_conv_m).get_node_shared_ptr()); + auto activations = pattern_map.at(activations_input_m); + auto weights = std::dynamic_pointer_cast(pattern_map.at(weights_input_m).get_node_shared_ptr()); + auto fp_element_type = activations.get_element_type(); + + ov::Tensor new_weights_data(fp_element_type, weights->get_output_shape(0)); + auto src_ptr = static_cast(weights->get_data_ptr()); + auto size = ov::shape_size(weights->get_shape()); + switch (fp_element_type) { + case ov::element::f16: convert_packed_bin_to_fp(src_ptr, static_cast(new_weights_data.data()), size); break; + case ov::element::f32: convert_packed_bin_to_fp(src_ptr, static_cast(new_weights_data.data()), size); break; + default: return false; + } + + auto new_weights_const = std::make_shared(new_weights_data); + auto rank = activations.get_partial_shape().size(); + + auto in_lo = pattern_map.at(in_lo_m); + auto in_hi = pattern_map.at(in_hi_m); + auto out_lo = std::make_shared(fp_element_type, ov::Shape(rank, 1), std::vector{-1.0f}); + auto out_hi = std::make_shared(fp_element_type, ov::Shape(rank, 1), std::vector{1.0f}); + + auto new_fq = std::make_shared(activations, in_lo, in_hi, out_lo, out_hi, 2); + std::vector> result_nodes = { new_fq }; + + std::shared_ptr conv_input = new_fq; + auto pb = binary_conv->get_pads_begin(); + auto pe = binary_conv->get_pads_end(); + if (binary_conv->get_pad_value() != 0.0f) { + pb.insert(pb.begin(), rank - pb.size(), 0); + pe.insert(pe.begin(), rank - pe.size(), 0); + auto pad_b = std::make_shared(ov::element::i32, ov::Shape{pb.size()}, pb); + auto pad_e = std::make_shared(ov::element::i32, ov::Shape{pe.size()}, pe); + auto pad_v = std::make_shared(fp_element_type, ov::Shape{}, std::vector{binary_conv->get_pad_value()}); + auto pad = std::make_shared(new_fq, pad_b, pad_e, pad_v, ov::op::PadMode::CONSTANT); + conv_input = pad; + + pb = ov::CoordinateDiff(binary_conv->get_pads_begin().size(), 0); + pe = ov::CoordinateDiff(binary_conv->get_pads_end().size(), 0); + result_nodes.push_back(pad); + } + auto convolution = std::make_shared(conv_input, + new_weights_const, + binary_conv->get_strides(), + pb, + pe, + binary_conv->get_dilations(), + ov::op::PadType::EXPLICIT); + + result_nodes.push_back(convolution); + convolution->set_friendly_name(binary_conv->get_friendly_name()); + ov::copy_runtime_info(m.get_matched_nodes(), result_nodes); + ov::replace_node(binary_conv, convolution); + return true; + }; + + auto m = std::make_shared(binary_conv_m, "ConvertBinaryConvolutionToConvolution"); + this->register_matcher(m, callback); +} + +} // namespace intel_gpu +} // namespace ov diff --git a/src/plugins/intel_gpu/src/plugin/transformations/binary_conv_to_conv.hpp b/src/plugins/intel_gpu/src/plugin/transformations/binary_conv_to_conv.hpp new file mode 100644 index 00000000000..780d3b19d34 --- /dev/null +++ b/src/plugins/intel_gpu/src/plugin/transformations/binary_conv_to_conv.hpp @@ -0,0 +1,19 @@ +// Copyright (C) 2018-2023 Intel Corporation +// SPDX-License-Identifier: Apache-2.0 +// + +#pragma once + +#include "openvino/pass/graph_rewrite.hpp" + +namespace ov { +namespace intel_gpu { + +class ConvertBinaryConvolutionToConvolution: public ov::pass::MatcherPass { +public: + OPENVINO_RTTI("ConvertBinaryConvolutionToConvolution", "0"); + ConvertBinaryConvolutionToConvolution(); +}; + +} // namespace intel_gpu +} // namespace ov diff --git a/src/plugins/intel_gpu/src/plugin/transformations_pipeline.cpp b/src/plugins/intel_gpu/src/plugin/transformations_pipeline.cpp index 68268790c62..9d1bdbe01d2 100644 --- a/src/plugins/intel_gpu/src/plugin/transformations_pipeline.cpp +++ b/src/plugins/intel_gpu/src/plugin/transformations_pipeline.cpp @@ -116,6 +116,7 @@ #include "plugin/transformations/move_fc_reshape_to_weights.hpp" #include "plugin/transformations/convert_fc_to_compressed.hpp" #include "plugin/transformations/rms_fusion.hpp" +#include "plugin/transformations/binary_conv_to_conv.hpp" #include "transformations/low_precision/mark_dequantization_subgraph.hpp" #include "low_precision/pull_reshape_through_dequantization.hpp" @@ -260,6 +261,7 @@ void TransformationsPipeline::apply(std::shared_ptr func) { manager.register_pass(); } + manager.register_pass(); manager.register_pass(); manager.register_pass(); manager.register_pass(); diff --git a/src/plugins/intel_gpu/src/runtime/format.cpp b/src/plugins/intel_gpu/src/runtime/format.cpp index 095bda14bbf..db229fe2abe 100644 --- a/src/plugins/intel_gpu/src/runtime/format.cpp +++ b/src/plugins/intel_gpu/src/runtime/format.cpp @@ -48,7 +48,6 @@ static const std::map format_traits_map { FMT_TRAITS(bfuwzyx, 1, 1, 5, 0, {0, 1, 2, 3, 4, 5, 6}, "bfuwzyx", "bfxyzwu", {}), FMT_TRAITS(bfvuwzyx, 1, 1, 6, 0, {0, 1, 2, 3, 4, 5, 6, 7}, "bfvuwzyx", "bfxyzwuv", {}), FMT_TRAITS(fs_b_yx_fsv32, 1, 1, 2, 0, {1, 0, 2, 3}, "fbyx", "bfxy", {{1, 32}}), - FMT_TRAITS(b_fs_yx_32fp, 1, 1, 2, 0, {0, 1, 2, 3}, "bfyx", "bfxy", {{1, 32}}), FMT_TRAITS(b_fs_zyx_fsv16, 1, 1, 3, 0, {0, 1, 2, 3, 4}, "bfzyx", "bfxyz", {{1, 16}}), FMT_TRAITS(bs_fs_zyx_bsv16_fsv32, 1, 1, 3, 0, {0, 1, 2, 3, 4}, "bfzyx", "bfxyz", {{0, 16 }, {1, 32}}), FMT_TRAITS(bs_fs_zyx_bsv16_fsv16, 1, 1, 3, 0, {0, 1, 2, 3, 4}, "bfzyx", "bfxyz", {{0, 16 }, {1, 16}}), @@ -132,7 +131,6 @@ static const std::map format_traits_map { FMT_TRAITS(os_is_yx_osv32_isv4_swizzled_by_2, 1, 1, 2, 0, {0, 1, 2, 3}, "oiyx", "oixy", {{0, 32}, {1, 4}}), FMT_TRAITS(os_is_yx_osv32_isv4, 1, 1, 2, 0, {0, 1, 2, 3}, "oiyx", "oixy", {{0, 32}, {1, 4}}), FMT_TRAITS(os_is_zyx_osv32_isv4, 1, 1, 3, 0, {0, 1, 2, 3, 4}, "oizyx", "oixyz", {{0, 32}, {1, 4}}), - FMT_TRAITS(os_is_yx_osv32_isv32p, 1, 1, 1, 0, {0, 1, 2, 3}, "oiyx", "oixy", {{0, 32}, {1, 32}}), FMT_TRAITS(os_is_zyx_isv16_osv16, 1, 1, 3, 0, {0, 1, 2, 3, 4}, "oizyx", "oixyz", {{1, 16}, {0, 16}}), FMT_TRAITS(is_os_zyx_isv16_osv16, 1, 1, 3, 0, {1, 0, 2, 3, 4}, "iozyx", "oixyz", {{1, 16}, {0, 16}}), FMT_TRAITS(is_os_yx_osv8_isv4, 1, 1, 2, 0, {1, 0, 2, 3}, "ioyx", "oixy", {{0, 8}, {1, 4}}), @@ -294,8 +292,7 @@ format format::adjust_to_rank(format fmt, size_t new_rank) { auto is_adjustable = [](const format& fmt) -> bool { return !format::is_weights_format(fmt) && !format::is_image_2d(fmt) && - !format::is_winograd(fmt) && - fmt != format::b_fs_yx_32fp; + !format::is_winograd(fmt); }; // Skip special formats as order + blocking desc may be not enough to properly match them diff --git a/src/plugins/intel_gpu/src/runtime/layout.cpp b/src/plugins/intel_gpu/src/runtime/layout.cpp index a18c5bd7321..5534e9a1248 100644 --- a/src/plugins/intel_gpu/src/runtime/layout.cpp +++ b/src/plugins/intel_gpu/src/runtime/layout.cpp @@ -151,8 +151,6 @@ static format to_weights_format(format f, bool is_grouped) { return format::o_is_yx_isv16; case format::bs_fs_fsv8_bsv8: return format::os_i_osv8__ai8; - case format::b_fs_yx_32fp: - return format::os_is_yx_osv32_isv32p; default: throw std::invalid_argument("Unable to convert data format " + f.to_string() + " to weights format"); } @@ -377,11 +375,6 @@ size_t layout::get_linear_size() const { sizes[1] = align_to(sizes[1], 4); sizes[0] = align_to(sizes[0], 8); sizes[2] = align_to(sizes[2], 8); - } else if (this->format == cldnn::format::b_fs_yx_32fp) { - sizes[1] = align_to(sizes[1], 32); - } else if (this->format == cldnn::format::os_is_yx_osv32_isv32p) { - sizes[0] = align_to(sizes[0], 32); - sizes[1] = align_to(sizes[1], 32); } else if (this->format == cldnn::format::image_2d_rgba) { sizes[1] = 4; } else if (this->format == cldnn::format::gs_oi_yxs_gsv4_yxsv4 || diff --git a/src/plugins/intel_gpu/tests/unit/fusions/binary_convolution_fusion_test.cpp b/src/plugins/intel_gpu/tests/unit/fusions/binary_convolution_fusion_test.cpp deleted file mode 100644 index 5079e06e0e8..00000000000 --- a/src/plugins/intel_gpu/tests/unit/fusions/binary_convolution_fusion_test.cpp +++ /dev/null @@ -1,277 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "fusion_test_common.hpp" - -#include -#include -#include -#include -#include -#include - -#include - -using namespace cldnn; -using namespace ::tests; - -namespace { - -struct binary_convolution_test_params { - tensor in_shape; - tensor out_shape; - tensor kernel; - ov::Strides stride; - ov::CoordinateDiff pad; - ov::Strides dilation; - uint32_t groups; - data_types data_type; - format input_format; - data_types weights_type; - format weights_format; - data_types default_type; - format default_format; - size_t expected_fused_primitives; - size_t expected_not_fused_primitives; -}; - -class BinaryConvolutionFusingTest : public BaseFusingTest { -public: - void execute(binary_convolution_test_params& p) { - auto input_prim = get_mem(get_input_layout(p)); - network network_not_fused(this->engine, this->topology_non_fused, cfg_not_fused); - network network_fused(this->engine, this->topology_fused, cfg_fused); - network_fused.set_input_data("input", input_prim); - network_not_fused.set_input_data("input", input_prim); - - compare(network_not_fused, network_fused, p); - auto find_conv = [](primitive_info& p) -> bool { - if (p.original_id == "conv_prim") - return true; - return false; - }; - - auto pi_fused = network_fused.get_primitives_info(); - auto info_fused = std::find_if(pi_fused.begin(), pi_fused.end(), find_conv); - if (info_fused != pi_fused.end()) - std::cout << "kernel: " << info_fused->kernel_id << std::endl; - } - - layout get_input_layout(binary_convolution_test_params& p) { - auto pad = p.pad; - std::vector pad_ = { 0, 0, static_cast(pad[1]), static_cast(pad[0]) }; - return layout{ p.data_type, p.input_format, p.in_shape, padding{ pad_ } }; - } - - layout get_per_channel_layout(binary_convolution_test_params& p) { - return layout{ p.default_type, p.default_format, tensor{1, p.out_shape.feature[0], 1, 1} }; - } -}; - -} // namespace - -#define CASE_BIN_CONV1 { 1, 16, 4, 5 }, { 1, 16, 4, 5 }, { 1, 1, 3, 3 }, { 1, 1 }, { 1, 1 }, { 1, 1 }, 1, data_types::u1, format::b_fs_yx_32fp, data_types::u1, format::os_is_yx_osv32_isv32p, data_types::f32, format::bfyx -#define CASE_BIN_CONV2 { 1, 16, 4, 5 }, { 1, 30, 4, 5 }, { 1, 1, 1, 1 }, { 1, 1 }, { 0, 0 }, { 1, 1 }, 1, data_types::u1, format::b_fs_yx_32fp, data_types::u1, format::os_is_yx_osv32_isv32p, data_types::f32, format::bfyx -#define CASE_BIN_CONV3 { 1, 184, 12, 21 }, { 1, 224, 12, 21 }, { 1, 1, 1, 1 }, { 1, 1 }, { 0, 0 }, { 1, 1 }, 1, data_types::u1, format::b_fs_yx_32fp, data_types::u1, format::os_is_yx_osv32_isv32p, data_types::f32, format::bfyx - -/* ----------------------------------------------------------------------------------------------------- */ -/* -------------------------------------- binary convolution cases ------------------------------------- */ -/* ----------------------------------------------------------------------------------------------------- */ - -class conv_bin_activation : public BinaryConvolutionFusingTest {}; -TEST_P(conv_bin_activation, basic) { - auto p = GetParam(); - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - activation("activation", input_info("bin_conv_prim"), activation_func::relu), - reorder("reorder_bfyx", input_info("activation"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -INSTANTIATE_TEST_SUITE_P(fusings_gpu, conv_bin_activation, ::testing::ValuesIn(std::vector{ - binary_convolution_test_params{ CASE_BIN_CONV1, 2, 3 }, -})); - -class conv_bin_scale_activation : public BinaryConvolutionFusingTest {}; -TEST_P(conv_bin_scale_activation, basic) { - auto p = GetParam(); - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - data("scale_data", get_mem(get_per_channel_layout(p), 1.0f/p.kernel.count())), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - eltwise("scale", { input_info("bin_conv_prim"), input_info("scale_data") }, eltwise_mode::prod, p.default_type), - activation("activation", input_info("scale"), activation_func::relu), - reorder("reorder_bfyx", input_info("activation"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -INSTANTIATE_TEST_SUITE_P(fusings_gpu, conv_bin_scale_activation, ::testing::ValuesIn(std::vector{ - binary_convolution_test_params{ CASE_BIN_CONV1, 2, 4 }, - binary_convolution_test_params{ CASE_BIN_CONV2, 2, 4 }, -})); - -class conv_bin_quantize_bin : public BinaryConvolutionFusingTest {}; -TEST_P(conv_bin_quantize_bin, channel_wise_quantize) { - auto p = GetParam(); - auto in_thresh = get_mem(get_per_channel_layout(p), min_random, max_random); - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - data("in_lo", in_thresh), - data("in_hi", in_thresh), - data("out_lo", get_mem(get_per_channel_layout(p), -1)), - data("out_hi", get_mem(get_per_channel_layout(p), 1)), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - quantize("quantize_data", input_info("bin_conv_prim"), input_info("in_lo"), input_info("in_hi"), - input_info("out_lo"), input_info("out_hi"), 2, data_types::u1), - reorder("reorder_bfyx", input_info("quantize_data"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -TEST_P(conv_bin_quantize_bin, blob_wise_quantize) { - auto p = GetParam(); - auto in_thresh = get_mem(get_single_element_layout(p), min_random, max_random); - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - data("in_lo", in_thresh), - data("in_hi", in_thresh), - data("out_lo", get_mem(get_single_element_layout(p), -1)), - data("out_hi", get_mem(get_single_element_layout(p), 1)), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - quantize("quantize_data", input_info("bin_conv_prim"), input_info("in_lo"), input_info("in_hi"), - input_info("out_lo"), input_info("out_hi"), 2, data_types::u1), - reorder("reorder_bfyx", input_info("quantize_data"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -INSTANTIATE_TEST_SUITE_P(fusings_gpu, conv_bin_quantize_bin, ::testing::ValuesIn(std::vector{ - binary_convolution_test_params{ CASE_BIN_CONV1, 2, 3 }, - binary_convolution_test_params{ CASE_BIN_CONV2, 2, 3 }, -})); - -class conv_bin_scale_conv_dw : public BinaryConvolutionFusingTest {}; -TEST_P(conv_bin_scale_conv_dw, dw_kernel_3x3_stride2) { - auto p = GetParam(); - auto dw_tensor = cldnn::tensor(group(p.out_shape.feature[0]), batch(1), feature(1), spatial(3, 3)); - auto dw_weights_layout = layout{ p.default_type, format::goiyx, dw_tensor }; - - ov::Strides dw_stride = {2, 2}; - ov::Strides dw_dilation = {1, 1}; - ov::CoordinateDiff dw_pad = p.pad; - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - data("weights_dw", get_mem(dw_weights_layout, -127, 127)), - data("scale_data", get_mem(get_per_channel_layout(p), 1e-1f)), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - eltwise("scale", { input_info("bin_conv_prim"), input_info("scale_data") }, eltwise_mode::prod, p.default_type), - convolution("conv_dw", input_info("scale"), "weights_dw", "", p.out_shape.feature[0], dw_stride, dw_dilation, dw_pad, dw_pad, true), - reorder("reorder_bfyx", input_info("conv_dw"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -TEST_P(conv_bin_scale_conv_dw, dw_kernel_3x3_stride1) { - auto p = GetParam(); - auto dw_tensor = cldnn::tensor(group(p.out_shape.feature[0]), batch(1), feature(1), spatial(3, 3)); - auto dw_weights_layout = layout{ p.default_type, format::goiyx, dw_tensor }; - - ov::Strides dw_stride = {1, 1}; - ov::Strides dw_dilation = {1, 1}; - ov::CoordinateDiff dw_pad = p.pad; - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - data("weights_dw", get_mem(dw_weights_layout, -127, 127)), - data("scale_data", get_mem(get_per_channel_layout(p), 1e-1f)), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - eltwise("scale", { input_info("bin_conv_prim"), input_info("scale_data") }, eltwise_mode::prod, p.default_type), - convolution("conv_dw", input_info("scale"), "weights_dw", "", p.out_shape.feature[0], dw_stride, dw_dilation, dw_pad, dw_pad, true), - reorder("reorder_bfyx", input_info("conv_dw"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -INSTANTIATE_TEST_SUITE_P(fusings_gpu, conv_bin_scale_conv_dw, ::testing::ValuesIn(std::vector{ - binary_convolution_test_params{ CASE_BIN_CONV2, 3, 4 }, - binary_convolution_test_params{ CASE_BIN_CONV3, 3, 4 }, -})); - -class conv_bin_scale_conv_dw_prelu : public BinaryConvolutionFusingTest {}; -TEST_P(conv_bin_scale_conv_dw_prelu, dw_kernel_3x3_stride2) { - auto p = GetParam(); - auto dw_tensor = cldnn::tensor(group(p.out_shape.feature[0]), batch(1), feature(1), spatial(3, 3)); - auto dw_weights_layout = layout{ p.default_type, format::goiyx, dw_tensor }; - - ov::Strides dw_stride = {2, 2}; - ov::Strides dw_dilation = {1, 1}; - ov::CoordinateDiff dw_pad = p.pad; - auto in_thresh = get_mem(get_per_channel_layout(p), min_random, max_random); - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - data("weights_dw", get_mem(dw_weights_layout, -127, 127)), - data("scale_data", get_mem(get_per_channel_layout(p), 1e-1f)), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - eltwise("scale", { input_info("bin_conv_prim"), input_info("scale_data") }, eltwise_mode::prod, p.default_type), - convolution("conv_dw", input_info("scale"), "weights_dw", "", p.out_shape.feature[0], dw_stride, dw_dilation, dw_pad, dw_pad, true), - data("slope_data", get_mem(get_per_channel_layout(p))), - activation("activation", input_info("conv_dw"), "slope_data", activation_func::relu_negative_slope), - reorder("reorder_bfyx", input_info("activation"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -TEST_P(conv_bin_scale_conv_dw_prelu, dw_kernel_3x3_stride1) { - auto p = GetParam(); - auto dw_tensor = cldnn::tensor(group(p.out_shape.feature[0]), batch(1), feature(1), spatial(3, 3)); - auto dw_weights_layout = layout{ p.default_type, format::goiyx, dw_tensor }; - - ov::Strides dw_stride = {1, 1}; - ov::Strides dw_dilation = {1, 1}; - ov::CoordinateDiff dw_pad = p.pad; - auto in_thresh = get_mem(get_per_channel_layout(p), min_random, max_random); - create_topologies( - input_layout("input", get_input_layout(p)), - data("weights", get_mem(get_weights_layout(p), -127, 127)), - data("weights_dw", get_mem(dw_weights_layout, -127, 127)), - data("scale_data", get_mem(get_per_channel_layout(p), 1e-1f)), - binary_convolution("bin_conv_prim", input_info("input"), { "weights" }, p.stride, p.pad, p.dilation, p.out_shape, p.groups), - eltwise("scale", { input_info("bin_conv_prim"), input_info("scale_data") }, eltwise_mode::prod, p.default_type), - convolution("conv_dw", input_info("scale"), "weights_dw", "", p.out_shape.feature[0], dw_stride, dw_dilation, dw_pad, dw_pad, true), - data("slope_data", get_mem(get_per_channel_layout(p))), - activation("activation", input_info("conv_dw"), "slope_data", activation_func::relu_negative_slope), - reorder("reorder_bfyx", input_info("activation"), p.default_format, data_types::f32) - ); - - tolerance = 1e-5f; - execute(p); -} - -INSTANTIATE_TEST_SUITE_P(fusings_gpu, conv_bin_scale_conv_dw_prelu, ::testing::ValuesIn(std::vector{ - binary_convolution_test_params{ CASE_BIN_CONV2, 3, 5 }, - binary_convolution_test_params{ CASE_BIN_CONV3, 3, 5 }, -})); diff --git a/src/plugins/intel_gpu/tests/unit/fusions/fusion_test_common.hpp b/src/plugins/intel_gpu/tests/unit/fusions/fusion_test_common.hpp index 58694f91d4c..06c6f7aac1c 100644 --- a/src/plugins/intel_gpu/tests/unit/fusions/fusion_test_common.hpp +++ b/src/plugins/intel_gpu/tests/unit/fusions/fusion_test_common.hpp @@ -114,10 +114,7 @@ public: cldnn::memory::ptr get_mem(cldnn::layout l) { auto prim = engine.allocate_memory(l); tensor s = l.get_tensor(); - if (l.data_type == data_types::u1) { - VF rnd_vec = rg.generate_random_1d(s.count() / 32, min_random, max_random); - set_values(prim, rnd_vec); - } else if (l.data_type == data_types::i8 || l.data_type == data_types::u8) { + if (l.data_type == data_types::i8 || l.data_type == data_types::u8) { VF rnd_vec = rg.generate_random_1d(s.count(), min_random, max_random); set_values(prim, rnd_vec); } else if (l.data_type == data_types::f16) { @@ -134,10 +131,7 @@ public: cldnn::memory::ptr get_mem(cldnn::layout l, float fill_value) { auto prim = engine.allocate_memory(l); tensor s = l.get_tensor(); - if (l.data_type == data_types::u1) { - VF rnd_vec(s.count() / 32, static_cast(fill_value)); - set_values(prim, rnd_vec); - } else if (l.data_type == data_types::f16) { + if (l.data_type == data_types::f16) { VF rnd_vec(s.count(), ov::float16(fill_value).to_bits()); set_values(prim, rnd_vec); } else if (l.data_type == data_types::f32) { @@ -169,10 +163,6 @@ public: VF rnd_vec = rg.generate_random_norepetitions(s.count(), min, max); set_values(prim, rnd_vec); } - else if (l.data_type == data_types::u1) { - VF rnd_vec = rg.generate_random_norepetitions(s.count(), min, max); - set_values(prim, rnd_vec); - } return prim; } @@ -192,9 +182,6 @@ public: } else if (l.data_type == data_types::u8) { VF rnd_vec = rg.generate_random_1d(s.count(), min, max); set_values(prim, rnd_vec); - } else if (l.data_type == data_types::u1) { - VF rnd_vec = rg.generate_random_1d(s.count() / 32, min, max); - set_values(prim, rnd_vec); } return prim; diff --git a/src/plugins/intel_gpu/tests/unit/shape_infer/quantize_si_test.cpp b/src/plugins/intel_gpu/tests/unit/shape_infer/quantize_si_test.cpp index 342386aae47..c095f486c93 100644 --- a/src/plugins/intel_gpu/tests/unit/shape_infer/quantize_si_test.cpp +++ b/src/plugins/intel_gpu/tests/unit/shape_infer/quantize_si_test.cpp @@ -72,10 +72,6 @@ INSTANTIATE_TEST_SUITE_P(smoke, quantize_test, layout{ov::PartialShape{1, 2, 3, 4}, data_types::f32, format::bfyx}, layout{ov::PartialShape{1, 2, 3, 4}, data_types::i8, format::bfyx} }, - { - layout{ov::PartialShape{1, 2, 3, 4}, data_types::f32, format::bfyx}, - layout{ov::PartialShape{1, 2, 3, 4}, data_types::u1, format::b_fs_yx_32fp} - }, { layout{ov::PartialShape{1, 2, 3, 4, 5}, data_types::f32, format::bfzyx}, layout{ov::PartialShape{1, 2, 3, 4, 5}, data_types::u8, format::bfzyx} diff --git a/src/plugins/intel_gpu/tests/unit/test_cases/binary_convolution_gpu_test.cpp b/src/plugins/intel_gpu/tests/unit/test_cases/binary_convolution_gpu_test.cpp deleted file mode 100644 index 9561f7a94f4..00000000000 --- a/src/plugins/intel_gpu/tests/unit/test_cases/binary_convolution_gpu_test.cpp +++ /dev/null @@ -1,495 +0,0 @@ -// Copyright (C) 2018-2023 Intel Corporation -// SPDX-License-Identifier: Apache-2.0 -// - -#include "test_utils.h" - -#include -#include -#include -#include - -#include - -using namespace cldnn; -using namespace ::tests; - -// Batch, groups, IC, IW, IH, OC, OW, OH, KH, KW, SH, SW, PH, PW -struct TestParams { - int b; - int g; - - int ic; - int ih; - int iw; - - int oc; - int oh; - int ow; - - int kh; - int kw; - - int sh; - int sw; - - int ph; - int pw; - - float pad_value; - data_types dt; - std::string name; - bool is_caching_test; - - bool isConsistent() const - { - bool res = true; - - res &= (((iw - kw + 2*pw) / sw + 1) == ow); - res &= (((ih - kh + 2*ph) / sh + 1) == oh); - return res; - } - - friend ::std::ostream& operator<<(::std::ostream& os, const TestParams& p) { - return os << "Params: [ b=" << p.b - << "; g=" << p.g - << "; src=[" << p.ic << "; " << p.ih << "; " << p.iw << "]" - << "; dst=[" << p.oc << "; " << p.oh << "; " << p.ow << "]" - << "; k=[" << p.kh << "; " << p.kw << "]" - << "; stride=[" << p.sh << "; " << p.sw << "]" - << "; pad=[" << p.ph << "; " << p.pw << "]" - << "; pad_value=" << p.pad_value - << "; name=" << p.name - << "; is_caching_test=" << p.is_caching_test - << "]"; - } - friend void PrintTo(const TestParams& p, ::std::ostream* os) { - *os << p; - } -}; - -static void fill(cldnn::memory::ptr mem) { - cldnn::mem_lock ptr(mem, get_test_stream()); - for (size_t i = 0; i < div_up(mem->get_layout().count(), 32); i++) { - ptr[i] = (uint32_t)rand() % (1 << 31); - } -} - -template -void compute_ref_conv_bin(const cldnn::memory::ptr src, - const cldnn::memory::ptr weights, - cldnn::memory::ptr dst, - TestParams &p) { - - cldnn::mem_lock src_data(src, get_test_stream()); - cldnn::mem_lock weights_data(weights, get_test_stream()); - cldnn::mem_lock dst_data(dst, get_test_stream()); - - int pack_size = sizeof(data_t_src) * 8; - - int B = p.b; - int NG = p.g; - int IC = p.ic; - int IH = p.ih; - int IW = p.iw; - - int OC = p.oc; - int OH = p.oh; - int OW = p.ow; - - int KH = p.kh; - int KW = p.kw; - - int SH = p.sh; - int SW = p.sw; - - int PH = p.ph; - int PW = p.pw; - - auto extract_bit = [&](data_t_src val, data_t_src bit) -> data_t_src { - return (data_t_src)((val >> bit) & 0x1); - }; - - auto ker = [&](data_t_acc &d, int g, int mb, int oc,int oh, int ow, int& ks) { - for (int ic = 0; ic < IC / NG; ++ic) { - for (int kh = 0; kh < KH; ++kh) - for (int kw = 0; kw < KW; ++kw) { - const int ih = oh * SH - PH + kh; - const int iw = ow * SW - PW + kw; - - int widx = g * OC / NG *IC / NG * KH * KW - + oc * IC / NG * KH * KW - + ic * KH * KW - + kh * KW - + kw; - int iidx = -1; - uint8_t w = extract_bit(weights_data[widx / pack_size], widx % pack_size); - uint8_t s = 0; - - if ((ih < 0 || ih >= IH || iw < 0 || iw >= IW)) { - if (p.pad_value == 0.0f) - continue; - else - s = (p.pad_value == -1.0f) ? 0 : 1; - } else { - if (ic == 0) ks++; - iidx = mb * div_up(IC, pack_size) * IH * IW - + g * div_up(IC, pack_size) / NG * IH * IW - + (ic/pack_size) * IH * IW - + ih * IW - + iw; - - s = extract_bit(src_data[iidx], ic % pack_size); - } - d += (data_t_acc)(s ^ w); - } - } - }; - - for (int g = 0; g < NG; g++) { - for (int b = 0; b < B; b++) { - for (int oc = 0; oc < OC / NG; oc++) { - for (int oh = 0; oh < OH; oh++) { - for (int ow = 0; ow < OW; ow++) { - data_t_acc a = 0; - int ks = 0; - ker(a, g, b, oc, oh, ow, ks); - int dst_off = b * OC * OH* OW - + g * OC / NG * OH * OW - + oc * OH * OW - + oh * OW - + ow; - if (p.pad_value == 0.0f) - dst_data[dst_off] =(data_t_dst)(IC*ks - 2*a); - else - dst_data[dst_off] = (data_t_dst)(IC*KH*KW - 2*a); - } - } - } - } - } -} - -class binary_convolution_test : public ::testing::TestWithParam { - void SetUp() override { - std::cout << GetParam() << std::endl; - ASSERT_TRUE(GetParam().isConsistent()); - } -}; - -TEST_P(binary_convolution_test, conv) { - auto& engine = get_test_engine(); - - // DG2 is not validated for binary convolution: https://github.com/openvinotoolkit/openvino/pull/12486 - if(engine.get_device_info().supports_immad) - return; - - ov::intel_gpu::ExecutionConfig config = get_test_default_config(engine); - config.set_property(ov::intel_gpu::optimize_data(true)); - topology topology_bin; - - std::string weights_suffix = "_w_"; - - std::string input_name = "input"; - std::string output_name = "conv"; - - TestParams p = GetParam(); - - ov::Strides stride = {static_cast(p.sh), static_cast(p.sw)}; - ov::CoordinateDiff pad = {p.ph, p.pw}; - ov::Strides dilation = {1,1}; - - cldnn::tensor is_size{ cldnn::batch(p.b), - cldnn::feature(p.ic), - cldnn::spatial(p.iw, p.ih) }; - cldnn::tensor wei_size{ cldnn::batch(p.oc), - cldnn::feature(p.ic), - cldnn::spatial(p.kw, p.kh) }; - cldnn::tensor os_size{ cldnn::batch(p.b), - cldnn::feature(p.oc), - cldnn::spatial(p.ow, p.oh)}; - - auto input = engine.allocate_memory({ cldnn::data_types::u1, cldnn::format::b_fs_yx_32fp, is_size }); - auto weights = engine.allocate_memory({ cldnn::data_types::u1, cldnn::format::bfyx, wei_size }); - auto output_ref = engine.allocate_memory({ cldnn::data_types::f32, cldnn::format::bfyx, os_size }); - - fill(input); - fill(weights); - - compute_ref_conv_bin(input, weights, output_ref, p); - -// print_bin_blob(input,"input"); -// print_bin_blob_packed(input,"input"); -// print_bin_blob(weights, "weights"); -// print_blob(output_ref, "ref_out"); - - topology_bin.add(input_layout(input_name, input->get_layout())); - topology_bin.add(data(output_name + weights_suffix, weights)); - - topology_bin.add(binary_convolution(output_name, input_info(input_name), {output_name + weights_suffix}, - stride, pad, dilation, os_size, 1, p.pad_value, p.dt)); - - cldnn::network::ptr network_bin = get_network(engine, topology_bin, config, get_test_stream_ptr(), p.is_caching_test); - - network_bin->set_input_data(input_name, input); - - std::map outputs = network_bin->execute(); - auto outputMemory = outputs.at(output_name).get_memory(); - - for (size_t i = 0; i < output_ref->count(); i++) { - if (p.dt == data_types::f32) { - cldnn::mem_lock ref(output_ref, get_test_stream()); - cldnn::mem_lock opt(outputMemory, get_test_stream()); - - ASSERT_EQ(ref[i], opt[i]) << i; - } else if (p.dt == data_types::f16) { - cldnn::mem_lock ref(output_ref, get_test_stream()); - cldnn::mem_lock opt(outputMemory, get_test_stream()); - - ASSERT_EQ(ref[i], half_to_float(opt[i])) << i; - } - } -} - -// Batch, groups, IC, IW, IH, OC, OW, OH, KH, KW, SH, SW, PH, PW -INSTANTIATE_TEST_SUITE_P(BinaryConvTest, binary_convolution_test, ::testing::Values( - TestParams{1, 1, 16,2,2, 4,2,2, 3,3, 1,1, 1,1, -1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 17,2,2, 4,2,2, 3,3, 1,1, 1,1, -1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 17,2,2, 4,2,2, 3,3, 1,1, 1,1, 0.0f, data_types::f32, "small", false}, - TestParams{1, 1, 17,2,2, 4,2,2, 3,3, 1,1, 1,1, 1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 16,2,2, 16,2,2, 3,3, 1,1, 1,1, 1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 32,2,2, 32,2,2, 3,3, 1,1, 1,1, 1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 32,2,2, 32,2,2, 1,1, 1,1, 0,0, 1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 128,2,2, 128,2,2, 1,1, 1,1, 0,0, -1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 16,4,3, 4,4,3, 1,1, 1,1, 0,0, -1.0f, data_types::f32, "small", false}, - TestParams{1, 1, 16,2,2, 4,2,2, 3,3, 1,1, 1,1, -1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 17,2,2, 4,2,2, 3,3, 1,1, 1,1, -1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 17,2,2, 4,2,2, 3,3, 1,1, 1,1, 0.0f, data_types::f16, "small", false}, - TestParams{1, 1, 17,2,2, 4,2,2, 3,3, 1,1, 1,1, 1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 16,2,2, 16,2,2, 3,3, 1,1, 1,1, 1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 32,2,2, 32,2,2, 3,3, 1,1, 1,1, 1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 32,2,2, 32,2,2, 1,1, 1,1, 0,0, 1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 128,2,2, 128,2,2, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 16,4,3, 4,4,3, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 9,16,32, 17,8,16, 7,7, 2,2, 3,3, -1.0f, data_types::f16, "small", false}, - TestParams{1, 1, 9,16,32, 17,8,16, 7,7, 2,2, 3,3, 1.0f, data_types::f16, "small", false}, - - // Resnet-18 3x3 - TestParams{1, 1, 64,56,56, 64,56,56, 3,3, 1,1, 1,1, -1.0f, data_types::f16, "resnet18_0", false}, - TestParams{1, 1, 64,56,56, 128,28,28, 3,3, 2,2, 1,1, -1.0f, data_types::f16, "resnet18_1", false}, - TestParams{1, 1, 128,28,28, 128,28,28, 3,3, 1,1, 1,1, -1.0f, data_types::f16, "resnet18_2", false}, - TestParams{1, 1, 128,28,28, 256,14,14, 3,3, 2,2, 1,1, -1.0f, data_types::f16, "resnet18_3", false}, - TestParams{1, 1, 256,14,14, 256,14,14, 3,3, 1,1, 1,1, -1.0f, data_types::f16, "resnet18_4", false}, - TestParams{1, 1, 256,14,14, 512, 7, 7, 3,3, 2,2, 1,1, -1.0f, data_types::f16, "resnet18_5", false}, - TestParams{1, 1, 512, 7, 7, 512, 7, 7, 3,3, 1,1, 1,1, -1.0f, data_types::f16, "resnet18_6", false}, - // Resnet-50 - TestParams{1, 1, 64,56,56, 64,56,56, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_0", false}, - TestParams{1, 1, 64,56,56, 256,56,56, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_1", false}, - TestParams{1, 1, 256,56,56, 128,28,28, 1,1, 2,2, 0,0, -1.0f, data_types::f16, "resnet50_2", false}, - TestParams{1, 1, 128,28,28, 512,28,28, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_3", false}, - TestParams{1, 1, 512,28,28, 128,28,28, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_4", false}, - TestParams{1, 1, 512,28,28, 256,14,14, 1,1, 2,2, 0,0, -1.0f, data_types::f16, "resnet50_5", false}, - TestParams{1, 1, 256,14,14, 1024,14,14, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_6", false}, - TestParams{1, 1, 1024,14,14, 256,14,14, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_7", false}, - TestParams{1, 1, 1024,14,14, 512,7,7, 1,1, 2,2, 0,0, -1.0f, data_types::f16, "resnet50_8", false}, - TestParams{1, 1, 512,7,7, 2048,7,7, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_9", false}, - TestParams{1, 1, 2048,7,7, 512,7,7, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "resnet50_10", false}, - // Mobilenet-ssd-vd - TestParams{1, 1, 56,96,168, 112,96,168, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv2_2_sep_BIN", false}, // back_bone_seq_conv2_2_sep_BIN - TestParams{1, 1, 112,96,168, 112,96,168, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv3_1_sep_BIN", false}, // back_bone_seq_conv3_1_sep_BIN - TestParams{1, 1, 112,48,84, 208,48, 84, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv3_2_sep_BIN", false}, // back_bone_seq_conv3_2_sep_BIN - TestParams{1, 1, 208,48,84, 216,48, 84, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv4_1_sep_BIN", false}, // back_bone_seq_conv4_1_sep_BIN - TestParams{1, 1, 216,24,42, 328,24, 42, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv4_2_sep_BIN", false}, // back_bone_seq_conv4_2_sep_BIN - TestParams{1, 1, 328,24,42, 288,24, 42, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv5_1_sep_BIN", false}, // back_bone_seq_conv5_1_sep_BIN - TestParams{1, 1, 288,24,42, 288,24, 42, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv5_2_sep_BIN", false}, // back_bone_seq_conv5_2_sep_BIN - TestParams{1, 1, 288,24,42, 240,24, 42, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv5_3_sep_BIN", false}, // back_bone_seq_conv5_3_sep_BIN - TestParams{1, 1, 240,24,42, 264,24, 42, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv5_4_sep_BIN", false}, // back_bone_seq_conv5_4_sep_BIN - TestParams{1, 1, 264,24,42, 192,24, 42, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv5_5_sep_BIN", false}, // back_bone_seq_conv5_5_sep_BIN - TestParams{1, 1, 192,12,21, 208,12, 21, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv5_6_sep_BIN", false}, // back_bone_seq_conv5_6_sep_BIN - TestParams{1, 1, 208,12,21, 88,12, 21, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv6_sep_BN", false} // back_bone_seq_conv6_sep_BN -)); - -INSTANTIATE_TEST_SUITE_P(export_import, binary_convolution_test, ::testing::Values( - TestParams{1, 1, 208,12,21, 88,12, 21, 1,1, 1,1, 0,0, -1.0f, data_types::f16, "conv6_sep_BN", true} -)); - -template -static void set_binary_values(cldnn::memory::ptr mem, std::vector args) { - cldnn::mem_lock ptr(mem, get_test_stream()); - - auto it = ptr.begin(); - for (auto x : args) - *it++ = x; -} - -TEST(binary_convolution, basic_convolution_1x1_single_packed_channel) { - auto& engine = get_test_engine(); - // DG2 is not validated for binary convolution: https://github.com/openvinotoolkit/openvino/pull/12486 - if(engine.get_device_info().supports_immad) - return; - - auto input = engine.allocate_memory({ data_types::u1, format::b_fs_yx_32fp, { 1, 16, 2, 2 } }); - auto weights = engine.allocate_memory({ data_types::u1, format::bfyx, { 4, 16, 1, 1 } }); - - // 0 0 1 0 0 1 0 0 1 0 1 0 1 0 1 0 - // 1 0 0 0 0 1 1 0 0 1 1 0 1 0 1 0 - // 1 1 0 0 1 0 1 1 1 1 1 1 1 0 1 0 - // 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 1 - set_binary_values(input, { 21796, 22113, 24531, 32768 }); - - // 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 - // 0 1 0 1 0 1 0 1 1 0 1 0 1 0 1 0 - // 1 0 1 0 1 0 1 0 0 1 0 1 0 1 0 1 - // 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 - set_binary_values(weights, { 65535, 21930, 43605, 0 }); - - // 16 - 2*popcount(1 1 0 1 1 0 1 1 0 1 0 1 0 1 0 1) = -4 - // 16 - 2*popcount(0 1 1 1 1 0 0 1 1 0 0 1 0 1 0 1) = -2 - // 16 - 2*popcount(0 0 1 1 0 1 0 0 0 0 0 0 0 1 0 1) = 6 - // 16 - 2*popcount(1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 0) = -14 - - // 16 - 2*popcount(0 1 1 1 0 0 0 1 0 0 0 0 0 0 0 0) = 8 - // 16 - 2*popcount(1 1 0 1 0 0 1 1 1 1 0 0 0 0 0 0) = 2 - // 16 - 2*popcount(1 0 0 1 1 1 1 0 0 1 0 1 0 0 0 0) = 2 - // 16 - 2*popcount(0 1 0 1 0 1 0 1 1 0 1 0 1 0 1 1) = -2 - - // 16 - 2*popcount(1 0 0 0 1 1 1 0 1 1 1 1 1 1 1 1) = -8 - // 16 - 2*popcount(0 0 1 0 1 1 0 0 0 0 1 1 1 1 1 1) = -2 - // 16 - 2*popcount(0 1 1 0 0 0 0 1 1 0 1 0 1 1 1 1) = -2 - // 16 - 2*popcount(1 0 1 0 1 0 1 0 0 1 0 1 0 1 0 0) = 2 - - // 16 - 2*popcount(0 0 1 0 0 1 0 0 1 0 1 0 1 0 1 0) = 4 - // 16 - 2*popcount(1 0 0 0 0 1 1 0 0 1 1 0 1 0 1 0) = 2 - // 16 - 2*popcount(1 1 0 0 1 0 1 1 1 1 1 1 1 0 1 0) = -6 - // 16 - 2*popcount(0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 1) = 14 - VF output_vec = { - -4.0f, -2.0f, 6.0f, -14.0f, - 8.0f, 2.0f, 2.0f, -2.0f, - -8.0f, -2.0f, -2.0f, 2.0f, - 4.0f, 2.0f, -6.0f, 14.0f }; - - topology topology( - input_layout("input", input->get_layout()), - data("weights", weights), - binary_convolution("binary_conv", input_info("input"), { "weights" }, - { 1,1 }, - { 0,0 }, - { 1,1 }, - { 1,4,2,2 }, - 0, 0.0f, - data_types::f32, - padding{ { 0,0,0,0 }, 0 }) - ); - - ov::intel_gpu::ExecutionConfig config = get_test_default_config(engine); - config.set_property(ov::intel_gpu::optimize_data(true)); - - network network(engine, topology, config); - network.set_input_data("input", input); - - auto outputs = network.execute(); - ASSERT_EQ(outputs.size(), size_t(1)); - ASSERT_EQ(outputs.begin()->first, "binary_conv"); - - auto output_memory = outputs.at("binary_conv").get_memory(); - auto output_layout = output_memory->get_layout(); - cldnn::mem_lock output_ptr(output_memory, get_test_stream()); - - ASSERT_EQ(output_layout.format, format::bfyx); - ASSERT_EQ(output_layout.data_type, data_types::f32); - ASSERT_EQ(output_layout.batch(), 1); - ASSERT_EQ(output_layout.feature(), 4); - ASSERT_EQ(output_layout.spatial(1), 2); - ASSERT_EQ(output_layout.spatial(0), 2); - - for (size_t i = 0; i < output_layout.count(); i++) - { - ASSERT_EQ(output_ptr[i], output_vec[i]) << "index="<< i; - } -} - -TEST(binary_convolution, basic_convolution_1x1_single_packed_channel_fp16) { - auto& engine = get_test_engine(); - // DG2 is not validated for binary convolution: https://github.com/openvinotoolkit/openvino/pull/12486 - if(engine.get_device_info().supports_immad) - return; - - auto input = engine.allocate_memory({ data_types::u1, format::b_fs_yx_32fp, { 1, 16, 2, 2 } }); - auto weights = engine.allocate_memory({ data_types::u1, format::bfyx, { 4, 16, 1, 1 } }); - - // 0 0 1 0 0 1 0 0 1 0 1 0 1 0 1 0 - // 1 0 0 0 0 1 1 0 0 1 1 0 1 0 1 0 - // 1 1 0 0 1 0 1 1 1 1 1 1 1 0 1 0 - // 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 1 - set_binary_values(input, { 21796, 22113, 24531, 32768 }); - - // 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 - // 0 1 0 1 0 1 0 1 1 0 1 0 1 0 1 0 - // 1 0 1 0 1 0 1 0 0 1 0 1 0 1 0 1 - // 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 - set_binary_values(weights, { 65535, 21930, 43605, 0 }); - - // 16 - 2*popcount(1 1 0 1 1 0 1 1 0 1 0 1 0 1 0 1) = -4 - // 16 - 2*popcount(0 1 1 1 1 0 0 1 1 0 0 1 0 1 0 1) = -2 - // 16 - 2*popcount(0 0 1 1 0 1 0 0 0 0 0 0 0 1 0 1) = 6 - // 16 - 2*popcount(1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 0) = -14 - - // 16 - 2*popcount(0 1 1 1 0 0 0 1 0 0 0 0 0 0 0 0) = 8 - // 16 - 2*popcount(1 1 0 1 0 0 1 1 1 1 0 0 0 0 0 0) = 2 - // 16 - 2*popcount(1 0 0 1 1 1 1 0 0 1 0 1 0 0 0 0) = 2 - // 16 - 2*popcount(0 1 0 1 0 1 0 1 1 0 1 0 1 0 1 1) = -2 - - // 16 - 2*popcount(1 0 0 0 1 1 1 0 1 1 1 1 1 1 1 1) = -8 - // 16 - 2*popcount(0 0 1 0 1 1 0 0 0 0 1 1 1 1 1 1) = -2 - // 16 - 2*popcount(0 1 1 0 0 0 0 1 1 0 1 0 1 1 1 1) = -2 - // 16 - 2*popcount(1 0 1 0 1 0 1 0 0 1 0 1 0 1 0 0) = 2 - - // 16 - 2*popcount(0 0 1 0 0 1 0 0 1 0 1 0 1 0 1 0) = 4 - // 16 - 2*popcount(1 0 0 0 0 1 1 0 0 1 1 0 1 0 1 0) = 2 - // 16 - 2*popcount(1 1 0 0 1 0 1 1 1 1 1 1 1 0 1 0) = -6 - // 16 - 2*popcount(0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 1) = 14 - VF output_vec = { - -4.0f, -2.0f, 6.0f, -14.0f, - 8.0f, 2.0f, 2.0f, -2.0f, - -8.0f, -2.0f, -2.0f, 2.0f, - 4.0f, 2.0f, -6.0f, 14.0f }; - - topology topology( - input_layout("input", input->get_layout()), - data("weights", weights), - binary_convolution("binary_conv", input_info("input"), { "weights" }, - { 1,1 }, - { 0,0 }, - { 1,1 }, - { 1,4,2,2 }, - 0, 0.0f, - data_types::f16, - padding{ { 0,0,0,0 }, 0 }) - ); - - ov::intel_gpu::ExecutionConfig config = get_test_default_config(engine); - config.set_property(ov::intel_gpu::optimize_data(true)); - - network network(engine, topology, config); - network.set_input_data("input", input); - - auto outputs = network.execute(); - ASSERT_EQ(outputs.size(), size_t(1)); - ASSERT_EQ(outputs.begin()->first, "binary_conv"); - - auto output_memory = outputs.at("binary_conv").get_memory(); - auto output_layout = output_memory->get_layout(); - cldnn::mem_lock output_ptr(output_memory, get_test_stream()); - - ASSERT_EQ(output_layout.format, format::bfyx); - ASSERT_EQ(output_layout.data_type, data_types::f16); - ASSERT_EQ(output_layout.batch(), 1); - ASSERT_EQ(output_layout.feature(), 4); - ASSERT_EQ(output_layout.spatial(1), 2); - ASSERT_EQ(output_layout.spatial(0), 2); - - for (size_t i = 0; i < output_layout.count(); i++) { - ASSERT_EQ(half_to_float(output_ptr[i]), output_vec[i]) << "index="<< i; - } -} diff --git a/src/plugins/intel_gpu/tests/unit/test_cases/quantize_gpu_test.cpp b/src/plugins/intel_gpu/tests/unit/test_cases/quantize_gpu_test.cpp index 3058d3b389e..42ad04775d3 100644 --- a/src/plugins/intel_gpu/tests/unit/test_cases/quantize_gpu_test.cpp +++ b/src/plugins/intel_gpu/tests/unit/test_cases/quantize_gpu_test.cpp @@ -167,72 +167,6 @@ TEST(quantize_gpu, quantize_levels_2_output_broadcast_inputs_1_ch8) { } } -TEST(quantize_gpu, quantize_levels_2_output_broadcast_inputs_1_ch8_binary_pack) { - auto& engine = get_test_engine(); - auto input = engine.allocate_memory({data_types::f32, format::bfyx, {1, 8, 2, 2}}); - auto input_thresh = engine.allocate_memory({ data_types::f32,format::bfyx,{ 1, 8, 1, 1 } }); - auto output_low = engine.allocate_memory({ data_types::f32,format::bfyx,{ 1, 1, 1, 1 } }); - auto output_high = engine.allocate_memory({ data_types::f32,format::bfyx,{ 1, 1, 1, 1 } }); - - set_values(input, { -1.0f, 2.0f, 3.0f, 4.0f, - 5.0f, 2.0f, 2.0f, 3.0f, - 4.0f, 6.0f, 3.0f, 3.0f, - 3.0f, 5.0f, 1.0f, 1.0f, - - 1.0f, 1.0f, 1.0f, 1.0f, - 4.0f, 6.0f, 3.0f, 3.0f, - 3.0f, 5.0f, 1.0f, 1.0f, - 1.0f, 1.0f, 1.0f, 1.0f }); - - set_values(input_thresh, { 0.0f, 1.0f, 2.0f, 3.0f, - 4.0f, 5.0f, 6.0f, 7.0f }); - set_values(output_low, { -1.0f }); - set_values(output_high, { 1.0f }); - - // 0 1 1 0 0 0 0 0 0 0 0 0 0 1 1 1 - // 1 1 1 1 0 1 0 0 0 0 1 1 0 1 1 1 - // 1 1 1 0 0 0 0 0 0 0 0 0 0 1 0 1 - // 1 1 1 0 0 0 0 0 0 0 0 0 0 1 0 1 - std::vector ref_data = { -1, 1, 1, 1, - 1, 1, 1, 1, - 1, 1, 1, 1, - -1, 1, -1, -1, - -1, -1, -1, -1, - -1, 1, -1, -1, - -1, -1, -1, -1, - -1, -1, -1, -1 }; - - topology topology; - topology.add( - input_layout("input", input->get_layout()), - data("input_low", input_thresh), - data("input_high", input_thresh), - data("output_low", output_low), - data("output_high", output_high), - quantize("quantize", input_info("input"), input_info("input_low"), input_info("input_high"), input_info("output_low"), input_info("output_high"), 2, data_types::u1), - reorder("reorder", input_info("quantize"), layout{data_types::f32, format::bfyx, tensor{1,8,2,2}}) - ); - - ExecutionConfig config = get_test_default_config(engine); - config.set_property(ov::intel_gpu::optimize_data(true)); - network network(engine, topology, config); - network.set_input_data("input", input); - auto outputs = network.execute(); - - auto output = outputs.at("reorder").get_memory(); - cldnn::mem_lock output_ptr(output, get_test_stream()); - - // Check that layout and memory contains logical size of tensor - ASSERT_EQ(output->count(), (size_t)32); - ASSERT_EQ(output->get_layout().count(), (size_t)32); - - ASSERT_EQ(output->size(), ref_data.size() * sizeof(uint32_t)); - - for (size_t i = 0; i < ref_data.size(); ++i) { - ASSERT_EQ(output_ptr[i], ref_data[i]) << " index = " << i; - } -} - TEST(quantize_gpu, quantize_levels_2_output_broadcast_inputs_2) { cldnn::engine& engine = get_test_engine(); auto input = engine.allocate_memory({data_types::f32, format::bfyx, {1, 16, 2, 2}}); diff --git a/src/plugins/intel_gpu/tests/unit/test_cases/reorder_gpu_test.cpp b/src/plugins/intel_gpu/tests/unit/test_cases/reorder_gpu_test.cpp index 3d71384fca8..6f4dcda23fe 100644 --- a/src/plugins/intel_gpu/tests/unit/test_cases/reorder_gpu_test.cpp +++ b/src/plugins/intel_gpu/tests/unit/test_cases/reorder_gpu_test.cpp @@ -2067,101 +2067,6 @@ TEST(reorder_gpu_i64, basic) ASSERT_EQ(*(a_ptr++), val); } -TEST(reorder_gpu_binary, binary_output) -{ - auto& engine = get_test_engine(); - - ov::intel_gpu::ExecutionConfig config = get_test_default_config(engine); - config.set_property(ov::intel_gpu::optimize_data(true)); - - auto input = engine.allocate_memory({ data_types::f32, format::bfyx,{ 2, 2, 2, 2 } }); - layout output_layout(data_types::u1, format::b_fs_yx_32fp, { 2, 2, 2, 2 }); - - // Data is supposed to be quantized to {0,1} values - set_values(input, { - 1.f, 0.f, 1.f, 1.f, - 0.f, 1.f, 1.f, 0.f, - - 1.f, 1.f, 0.f, 1.f, - 0.f, 0.f, 0.f, 1.f - }); - - topology topology( - input_layout("input", input->get_layout()), - reorder("reorder", input_info("input"), output_layout)); - - network network(engine, topology, get_test_default_config(engine)); - network.set_input_data("input", input); - - auto outputs = network.execute(); - ASSERT_EQ(outputs.size(), size_t(1)); - ASSERT_EQ(outputs.begin()->first, "reorder"); - - auto output = outputs.begin()->second.get_memory(); - cldnn::mem_lock output_ptr(output, get_test_stream()); - - std::vector answers = { 1, 2, 3, 1, - 1, 1, 0, 3 }; - - // Check that layout and memory contains logical size of tensor - ASSERT_EQ(output->count(), input->get_layout().count()); - ASSERT_EQ(output->get_layout().count(), input->get_layout().count()); - - // Check that memory physical size consider binary pack - ASSERT_EQ(output->size(), answers.size() * sizeof(uint32_t)); - - for (size_t i = 0; i < answers.size(); ++i) { - ASSERT_EQ(answers[i], output_ptr[i]) << "index: " << i; - } -} - -TEST(reorder_gpu_binary, binary_input) -{ - auto& engine = get_test_engine(); - - ov::intel_gpu::ExecutionConfig config = get_test_default_config(engine); - config.set_property(ov::intel_gpu::optimize_data(true)); - - auto input = engine.allocate_memory({ data_types::u1, format::b_fs_yx_32fp,{ 2, 2, 2, 2 } }); - layout output_layout(data_types::f32, format::bfyx, { 2, 2, 2, 2 }); - - // Data is supposed to be quantized to {0,1} values - std::vector answers = { - 1.f, -1.f, 1.f, 1.f, - -1.f, 1.f, 1.f, -1.f, - - 1.f, 1.f, -1.f, 1.f, - -1.f, -1.f, -1.f, 1.f - }; - - set_values(input, { 1, 2, 3, 1, - 1, 1, 0, 3 }); - - topology topology( - input_layout("input", input->get_layout()), - reorder("reorder", input_info("input"), output_layout)); - - network network(engine, topology, get_test_default_config(engine)); - network.set_input_data("input", input); - - auto outputs = network.execute(); - ASSERT_EQ(outputs.size(), size_t(1)); - ASSERT_EQ(outputs.begin()->first, "reorder"); - - auto output = outputs.begin()->second.get_memory(); - cldnn::mem_lock output_ptr(output, get_test_stream()); - - // Check that layout and memory contains logical size of tensor - ASSERT_EQ(output->count(), input->get_layout().count()); - ASSERT_EQ(output->get_layout().count(), input->get_layout().count()); - - ASSERT_EQ(output->size(), answers.size() * sizeof(float)); - - for (size_t i = 0; i < answers.size(); ++i) { - ASSERT_EQ(answers[i], output_ptr[i]) << "index: " << i; - } -} - TEST(reorder_gpu_f32, bfwzyx_bfyx_chain) { // Topology: @@ -2859,10 +2764,7 @@ public: cldnn::memory::ptr get_mem(cldnn::layout l) { auto prim = engine.allocate_memory(l); tensor s = l.get_tensor(); - if (l.data_type == data_types::u1) { - VF rnd_vec = rg.generate_random_1d(s.count() / 32, min_random, max_random); - set_values(prim, rnd_vec); - } else if (l.data_type == data_types::i8 || l.data_type == data_types::u8) { + if (l.data_type == data_types::i8 || l.data_type == data_types::u8) { VF rnd_vec = rg.generate_random_1d(s.count(), min_random, max_random); set_values(prim, rnd_vec); } else if (l.data_type == data_types::f16) { diff --git a/src/plugins/intel_gpu/tests/unit/test_utils/test_utils.h b/src/plugins/intel_gpu/tests/unit/test_utils/test_utils.h index 680a87283fc..b63fc2f5fa8 100644 --- a/src/plugins/intel_gpu/tests/unit/test_utils/test_utils.h +++ b/src/plugins/intel_gpu/tests/unit/test_utils/test_utils.h @@ -578,131 +578,6 @@ inline std::vector get_output_values_to_float(cldnn::network& net, const } double default_tolerance(cldnn::data_types dt); -// inline void print_bin_blob(cldnn::memory& mem, std::string name) -// { -// auto&& size = mem.get_layout().get_tensor(); - -// std::cerr << name; -// std::cerr << " shape: "; -// std::cerr << size.batch[0] << " "; -// std::cerr << size.feature[0] << " "; -// std::cerr << size.spatial[1] << " "; -// std::cerr << size.spatial[0] << " "; -// std::cerr << "(" << size.batch[0] * size.feature[0] * size.spatial[1] * size.spatial[0] << ")" << std::endl; - -// auto mem_ptr = mem.pointer(); - -// bool packed_ic = mem.get_layout().format == cldnn::format::b_fs_yx_32fp ? 1 : 0; -// int B = size.batch[0]; -// int C = size.feature[0]; -// int H = size.spatial[1]; -// int W = size.spatial[0]; - -// for (cldnn::tensor::value_type b = 0; b < B; ++b) -// { -// for (cldnn::tensor::value_type f = 0; f < C; ++f) -// { -// for (cldnn::tensor::value_type y = 0; y < H; ++y) -// { -// for (cldnn::tensor::value_type x = 0; x < W; ++x) -// { -// if (!packed_ic) -// { -// size_t input_it = b * C*H*W + f * W*H + y * W + x; -// size_t elem = input_it / 32; -// size_t bit = input_it % 32; -// std::cerr << ((mem_ptr[elem] & (1 << bit)) >> bit) << " "; -// } -// else -// { -// size_t input_it = b * (C / 32)*W*H + (f / 32)*W*H + y * W + x; -// size_t bit = f % 32; -// std::cerr << ((mem_ptr[input_it] & (1 << bit)) >> bit) << " "; -// } -// } -// std::cerr << std::endl; -// } -// std::cerr << std::endl; -// } -// std::cerr << "==============" << std::endl; -// } -// } - -// inline void print_bin_blob_packed(cldnn::memory& mem, std::string name) -// { -// auto&& size = mem.get_layout().get_tensor(); - -// std::cerr << name; -// std::cerr << " shape: "; -// std::cerr << size.batch[0] << " "; -// std::cerr << size.feature[0] << " "; -// std::cerr << size.spatial[1] << " "; -// std::cerr << size.spatial[0] << " "; -// std::cerr << "(" << size.batch[0] * size.feature[0] * size.spatial[1] * size.spatial[0] << ")" << std::endl; - -// auto mem_ptr = mem.pointer(); - -// int B = size.batch[0]; -// int C = size.feature[0]; -// int H = size.spatial[1]; -// int W = size.spatial[0]; - -// for (cldnn::tensor::value_type b = 0; b < B; ++b) -// { -// for (cldnn::tensor::value_type f = 0; f < div_up(C, 32); ++f) -// { -// for (cldnn::tensor::value_type y = 0; y < H; ++y) -// { -// for (cldnn::tensor::value_type x = 0; x < W; ++x) -// { -// size_t input_it = b * div_up(C, 32)*W*H + f * W*H + y * W + x; -// std::cerr << mem_ptr[input_it] << " "; -// } -// std::cerr << std::endl; -// } -// std::cerr << std::endl; -// } -// std::cerr << "==============" << std::endl; -// } -// } - -// inline void print_blob(cldnn::memory& mem, std::string name) -// { -// auto&& size = mem.get_layout().get_tensor(); - -// std::cerr << name; -// std::cerr << " shape: "; -// std::cerr << size.batch[0] << " "; -// std::cerr << size.feature[0] << " "; -// std::cerr << size.spatial[1] << " "; -// std::cerr << size.spatial[0] << " "; -// std::cerr << "(" << size.batch[0] * size.feature[0] * size.spatial[1] * size.spatial[0] << ")" << std::endl; - -// auto mem_ptr = mem.pointer(); - -// int B = size.batch[0]; -// int C = size.feature[0]; -// int H = size.spatial[1]; -// int W = size.spatial[0]; - -// for (cldnn::tensor::value_type b = 0; b < B; ++b) -// { -// for (cldnn::tensor::value_type f = 0; f < C; ++f) -// { -// for (cldnn::tensor::value_type y = 0; y < H; ++y) -// { -// for (cldnn::tensor::value_type x = 0; x < W; ++x) -// { -// size_t input_it = b * C*W*H + f * W*H + y * W + x; -// std::cerr << std::setw(4) << mem_ptr[input_it] << " "; -// } -// std::cerr << std::endl; -// } -// std::cerr << std::endl; -// } -// std::cerr << "==============" << std::endl; -// } -// } inline cldnn::network::ptr get_network(cldnn::engine& engine, cldnn::topology& topology, diff --git a/src/plugins/intel_gpu/tests/unit/transformations/convert_binary_conv_to_conv_test.cpp b/src/plugins/intel_gpu/tests/unit/transformations/convert_binary_conv_to_conv_test.cpp new file mode 100644 index 00000000000..23c30f0e95e --- /dev/null +++ b/src/plugins/intel_gpu/tests/unit/transformations/convert_binary_conv_to_conv_test.cpp @@ -0,0 +1,72 @@ +// Copyright (C) 2023 Intel Corporation +// SPDX-License-Identifier: Apache-2.0 +// + +#include + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "common_test_utils/ov_test_utils.hpp" +#include "openvino/core/coordinate_diff.hpp" +#include "openvino/core/type/element_type.hpp" + +using namespace testing; +using namespace ov::intel_gpu; + +TEST_F(TransformationTestsF, ConvertBinaryConvolutionToConvolutionTest1) { + { + auto input = std::make_shared(ov::element::f32, ov::Shape{ 1, 256, 56, 56 }); + auto in_lo = std::make_shared(ov::element::f32, ov::Shape{ 1, 256, 1, 1 }); + auto in_hi = std::make_shared(ov::element::f32, ov::Shape{ 1, 256, 1, 1 }); + auto out_lo = std::make_shared(ov::element::f32, ov::Shape{ 1, 1, 1, 1 }, std::vector{0.0f}); + auto out_hi = std::make_shared(ov::element::f32, ov::Shape{ 1, 1, 1, 1 }, std::vector{1.0f}); + auto fq = std::make_shared(input, in_lo, in_hi, out_lo, out_hi, 2); + auto weights = std::make_shared(ov::element::u1, ov::Shape{ 32, 256, 3, 3 }); + auto binary_conv = std::make_shared(fq, + weights, + ov::Strides{1, 1}, + ov::CoordinateDiff{1, 1}, + ov::CoordinateDiff{1, 1}, + ov::Strides{1, 1}, + ov::op::v1::BinaryConvolution::BinaryConvolutionMode::XNOR_POPCOUNT, + -1.0f); + + model = std::make_shared(ov::NodeVector{ binary_conv }, ov::ParameterVector{ input }); + manager.register_pass(); + } + { + auto input = std::make_shared(ov::element::f32, ov::Shape{ 1, 256, 56, 56 }); + auto in_lo = std::make_shared(ov::element::f32, ov::Shape{ 1, 256, 1, 1 }); + auto in_hi = std::make_shared(ov::element::f32, ov::Shape{ 1, 256, 1, 1 }); + auto out_lo = std::make_shared(ov::element::f32, ov::Shape{ 1, 1, 1, 1 }, std::vector{-1.0f}); + auto out_hi = std::make_shared(ov::element::f32, ov::Shape{ 1, 1, 1, 1 }, std::vector{1.0f}); + auto fq = std::make_shared(input, in_lo, in_hi, out_lo, out_hi, 2); + auto weights = std::make_shared(ov::element::f32, ov::Shape{ 32, 256, 3, 3 }); + + auto pb = std::make_shared(ov::element::i32, ov::Shape{ 4 }, std::vector{0, 0, 1, 1}); + auto pe = std::make_shared(ov::element::i32, ov::Shape{ 4 }, std::vector{0, 0, 1, 1}); + auto pv = std::make_shared(ov::element::f32, ov::Shape{ }, std::vector{1.0f}); + + auto pad = std::make_shared(fq, pb, pe, pv, ov::op::PadMode::CONSTANT); + + auto conv = std::make_shared(pad, + weights, + ov::Strides{1, 1}, + ov::CoordinateDiff{0, 0}, + ov::CoordinateDiff{0, 0}, + ov::Strides{1, 1}); + + model_ref = std::make_shared(ov::NodeVector{ conv }, ov::ParameterVector{ input }); + } +}