From 845463c72f1d397b32ca871a57808539cd357e43 Mon Sep 17 00:00:00 2001 From: unmeshna Date: Wed, 16 Sep 2026 04:59:45 -0700 Subject: [PATCH 1/2] Add HiFi-optimized Xtensa Conv kernels Optimize the Xtensa conv kernels using the HiFi NNLib: dilated/group conv, float32, and Int4 filter dispatch, plus new reference fallbacks (conv_float32_reference.cc, conv_int8_int16_float32.cc). Registers the new helper sources in xtensa.inc. Depends on the 09_03_2026 NNLib release (PR#3703) for the v2 per-channel conv APIs (e.g. xa_nn_conv2d_std_v2_per_chan_sym4sxasym8s). --- tensorflow/lite/micro/kernels/xtensa/conv.cc | 129 ++-- .../kernels/xtensa/conv_common_xtensa.cc | 16 +- .../kernels/xtensa/conv_float32_reference.cc | 86 +++ .../lite/micro/kernels/xtensa/conv_hifi.cc | 701 +++++++++++++++--- .../kernels/xtensa/conv_int16_reference.cc | 26 +- .../kernels/xtensa/conv_int8_int16_float32.cc | 145 ++++ .../lite/micro/kernels/xtensa/xtensa_conv.h | 30 +- .../lite/micro/tools/make/ext_libs/xtensa.inc | 105 ++- 8 files changed, 1023 insertions(+), 215 deletions(-) create mode 100644 tensorflow/lite/micro/kernels/xtensa/conv_float32_reference.cc create mode 100644 tensorflow/lite/micro/kernels/xtensa/conv_int8_int16_float32.cc diff --git a/tensorflow/lite/micro/kernels/xtensa/conv.cc b/tensorflow/lite/micro/kernels/xtensa/conv.cc index 31aebe86626..c04324279f6 100644 --- a/tensorflow/lite/micro/kernels/xtensa/conv.cc +++ b/tensorflow/lite/micro/kernels/xtensa/conv.cc @@ -48,85 +48,82 @@ TfLiteStatus Eval(TfLiteContext* context, TfLiteNode* node) { const TfLiteEvalTensor* filter = tflite::micro::GetEvalInput(context, node, kConvWeightsTensor); const TfLiteEvalTensor* bias = - tflite::micro::GetEvalInput(context, node, kConvBiasTensor); + (NumInputs(node) == 3) + ? tflite::micro::GetEvalInput(context, node, kConvBiasTensor) + : nullptr; switch (input->type) { case kTfLiteFloat32: { -#ifdef USE_TFLM_COMPRESSION - - MicroContext* micro_context = GetMicroContext(context); - - const CompressionTensorData* weights_comp_td = - micro_context->GetTensorCompressionData(node, kConvWeightsTensor); - const CompressionTensorData* bias_comp_td = - micro_context->GetTensorCompressionData(node, kConvBiasTensor); - -#endif // USE_TFLM_COMPRESSION - tflite::reference_ops::Conv( - ConvParamsFloat(params, op_data.reference_op_data), - tflite::micro::GetTensorShape(input), - tflite::micro::GetTensorData(input), - tflite::micro::GetTensorShape(filter), -#ifdef USE_TFLM_COMPRESSION - tflite::micro::GetTensorData( - micro_context, filter, weights_comp_td, - op_data.reference_op_data.weights_scratch_index), - tflite::micro::GetTensorShape(bias), - tflite::micro::GetOptionalTensorData( - micro_context, bias, bias_comp_td, - op_data.reference_op_data.bias_scratch_index), -#else // USE_TFLM_COMPRESSION - tflite::micro::GetTensorData(filter), - tflite::micro::GetTensorShape(bias), - tflite::micro::GetOptionalTensorData(bias), -#endif // USE_TFLM_COMPRESSION - tflite::micro::GetTensorShape(output), - tflite::micro::GetTensorData(output), - tflite::micro::GetTensorShape(nullptr), nullptr); +#if defined(INCLUDE_FLOAT_OPT) + return ConvEvalHifiFloat32(context, node, params, op_data, input, filter, + bias, output); +#else + return ConvReferenceEvalFloat32(context, node); +#endif break; } case kTfLiteInt8: { -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) - if (params.dilation_width_factor == 1 && - params.dilation_height_factor == 1) { - return ConvEvalHifiInt8(context, node, params, op_data, input, filter, - bias, output); - } else { - return ConvReferenceEvalInt8(context, node); - } + switch (filter->type) { + case kTfLiteInt4: { +#if defined(HIFI5) && defined(NNLIB_HIFI5) + return ConvEvalHifiInt4(context, node, params, op_data, input, filter, + bias, output); +#elif defined(HIFI4) + TfLiteEvalTensor filter_int8 = tflite::micro::MakeUnpackedInt4Tensor( + context, op_data.reference_op_data.filter_buffer_index, filter); + return ConvEvalHifiInt8(context, node, params, op_data, input, &filter_int8, + bias, output); +#else + return ConvReferenceEvalInt8(context, node); +#endif + break; + } + case kTfLiteInt8: { +#if defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + return ConvEvalHifiInt8(context, node, params, op_data, input, filter, + bias, output); #elif defined(VISION_P6) - // At this time the optimized implementation is failing the unit tests in - // ways that are not entirely clear why. For now, we have identified some - // of the problem cases and are manually inserting a reference fallback. - // See http://b/270720625 for more details. - if (op_data.is_per_channel_quantized || - input->dims->data[1] != input->dims->data[2]) { - return ConvReferenceEvalInt8(context, node); - } else { - return ConvEvalVision(context, node, params, op_data, input, filter, - bias, output); - } + // At this time the optimized implementation is failing the unit tests in + // ways that are not entirely clear why. For now, we have identified some + // of the problem cases and are manually inserting a reference fallback. + // See http://b/270720625 for more details. + if (op_data.is_per_channel_quantized || + input->dims->data[1] != input->dims->data[2]) { + return ConvReferenceEvalInt8(context, node); + } else { + return ConvEvalVision(context, node, params, op_data, input, filter, + bias, output); + } #else - return ConvReferenceEvalInt8(context, node); + return ConvReferenceEvalInt8(context, node); #endif + break; + } + default: + MicroPrintf("Type %s (%d) not supported.", TfLiteTypeGetName(filter->type), + filter->type); + return kTfLiteError; + } + break; } case kTfLiteInt16: { -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) - // Note that int32 bias is not widely supported and might be risky (e.g. - // http://b/262003750). As such, while we have a fallback to the reference - // implementation, production use-cases should only have int64 bias. - const bool requires_int32_accum = - (bias != nullptr && bias->type == kTfLiteInt32) || - (bias == nullptr && params.quantized_bias_type == kTfLiteInt32); - if (requires_int32_accum) { +#if defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + if (bias == nullptr || bias->type == kTfLiteInt64) { + return ConvEvalHifiInt16(context, node, params, op_data, input, filter, bias, + output); + } + else if (bias->type == kTfLiteInt32) { +#else // defined(HIFI4) || defined(HIFI5) + if (bias == nullptr || bias->type == kTfLiteInt64 || bias->type == kTfLiteInt32) { +#endif // defined(HIFI4) || defined(HIFI5) return ConvReferenceEvalInt16(context, node); - } else { - return ConvEvalHifiInt16(context, node, params, op_data, input, filter, - bias, output); } -#else - return ConvReferenceEvalInt16(context, node); -#endif + else { + MicroPrintf("Bias type %s (%d) not supported.", + TfLiteTypeGetName(bias->type), bias->type); + return kTfLiteError; + } + break; } default: MicroPrintf("Type %s (%d) not supported.", TfLiteTypeGetName(input->type), diff --git a/tensorflow/lite/micro/kernels/xtensa/conv_common_xtensa.cc b/tensorflow/lite/micro/kernels/xtensa/conv_common_xtensa.cc index 3063e77744d..9f853eca920 100644 --- a/tensorflow/lite/micro/kernels/xtensa/conv_common_xtensa.cc +++ b/tensorflow/lite/micro/kernels/xtensa/conv_common_xtensa.cc @@ -42,10 +42,18 @@ void* ConvInitXtensa(TfLiteContext* context, const char* buffer, TfLiteStatus ConvPrepareXtensa(TfLiteContext* context, TfLiteNode* node) { TF_LITE_ENSURE_OK(context, ConvPrepare(context, node)); -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) - TF_LITE_ENSURE_OK(context, ConvPrepareHifi(context, node)); -#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) - +#if defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) +#if defined(HIFI5) && defined(NNLIB_HIFI5) + const TfLiteEvalTensor* filter = + tflite::micro::GetEvalInput(context, node, kConvWeightsTensor); + const TfLiteEvalTensor* input = + tflite::micro::GetEvalInput(context, node, kConvInputTensor); + if(input->type == kTfLiteInt8 && filter->type == kTfLiteInt4) + TF_LITE_ENSURE_OK(context, ConvPrepareHifiInt4(context, node)); + else +#endif // defined(HIFI5) && defined(NNLIB_HIFI5) + TF_LITE_ENSURE_OK(context, ConvPrepareHifi(context, node)); +#endif #if defined(VISION_P6) TF_LITE_ENSURE_OK(context, ConvPrepareVision(context, node)); #endif // defined(VISION_P6) diff --git a/tensorflow/lite/micro/kernels/xtensa/conv_float32_reference.cc b/tensorflow/lite/micro/kernels/xtensa/conv_float32_reference.cc new file mode 100644 index 00000000000..118d3cb0720 --- /dev/null +++ b/tensorflow/lite/micro/kernels/xtensa/conv_float32_reference.cc @@ -0,0 +1,86 @@ +/* Copyright 2023 The TensorFlow Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#include "tensorflow/lite/c/builtin_op_data.h" +#include "tensorflow/lite/c/common.h" +#include "tensorflow/lite/kernels/internal/common.h" +#include "tensorflow/lite/kernels/internal/quantization_util.h" +#include "tensorflow/lite/kernels/internal/reference/conv.h" +#include "tensorflow/lite/kernels/internal/reference/integer_ops/conv.h" +#include "tensorflow/lite/kernels/internal/tensor_ctypes.h" +#include "tensorflow/lite/kernels/kernel_util.h" +#include "tensorflow/lite/kernels/padding.h" +#include "tensorflow/lite/micro/kernels/conv.h" +#include "tensorflow/lite/micro/kernels/kernel_util.h" +#include "tensorflow/lite/micro/micro_log.h" +#include "tensorflow/lite/micro/kernels/xtensa/xtensa_conv.h" + +namespace tflite { + +TfLiteStatus ConvReferenceEvalFloat32(TfLiteContext* context, TfLiteNode* node) { + TFLITE_DCHECK(node->user_data != nullptr); + TFLITE_DCHECK(node->builtin_data != nullptr); + + TfLiteEvalTensor* output = + tflite::micro::GetEvalOutput(context, node, kConvOutputTensor); + const TfLiteEvalTensor* input = + tflite::micro::GetEvalInput(context, node, kConvInputTensor); + const TfLiteEvalTensor* filter = + tflite::micro::GetEvalInput(context, node, kConvWeightsTensor); + const TfLiteEvalTensor* bias = + (NumInputs(node) == 3) + ? tflite::micro::GetEvalInput(context, node, kConvBiasTensor) + : nullptr; + + const auto& params = + *(reinterpret_cast(node->builtin_data)); + const auto& op_data = *(reinterpret_cast(node->user_data)); + +#ifdef USE_TFLM_COMPRESSION + + MicroContext* micro_context = GetMicroContext(context); + + const CompressionTensorData* weights_comp_td = + micro_context->GetTensorCompressionData(node, kConvWeightsTensor); + const CompressionTensorData* bias_comp_td = + micro_context->GetTensorCompressionData(node, kConvBiasTensor); + +#endif // USE_TFLM_COMPRESSION + tflite::reference_ops::Conv( + ConvParamsFloat(params, op_data.reference_op_data), + tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(filter), +#ifdef USE_TFLM_COMPRESSION + tflite::micro::GetTensorData( + micro_context, filter, weights_comp_td, + op_data.reference_op_data.weights_scratch_index), + tflite::micro::GetTensorShape(bias), + tflite::micro::GetOptionalTensorData( + micro_context, bias, bias_comp_td, + op_data.reference_op_data.bias_scratch_index), +#else // USE_TFLM_COMPRESSION + tflite::micro::GetTensorData(filter), + tflite::micro::GetTensorShape(bias), + tflite::micro::GetOptionalTensorData(bias), +#endif // USE_TFLM_COMPRESSION + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output), + tflite::micro::GetTensorShape(nullptr), nullptr); + + return kTfLiteOk; +} + +} // namespace tflite diff --git a/tensorflow/lite/micro/kernels/xtensa/conv_hifi.cc b/tensorflow/lite/micro/kernels/xtensa/conv_hifi.cc index 8d3e5ab22a4..b939a2d60e7 100644 --- a/tensorflow/lite/micro/kernels/xtensa/conv_hifi.cc +++ b/tensorflow/lite/micro/kernels/xtensa/conv_hifi.cc @@ -13,7 +13,7 @@ See the License for the specific language governing permissions and limitations under the License. ==============================================================================*/ -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) #include @@ -22,6 +22,8 @@ limitations under the License. #include "tensorflow/lite/kernels/internal/common.h" #include "tensorflow/lite/kernels/internal/portable_tensor_utils.h" #include "tensorflow/lite/kernels/internal/tensor_ctypes.h" +#include "tensorflow/lite/kernels/internal/reference/conv.h" +#include "tensorflow/lite/kernels/internal/reference/integer_ops/conv.h" #include "tensorflow/lite/kernels/kernel_util.h" #include "tensorflow/lite/micro/kernels/conv.h" #include "tensorflow/lite/micro/kernels/kernel_util.h" @@ -50,23 +52,12 @@ TfLiteStatus ConvPrepareHifi(TfLiteContext* context, TfLiteNode* node) { const RuntimeShape& filter_shape = GetTensorShape(filter); const RuntimeShape& output_shape = GetTensorShape(output); - // Check if the Xtensa optimized code can be used - // HIFI4 and HIFI5 do not allow bias data pointer to be nullptr - /* TODO(b/277112516): Dilation is currently not supported on HiFi 4 NN Library - */ - bool inputs_and_bias_ok = bias != nullptr; -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) - inputs_and_bias_ok = - inputs_and_bias_ok && + bool inputs_and_bias_ok = (input->type == kTfLiteInt8 || - (input->type == kTfLiteInt16 && bias->type == kTfLiteInt64)); -#else - inputs_and_bias_ok = inputs_and_bias_ok && (input->type == kTfLiteInt8); -#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) - if (!(inputs_and_bias_ok && params->dilation_width_factor == 1 && - params->dilation_height_factor == 1 && - input_shape.Dims(1) >= filter_shape.Dims(1) && - input_shape.Dims(2) >= filter_shape.Dims(2))) { + (input->type == kTfLiteInt16 && (!bias || bias->type == kTfLiteInt64)) || + input->type == kTfLiteFloat32); + + if (inputs_and_bias_ok == 0) { micro_context->DeallocateTempTfLiteTensor(input); micro_context->DeallocateTempTfLiteTensor(filter); micro_context->DeallocateTempTfLiteTensor(output); @@ -78,7 +69,7 @@ TfLiteStatus ConvPrepareHifi(TfLiteContext* context, TfLiteNode* node) { const int input_height = input_shape.Dims(1); const int input_width = input_shape.Dims(2); - const int input_depth = MatchingDim(input_shape, 3, filter_shape, 3); + const int input_depth = input_shape.Dims(3); const int filter_height = filter_shape.Dims(1); const int filter_width = filter_shape.Dims(2); const int filter_depth = filter_shape.Dims(3); @@ -91,25 +82,69 @@ TfLiteStatus ConvPrepareHifi(TfLiteContext* context, TfLiteNode* node) { const int pad_width = data->reference_op_data.padding.width; int required_scratch = 0; - // TODO(b/277112516): Dilation is currently not supported on HiFi 4 NN Library - if ((params->dilation_width_factor == 1) && + // Dilation is currently not supported for kTfLiteInt16 datatype. + if( ((params->dilation_height_factor > 1) || (params->dilation_width_factor > 1)) + && ((input->type == kTfLiteInt8) || (input->type == kTfLiteInt16)) && (input_depth == filter_depth)) { + // For HiFi5, with nnlib-hifi5 versions 1.7.0 onwards and for HiFi4 with nnlib-hifi4 versions 2.5.0 onwards, + // we use the below dilated_conv2d_std getsize() API. For the earlier versions, "output_channels" argument is not needed. +#if defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + if (input->type == kTfLiteInt8) { + required_scratch = xa_nn_dilated_conv2d_std_getsize( + input_height, input_depth, filter_height, filter_width, stride_height, + pad_height, output_height, output_channels, PREC_ASYM8S, params->dilation_height_factor); + } + else if (input->type == kTfLiteInt16) { + required_scratch = xa_nn_dilated_conv2d_std_getsize( + input_height, input_depth, filter_height, filter_width, stride_height, + pad_height, output_height, output_channels, PREC_SYM16S, params->dilation_height_factor); + } +#endif // defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) +#ifndef HIFI_IQ // Scratchpad may not be required in some cases on HiFi-iQ. + TF_LITE_ENSURE(context, required_scratch > 0); +#endif + } + else if ((params->dilation_width_factor == 1) && (params->dilation_height_factor == 1)) { if (input->type == kTfLiteInt8) { - required_scratch = xa_nn_conv2d_std_getsize( - input_height, input_width, input_depth, filter_height, filter_width, - filter_depth, stride_height, pad_height, stride_width, pad_width, - output_height, output_width, output_channels, PREC_ASYM8S, PREC_SYM8S, - 1, 1, 0); + if(input_depth == filter_depth){ + required_scratch = xa_nn_conv2d_std_getsize( + input_height, input_width, input_depth, filter_height, filter_width, filter_depth, stride_height, + pad_height, stride_width, pad_width, output_height, output_width, output_channels, PREC_ASYM8S, PREC_SYM8S, params->dilation_height_factor, params->dilation_width_factor, 0/*Out data format*/); + } + else{ + required_scratch = xa_nn_conv2d_getsize( + input_height, input_width, input_depth, filter_height, filter_width, filter_depth, params->dilation_height_factor, params->dilation_width_factor, stride_height, + pad_height, stride_width, pad_width, output_height, output_width, output_channels, PREC_ASYM8S, PREC_SYM8S, 0/*Out data format*/); + } +#ifndef HIFI_IQ // Scratchpad may not be required in some cases on HiFi-iQ. TF_LITE_ENSURE(context, required_scratch > 0); +#endif } if (input->type == kTfLiteInt16) { - required_scratch = xa_nn_conv2d_std_getsize( - input_height, input_width, input_depth, filter_height, filter_width, - filter_depth, stride_height, pad_height, stride_width, pad_width, - output_height, output_width, output_channels, PREC_SYM16S, PREC_SYM8S, - 1, 1, 0); + if(input_depth == filter_depth){ + required_scratch = xa_nn_conv2d_std_getsize( + input_height, input_width, input_depth, filter_height, filter_width, filter_depth, stride_height, + pad_height, stride_width, pad_width, output_height, output_width, output_channels, PREC_SYM16S, PREC_SYM8S, params->dilation_height_factor, params->dilation_width_factor, 0/*Out data format*/); + } + else{ + required_scratch = xa_nn_conv2d_getsize( + input_height, input_width, input_depth, filter_height, filter_width, filter_depth, params->dilation_height_factor, params->dilation_width_factor, stride_height, + pad_height, stride_width, pad_width, output_height, output_width, output_channels, PREC_SYM16S, PREC_SYM8S, 0/*Out data format*/); + } +#ifndef HIFI_IQ // Scratchpad may not be required in some cases on HiFi-iQ. TF_LITE_ENSURE(context, required_scratch > 0); +#endif } +#if defined(INCLUDE_FLOAT_OPT) + if ((input->type == kTfLiteFloat32) && (input_depth == filter_depth)) { + required_scratch = xa_nn_conv2d_std_getsize( + input_height, input_width, input_depth, filter_height, filter_width, filter_depth, stride_height, + pad_height, stride_width, pad_width, output_height, output_width, output_channels, PREC_F32, PREC_F32, params->dilation_height_factor, params->dilation_width_factor, 0/*Out data format*/); +#ifndef HIFI_IQ // Scratchpad may not be required in some cases on HiFi-iQ. + TF_LITE_ENSURE(context, required_scratch > 0); +#endif + } +#endif } TF_LITE_ENSURE_OK( context, context->RequestScratchBufferInArena( @@ -124,7 +159,69 @@ TfLiteStatus ConvPrepareHifi(TfLiteContext* context, TfLiteNode* node) { return kTfLiteOk; } -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if (defined(HIFI5) && defined(NNLIB_HIFI5)) || defined(HIFI_IQ) +TfLiteStatus ConvPrepareHifiInt4(TfLiteContext* context, TfLiteNode* node) { + XtensaConvOpData* data = static_cast(node->user_data); + const auto params = static_cast(node->builtin_data); + + MicroContext* micro_context = GetMicroContext(context); + + // Calculate scratch memory requirements and request scratch buffer + TfLiteTensor* output = + micro_context->AllocateTempOutputTensor(node, kConvOutputTensor); + TF_LITE_ENSURE(context, output != nullptr); + TfLiteTensor* input = + micro_context->AllocateTempInputTensor(node, kConvInputTensor); + TF_LITE_ENSURE(context, input != nullptr); + TfLiteTensor* filter = + micro_context->AllocateTempInputTensor(node, kConvWeightsTensor); + TF_LITE_ENSURE(context, filter != nullptr); + + const RuntimeShape& input_shape = GetTensorShape(input); + const RuntimeShape& filter_shape = GetTensorShape(filter); + const RuntimeShape& output_shape = GetTensorShape(output); + const int input_height = input_shape.Dims(1); + const int input_width = input_shape.Dims(2); + const int input_depth = input_shape.Dims(3); + const int filter_height = filter_shape.Dims(1); + const int filter_width = filter_shape.Dims(2); + const int filter_depth = filter_shape.Dims(3); + const int output_height = output_shape.Dims(1); + const int output_width = output_shape.Dims(2); + const int output_channels = output_shape.Dims(3); + const int stride_height = params->stride_height; + const int stride_width = params->stride_width; + const int pad_height = data->reference_op_data.padding.height; + const int pad_width = data->reference_op_data.padding.width; + + int required_scratch = 0; + + if ((params->dilation_width_factor == 1) && + (params->dilation_height_factor == 1) && + (filter_depth == input_depth)) { + required_scratch = xa_nn_conv2d_std_getsize( + input_height, input_width, input_depth, filter_height, filter_width, filter_depth, stride_height, + pad_height, stride_width, pad_width, output_height, output_width, output_channels, PREC_ASYM8S, PREC_SYM4S, params->dilation_height_factor, params->dilation_width_factor, 0/*Out data format*/); + TF_LITE_ENSURE(context, required_scratch > 0); + } + else + { + required_scratch = + RuntimeShape(filter->dims->size, + reinterpret_cast(filter->dims->data)) + .FlatSize(); + } + TF_LITE_ENSURE_OK( + context, context->RequestScratchBufferInArena( + context, required_scratch, &data->scratch_tensor_index)); + + micro_context->DeallocateTempTfLiteTensor(input); + micro_context->DeallocateTempTfLiteTensor(filter); + micro_context->DeallocateTempTfLiteTensor(output); + return kTfLiteOk; +} +#endif + TfLiteStatus ConvEvalHifiInt16(TfLiteContext* context, TfLiteNode* node, const TfLiteConvParams& params, const XtensaConvOpData& data, @@ -145,12 +242,13 @@ TfLiteStatus ConvEvalHifiInt16(TfLiteContext* context, TfLiteNode* node, const RuntimeShape& output_shape = tflite::micro::GetTensorShape(output); const int batches = MatchingDim(input_shape, 0, output_shape, 0); - const int input_depth = MatchingDim(input_shape, 3, filter_shape, 3); const int output_depth = MatchingDim(filter_shape, 0, output_shape, 3); const int input_height = input_shape.Dims(1); const int input_width = input_shape.Dims(2); + const int input_depth = input_shape.Dims(3); const int filter_height = filter_shape.Dims(1); const int filter_width = filter_shape.Dims(2); + const int filter_depth = filter_shape.Dims(3); const int output_height = output_shape.Dims(1); const int output_width = output_shape.Dims(2); @@ -175,71 +273,121 @@ TfLiteStatus ConvEvalHifiInt16(TfLiteContext* context, TfLiteNode* node, data.reference_op_data.bias_scratch_index); #else // USE_TFLM_COMPRESSION const int8_t* filter_data = tflite::micro::GetTensorData(filter); - const int64_t* bias_data = - tflite::micro::GetOptionalTensorData(bias); + const int64_t* bias_data = tflite::micro::GetOptionalTensorData(bias); #endif // USE_TFLM_COMPRESSION int16_t* output_data = tflite::micro::GetTensorData(output); int output_data_format = 0; int out_length = output_height * output_width * output_depth; - if (filter_height == 1 && filter_width == 1) { + if(params.dilation_height_factor == 1 && params.dilation_width_factor == 1) { + if (filter_height == 1 && filter_width == 1) { + for (int batch = 0; batch < batches; ++batch) { + int16_t* p_out_temp; + p_out_temp = &output_data[batch * out_length]; + + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_pointwise_v2_per_chan_sym8sxsym16s( + p_out_temp, const_cast(filter_data), + const_cast(&input_data[batch * input_height * + input_width * input_depth]), + const_cast(bias_data), input_height, input_width, + input_depth, output_depth, 0, + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, 0, + output_data_format, output_activation_min, output_activation_max, NULL), + 0); + } + } else { + void* p_scratch = static_cast( + context->GetScratchBuffer(context, data.scratch_tensor_index)); + + for (int batch = 0; batch < batches; ++batch) { + int16_t* p_out_temp; + p_out_temp = &output_data[batch * out_length]; + + { + if(filter_depth == input_depth){ + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_std_v2_per_chan_sym8sxsym16s( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + const_cast(filter_data), // filter_data, + bias_data, input_height, input_width, input_depth, + filter_height, filter_width, output_depth, stride_width, + stride_height, pad_width, pad_height, output_height, + output_width, 0, + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, 0, + output_data_format, static_cast(p_scratch), + output_activation_min, output_activation_max, NULL), + 0); + } + else{ + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_v2_per_chan_sym8sxsym16s( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + const_cast(filter_data), // filter_data, + bias_data, input_height, input_width, input_depth, + filter_height, filter_width, filter_depth, params.dilation_height_factor, params.dilation_width_factor, output_depth, stride_width, + stride_height, pad_width, pad_height, output_height, + output_width, 0, + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + 0, output_data_format, + static_cast(p_scratch), + output_activation_min, output_activation_max, NULL), + 0); + } + } + } + } + } else if (filter_depth == input_depth) { + /* dilated convolution available only for filter_depth = input_depth */ + void* p_scratch = static_cast( + context->GetScratchBuffer(context, data.scratch_tensor_index)); + for (int batch = 0; batch < batches; ++batch) { int16_t* p_out_temp; p_out_temp = &output_data[batch * out_length]; TF_LITE_ENSURE_EQ( context, - xa_nn_conv2d_pointwise_per_chan_sym8sxsym16s( - p_out_temp, const_cast(filter_data), - const_cast(&input_data[batch * input_height * - input_width * input_depth]), - const_cast(bias_data), input_height, input_width, - input_depth, output_depth, 0, + xa_nn_dilated_conv2d_std_v2_per_chan_sym8sxsym16s( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + const_cast(filter_data), // filter_data, + bias_data, input_height, input_width, input_depth, + filter_height, filter_width, output_depth, stride_width, + stride_height, pad_width, pad_height, output_height, + output_width, 0, data.reference_op_data.per_channel_output_multiplier, data.reference_op_data.per_channel_output_shift, 0, - output_data_format), + output_data_format, static_cast(p_scratch), + params.dilation_height_factor, params.dilation_width_factor, + output_activation_min, output_activation_max, NULL), 0); - - TF_LITE_ENSURE_EQ(context, - xa_nn_vec_activation_min_max_16_16( - p_out_temp, p_out_temp, output_activation_min, - output_activation_max, out_length), - 0); } } else { - void* p_scratch = static_cast( - context->GetScratchBuffer(context, data.scratch_tensor_index)); - for (int batch = 0; batch < batches; ++batch) { - int16_t* p_out_temp; - p_out_temp = &output_data[batch * out_length]; - - { - TF_LITE_ENSURE_EQ( - context, - xa_nn_conv2d_std_per_chan_sym8sxsym16s( - p_out_temp, - &input_data[batch * input_height * input_width * input_depth], - const_cast(filter_data), // filter_data, - bias_data, input_height, input_width, input_depth, - filter_height, filter_width, output_depth, stride_width, - stride_height, pad_width, pad_height, output_height, - output_width, 0, - data.reference_op_data.per_channel_output_multiplier, - data.reference_op_data.per_channel_output_shift, 0, - output_data_format, static_cast(p_scratch)), - 0); - } - TF_LITE_ENSURE_EQ(context, - xa_nn_vec_activation_min_max_16_16( - p_out_temp, p_out_temp, output_activation_min, - output_activation_max, out_length), - 0); - } + reference_integer_ops::ConvPerChannel( + ConvParamsQuantized(params, data.reference_op_data), + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(filter), + tflite::micro::GetTensorData(filter), + tflite::micro::GetTensorShape(bias), + tflite::micro::GetOptionalTensorData(bias), + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output)); } return kTfLiteOk; } -#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) TfLiteStatus ConvEvalHifiInt8(TfLiteContext* context, TfLiteNode* node, const TfLiteConvParams& params, @@ -263,12 +411,13 @@ TfLiteStatus ConvEvalHifiInt8(TfLiteContext* context, TfLiteNode* node, const RuntimeShape& output_shape = tflite::micro::GetTensorShape(output); const int batches = MatchingDim(input_shape, 0, output_shape, 0); - const int input_depth = MatchingDim(input_shape, 3, filter_shape, 3); + const int input_depth = input_shape.Dims(3); const int output_depth = MatchingDim(filter_shape, 0, output_shape, 3); const int input_height = input_shape.Dims(1); const int input_width = input_shape.Dims(2); const int filter_height = filter_shape.Dims(1); const int filter_width = filter_shape.Dims(2); + const int filter_depth = filter_shape.Dims(3); const int output_height = output_shape.Dims(1); const int output_width = output_shape.Dims(2); @@ -285,46 +434,30 @@ TfLiteStatus ConvEvalHifiInt8(TfLiteContext* context, TfLiteNode* node, const int8_t* input_data = tflite::micro::GetTensorData(input); #ifdef USE_TFLM_COMPRESSION + const int8_t* filter_data = tflite::micro::GetTensorData( + micro_context, filter, weights_comp_td, + data.reference_op_data.weights_scratch_index); const int32_t* bias_data = tflite::micro::GetOptionalTensorData( micro_context, bias, bias_comp_td, data.reference_op_data.bias_scratch_index); #else // USE_TFLM_COMPRESSION - const int32_t* bias_data = - tflite::micro::GetOptionalTensorData(bias); + const int8_t* filter_data = tflite::micro::GetTensorData(filter); + const int32_t* bias_data = tflite::micro::GetOptionalTensorData(bias); #endif // USE_TFLM_COMPRESSION int8_t* output_data = tflite::micro::GetTensorData(output); - const int8_t* filter_data; - if (filter->type == kTfLiteInt4) { - int8_t* unpacked_filter_data = - static_cast(context->GetScratchBuffer( - context, data.reference_op_data.filter_buffer_index)); - tflite::tensor_utils::UnpackDenseInt4IntoInt8( - tflite::micro::GetTensorData(filter), - tflite::micro::GetTensorShape(filter).FlatSize(), unpacked_filter_data); - filter_data = unpacked_filter_data; - } else { -#ifdef USE_TFLM_COMPRESSION - filter_data = tflite::micro::GetTensorData( - micro_context, filter, weights_comp_td, - data.reference_op_data.weights_scratch_index); -#else // USE_TFLM_COMPRESSION - filter_data = tflite::micro::GetTensorData(filter); -#endif // USE_TFLM_COMPRESSION - } - int output_data_format = 0; int out_length = output_height * output_width * output_depth; - if (filter_height == 1 && filter_width == 1) { + if (filter_height == 1 && filter_width == 1 && stride_width == 1 && stride_height == 1 && + pad_width == 0 && pad_height == 0 && (input_height == output_height) && (input_width == output_width) && (filter_depth == input_depth)) { for (int batch = 0; batch < batches; ++batch) { int8_t* p_out_temp; p_out_temp = &output_data[batch * out_length]; TF_LITE_ENSURE_EQ( context, - - xa_nn_conv2d_pointwise_per_chan_sym8sxasym8s( + xa_nn_conv2d_pointwise_v2_per_chan_sym8sxasym8s( p_out_temp, const_cast(filter_data), const_cast(&input_data[batch * input_height * input_width * input_depth]), @@ -332,50 +465,374 @@ TfLiteStatus ConvEvalHifiInt8(TfLiteContext* context, TfLiteNode* node, input_depth, output_depth, input_offset, data.reference_op_data.per_channel_output_multiplier, data.reference_op_data.per_channel_output_shift, output_offset, - output_data_format), + output_data_format, + output_activation_min, output_activation_max, NULL), 0); - - TF_LITE_ENSURE_EQ(context, - xa_nn_vec_activation_min_max_8_8( - p_out_temp, p_out_temp, output_activation_min, - output_activation_max, out_length), - 0); } } else { void* p_scratch = static_cast( context->GetScratchBuffer(context, data.scratch_tensor_index)); + if(((params.dilation_width_factor > 1) || (params.dilation_height_factor > 1)) && (filter_depth != input_depth) ) + { + /*Dilated Group-conv not supported*/ + reference_integer_ops::ConvPerChannel( + ConvParamsQuantized(params, data.reference_op_data), + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(filter), + tflite::micro::GetTensorData(filter), + tflite::micro::GetTensorShape(bias), + tflite::micro::GetOptionalTensorData(bias), + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output)); + return kTfLiteOk; + } + for (int batch = 0; batch < batches; ++batch) { int8_t* p_out_temp; p_out_temp = &output_data[batch * out_length]; + if (((params.dilation_width_factor > 1) || + (params.dilation_height_factor > 1)) && + (filter_depth == input_depth)) { TF_LITE_ENSURE_EQ( context, - xa_nn_conv2d_std_per_chan_sym8sxasym8s( - p_out_temp, - &input_data[batch * input_height * input_width * input_depth], - const_cast(filter_data), // filter_data, - bias_data, input_height, input_width, input_depth, - filter_height, filter_width, output_depth, stride_width, - stride_height, pad_width, pad_height, output_height, - output_width, input_offset, - data.reference_op_data.per_channel_output_multiplier, - data.reference_op_data.per_channel_output_shift, output_offset, - output_data_format, static_cast(p_scratch)), + xa_nn_dilated_conv2d_std_v2_per_chan_sym8sxasym8s(p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + const_cast(filter_data), // filter_data, + bias_data, input_height, input_width, input_depth, filter_height, + filter_width, output_depth, stride_width, stride_height, pad_width, + pad_height, output_height, output_width, input_offset, + data.reference_op_data.per_channel_output_multiplier, data.reference_op_data.per_channel_output_shift, + output_offset, output_data_format, + static_cast(p_scratch), params.dilation_height_factor, params.dilation_width_factor, + output_activation_min, output_activation_max, NULL), 0); } + else + { + if(filter_depth == input_depth){ + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_std_v2_per_chan_sym8sxasym8s( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + const_cast(filter_data), + bias_data, input_height, input_width, input_depth, + filter_height, filter_width, output_depth, stride_width, + stride_height, pad_width, pad_height, output_height, + output_width, input_offset, + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + output_offset, output_data_format, + static_cast(p_scratch), + output_activation_min, output_activation_max, NULL), + 0); + } + else{ + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_v2_per_chan_sym8sxasym8s( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + const_cast(filter_data), // filter_data, + bias_data, input_height, input_width, input_depth, + filter_height, filter_width, filter_depth, params.dilation_height_factor, params.dilation_width_factor, output_depth, stride_width, + stride_height, pad_width, pad_height, output_height, + output_width, input_offset, + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + output_offset, output_data_format, + static_cast(p_scratch), + output_activation_min, output_activation_max, NULL), + 0); + } + } + } + } + + return kTfLiteOk; +} + +#if (defined(HIFI5) && defined(NNLIB_HIFI5)) || defined(HIFI_IQ) +TfLiteStatus ConvEvalHifiInt4(TfLiteContext* context, TfLiteNode* node, + const TfLiteConvParams& params, + const XtensaConvOpData& data, + const TfLiteEvalTensor* input, + const TfLiteEvalTensor* filter, + const TfLiteEvalTensor* bias, + TfLiteEvalTensor* output) { + const RuntimeShape& input_shape = tflite::micro::GetTensorShape(input); + const RuntimeShape& filter_shape = tflite::micro::GetTensorShape(filter); + + const int32_t input_offset = -data.reference_op_data.input_zero_point; + const int32_t output_offset = data.reference_op_data.output_zero_point; + const int stride_width = params.stride_width; + const int stride_height = params.stride_height; + const int pad_width = data.reference_op_data.padding.width; + const int pad_height = data.reference_op_data.padding.height; + const int32_t output_activation_min = + data.reference_op_data.output_activation_min; + const int32_t output_activation_max = + data.reference_op_data.output_activation_max; + + const RuntimeShape& output_shape = tflite::micro::GetTensorShape(output); + const int batches = MatchingDim(input_shape, 0, output_shape, 0); + const int input_depth = input_shape.Dims(3); + const int output_depth = MatchingDim(filter_shape, 0, output_shape, 3); + const int input_height = input_shape.Dims(1); + const int input_width = input_shape.Dims(2); + const int filter_height = filter_shape.Dims(1); + const int filter_width = filter_shape.Dims(2); + const int filter_depth = filter_shape.Dims(3); + const int output_height = output_shape.Dims(1); + const int output_width = output_shape.Dims(2); + +#ifdef USE_TFLM_COMPRESSION + + MicroContext* micro_context = GetMicroContext(context); + + const CompressionTensorData* weights_comp_td = + micro_context->GetTensorCompressionData(node, kConvWeightsTensor); + const CompressionTensorData* bias_comp_td = + micro_context->GetTensorCompressionData(node, kConvBiasTensor); + +#endif // USE_TFLM_COMPRESSION + + const int8_t* input_data = tflite::micro::GetTensorData(input); +#ifdef USE_TFLM_COMPRESSION + const int8_t* filter_data = tflite::micro::GetTensorData( + micro_context, filter, weights_comp_td, + data.reference_op_data.weights_scratch_index); + const int32_t* bias_data = tflite::micro::GetOptionalTensorData( + micro_context, bias, bias_comp_td, + data.reference_op_data.bias_scratch_index); +#else // USE_TFLM_COMPRESSION + const int8_t* filter_data = tflite::micro::GetTensorData(filter); + const int32_t* bias_data = tflite::micro::GetOptionalTensorData(bias); +#endif // USE_TFLM_COMPRESSION + int8_t* output_data = tflite::micro::GetTensorData(output); + + int output_data_format = 0; + int out_length = output_height * output_width * output_depth; + + if ((params.dilation_width_factor == 1) && + (params.dilation_height_factor == 1) && + (filter_depth == input_depth)) + { + void* p_scratch = static_cast( + context->GetScratchBuffer(context, data.scratch_tensor_index)); + + for (int batch = 0; batch < batches; ++batch) { + int8_t* p_out_temp; + p_out_temp = &output_data[batch * out_length]; + +#ifndef HIFI_IQ + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_std_per_chan_sym4sxasym8s( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + filter_data, bias_data, + input_height, input_width, input_depth, + filter_height, filter_width, output_depth, + stride_width, stride_height, pad_width, pad_height, + output_height, output_width, input_offset, + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + output_offset, output_data_format, + static_cast(p_scratch)), + 0); TF_LITE_ENSURE_EQ(context, xa_nn_vec_activation_min_max_8_8( p_out_temp, p_out_temp, output_activation_min, output_activation_max, out_length), - 0); + 0); +#else + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_std_v2_per_chan_sym4sxasym8s( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + filter_data,bias_data, + input_height, input_width, input_depth, + filter_height, filter_width, output_depth, + stride_width, stride_height, pad_width, pad_height, + output_height, output_width, input_offset, + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + output_offset, output_data_format, + static_cast(p_scratch), + output_activation_min, output_activation_max, NULL), + 0); +#endif + } + } + else + { + int8_t* unpacked_filter_data = static_cast( + context->GetScratchBuffer(context, data.scratch_tensor_index)); + + tflite::tensor_utils::UnpackDenseInt4IntoInt8( + tflite::micro::GetTensorData(filter), + tflite::micro::GetTensorShape(filter).FlatSize(), unpacked_filter_data); + filter_data = unpacked_filter_data; + + reference_integer_ops::ConvPerChannel( + ConvParamsQuantized(params, data.reference_op_data), + data.reference_op_data.per_channel_output_multiplier, + data.reference_op_data.per_channel_output_shift, + tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(filter), + filter_data, + tflite::micro::GetTensorShape(bias), + tflite::micro::GetOptionalTensorData(bias), + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output)); } + return kTfLiteOk; +} +#endif // #if (defined(HIFI5) && defined(NNLIB_HIFI5)) || defined(HIFI_IQ) + +#if defined(INCLUDE_FLOAT_OPT) + +TfLiteStatus ConvEvalHifiFloat32(TfLiteContext* context, TfLiteNode* node, + const TfLiteConvParams& params, + const XtensaConvOpData& data, + const TfLiteEvalTensor* input, + const TfLiteEvalTensor* filter, + const TfLiteEvalTensor* bias, + TfLiteEvalTensor* output) { + const RuntimeShape& input_shape = tflite::micro::GetTensorShape(input); + const RuntimeShape& filter_shape = tflite::micro::GetTensorShape(filter); + const int stride_width = params.stride_width; + const int stride_height = params.stride_height; + const int pad_width = data.reference_op_data.padding.width; + const int pad_height = data.reference_op_data.padding.height; + + const RuntimeShape& output_shape = tflite::micro::GetTensorShape(output); + const int batches = MatchingDim(input_shape, 0, output_shape, 0); + const int output_depth = MatchingDim(filter_shape, 0, output_shape, 3); + const int input_height = input_shape.Dims(1); + const int input_width = input_shape.Dims(2); + const int input_depth = input_shape.Dims(3); + const int filter_height = filter_shape.Dims(1); + const int filter_width = filter_shape.Dims(2); + const int filter_depth = filter_shape.Dims(3); + const int output_height = output_shape.Dims(1); + const int output_width = output_shape.Dims(2); + +#ifdef USE_TFLM_COMPRESSION + + MicroContext* micro_context = GetMicroContext(context); + + const CompressionTensorData* weights_comp_td = + micro_context->GetTensorCompressionData(node, kConvWeightsTensor); + const CompressionTensorData* bias_comp_td = + micro_context->GetTensorCompressionData(node, kConvBiasTensor); + +#endif // USE_TFLM_COMPRESSION + TFLITE_DCHECK(node->user_data != nullptr); + const auto& op_data = *(reinterpret_cast(node->user_data)); + ConvParams op_params = ConvParamsFloat(params, op_data.reference_op_data); + + const float32_t* input_data = tflite::micro::GetTensorData(input); +#ifdef USE_TFLM_COMPRESSION + const float32_t* filter_data = tflite::micro::GetTensorData( + micro_context, filter, weights_comp_td, + data.reference_op_data.weights_scratch_index); + const float32_t* bias_data = tflite::micro::GetOptionalTensorData( + micro_context, bias, bias_comp_td, + data.reference_op_data.bias_scratch_index); +#else // USE_TFLM_COMPRESSION + const float32_t* filter_data = tflite::micro::GetTensorData(filter); + const float32_t* bias_data = tflite::micro::GetOptionalTensorData(bias); +#endif // USE_TFLM_COMPRESSION + float32_t* output_data = tflite::micro::GetTensorData(output); + + int output_data_format = 0; + int out_length = output_height * output_width * output_depth; + int err; + if (filter_height == 1 && filter_width == 1) { + for (int batch = 0; batch < batches; ++batch) { + float32_t* p_out_temp; + p_out_temp = &output_data[batch * out_length]; + + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_pointwise_f32( + p_out_temp, const_cast(filter_data), + const_cast(&input_data[batch * input_height * + input_width * input_depth]), + const_cast(bias_data), input_height, input_width, + input_depth, output_depth, output_data_format), + 0); + + err = xa_nn_vec_activation_min_max_f32_f32( + p_out_temp, + p_out_temp, + op_params.float_activation_min, + op_params.float_activation_max, + out_length); + TF_LITE_ENSURE(context, err == 0); + } + } else if ((filter_depth == input_depth) && + ((params.dilation_width_factor == 1) && + (params.dilation_height_factor == 1))) + { + void* p_scratch = static_cast( + context->GetScratchBuffer(context, data.scratch_tensor_index)); + + for (int batch = 0; batch < batches; ++batch) { + float32_t* p_out_temp; + p_out_temp = &output_data[batch * out_length]; + + TF_LITE_ENSURE_EQ( + context, + xa_nn_conv2d_std_f32( + p_out_temp, + &input_data[batch * input_height * input_width * input_depth], + const_cast(filter_data), // filter_data, + bias_data, input_height, input_width, input_depth, + filter_height, filter_width, output_depth, stride_width, + stride_height, pad_width, pad_height, output_height, + output_width,output_data_format, static_cast(p_scratch)), + 0); + + err = xa_nn_vec_activation_min_max_f32_f32( + p_out_temp, + p_out_temp, + op_params.float_activation_min, + op_params.float_activation_max, + out_length); + TF_LITE_ENSURE(context, err == 0); + } + } + else{ + TFLITE_DCHECK(node->user_data != nullptr); + const auto& op_data = *(reinterpret_cast(node->user_data)); + tflite::reference_ops::Conv( + ConvParamsFloat(params, op_data.reference_op_data), + tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(filter), + tflite::micro::GetTensorData(filter), + tflite::micro::GetTensorShape(bias), + tflite::micro::GetOptionalTensorData(bias), + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output), + tflite::micro::GetTensorShape(nullptr), nullptr); } return kTfLiteOk; } +#endif } // namespace tflite -#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) diff --git a/tensorflow/lite/micro/kernels/xtensa/conv_int16_reference.cc b/tensorflow/lite/micro/kernels/xtensa/conv_int16_reference.cc index 317ede6d381..a81d3188294 100644 --- a/tensorflow/lite/micro/kernels/xtensa/conv_int16_reference.cc +++ b/tensorflow/lite/micro/kernels/xtensa/conv_int16_reference.cc @@ -56,17 +56,7 @@ TfLiteStatus ConvReferenceEvalInt16(TfLiteContext* context, TfLiteNode* node) { #endif // USE_TFLM_COMPRESSION - if (bias != nullptr && bias->type != kTfLiteInt32 && - bias->type != kTfLiteInt64) { - MicroPrintf("Bias type %s (%d) not supported.", - TfLiteTypeGetName(bias->type), bias->type); - return kTfLiteError; - } - const bool requires_int32_accum = - (bias != nullptr && bias->type == kTfLiteInt32) || - (bias == nullptr && params.quantized_bias_type == kTfLiteInt32); - - if (requires_int32_accum) { + if (bias != nullptr && bias->type == kTfLiteInt32) { reference_integer_ops::ConvPerChannel( ConvParamsQuantized(params, op_data), op_data.per_channel_output_multiplier, op_data.per_channel_output_shift, @@ -78,16 +68,16 @@ TfLiteStatus ConvReferenceEvalInt16(TfLiteContext* context, TfLiteNode* node) { weights_comp_td, op_data.weights_scratch_index), tflite::micro::GetTensorShape(bias), - tflite::micro::GetOptionalTensorData( + tflite::micro::GetTensorData( micro_context, bias, bias_comp_td, op_data.bias_scratch_index), #else // USE_TFLM_COMPRESSION tflite::micro::GetTensorData(filter), tflite::micro::GetTensorShape(bias), - tflite::micro::GetOptionalTensorData(bias), + tflite::micro::GetTensorData(bias), #endif // USE_TFLM_COMPRESSION tflite::micro::GetTensorShape(output), tflite::micro::GetTensorData(output)); - } else { + } else if (bias == nullptr || bias->type == kTfLiteInt64) { reference_integer_ops::ConvPerChannel( ConvParamsQuantized(params, op_data), op_data.per_channel_output_multiplier, op_data.per_channel_output_shift, @@ -99,8 +89,8 @@ TfLiteStatus ConvReferenceEvalInt16(TfLiteContext* context, TfLiteNode* node) { weights_comp_td, op_data.weights_scratch_index), tflite::micro::GetTensorShape(bias), - tflite::micro::GetOptionalTensorData( - micro_context, bias, bias_comp_td, op_data.bias_scratch_index), + tflite::micro::GetOptionalTensorData(micro_context, bias, bias_comp_td, + op_data.bias_scratch_index), #else // USE_TFLM_COMPRESSION tflite::micro::GetTensorData(filter), tflite::micro::GetTensorShape(bias), @@ -108,6 +98,10 @@ TfLiteStatus ConvReferenceEvalInt16(TfLiteContext* context, TfLiteNode* node) { #endif // USE_TFLM_COMPRESSION tflite::micro::GetTensorShape(output), tflite::micro::GetTensorData(output)); + } else { + MicroPrintf("Bias type %s (%d) not supported.", + TfLiteTypeGetName(bias->type), bias->type); + return kTfLiteError; } return kTfLiteOk; diff --git a/tensorflow/lite/micro/kernels/xtensa/conv_int8_int16_float32.cc b/tensorflow/lite/micro/kernels/xtensa/conv_int8_int16_float32.cc new file mode 100644 index 00000000000..2518b86a223 --- /dev/null +++ b/tensorflow/lite/micro/kernels/xtensa/conv_int8_int16_float32.cc @@ -0,0 +1,145 @@ +/* Copyright 2023 The TensorFlow Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#include "tensorflow/lite/c/builtin_op_data.h" +#include "tensorflow/lite/c/common.h" +#include "tensorflow/lite/kernels/internal/common.h" +#include "tensorflow/lite/kernels/internal/tensor_ctypes.h" +#include "tensorflow/lite/kernels/kernel_util.h" +#include "tensorflow/lite/micro/kernels/kernel_util.h" +#include "tensorflow/lite/micro/kernels/xtensa/xtensa.h" +#include "tensorflow/lite/micro/kernels/xtensa/xtensa_conv.h" + +namespace tflite { +namespace { + +TfLiteStatus EvalInt8(TfLiteContext* context, TfLiteNode* node) { +#if defined(HIFIMINI) + return ConvReferenceEvalInt8(context, node); +#else + const auto& op_data = *(reinterpret_cast(node->user_data)); + const auto& params = + *(reinterpret_cast(node->builtin_data)); + + const TfLiteEvalTensor* input = + tflite::micro::GetEvalInput(context, node, kConvInputTensor); + TfLiteEvalTensor* output = + tflite::micro::GetEvalOutput(context, node, kConvOutputTensor); + const TfLiteEvalTensor* filter = + tflite::micro::GetEvalInput(context, node, kConvWeightsTensor); + const TfLiteEvalTensor* bias = + tflite::micro::GetEvalInput(context, node, kConvBiasTensor); + + switch (filter->type) { + case kTfLiteInt4: { + #if defined(HIFI5) && defined(NNLIB_HIFI5) || defined(HIFI_IQ) + return ConvEvalHifiInt4(context, node, params, op_data, input, filter, + bias, output); + #elif defined(HIFI4) + TfLiteEvalTensor filter_int8 = tflite::micro::MakeUnpackedInt4Tensor( + context, op_data.reference_op_data.filter_buffer_index, filter); + return ConvEvalHifiInt8(context, node, params, op_data, input, &filter_int8, + bias, output); + #else + return ConvReferenceEvalInt8(context, node); + #endif + } + case kTfLiteInt8: { + #if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + return ConvEvalHifiInt8(context, node, params, op_data, input, filter, bias, + output); + #else + return ConvReferenceEvalInt8(context, node); + #endif + } + default: + MicroPrintf("Type %s (%d) not supported.", TfLiteTypeGetName(filter->type), + filter->type); + return kTfLiteError; + } +#endif // defined(HIFIMINI) +} + +TfLiteStatus EvalInt16(TfLiteContext* context, TfLiteNode* node) { +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + const auto& op_data = *(reinterpret_cast(node->user_data)); + const auto& params = + *(reinterpret_cast(node->builtin_data)); + + const TfLiteEvalTensor* input = + tflite::micro::GetEvalInput(context, node, kConvInputTensor); + TfLiteEvalTensor* output = + tflite::micro::GetEvalOutput(context, node, kConvOutputTensor); + const TfLiteEvalTensor* filter = + tflite::micro::GetEvalInput(context, node, kConvWeightsTensor); + const TfLiteEvalTensor* bias = + tflite::micro::GetEvalInput(context, node, kConvBiasTensor); + + if(bias == nullptr || bias->type == kTfLiteInt64){ + return ConvEvalHifiInt16(context, node, params, op_data, input, filter, bias, + output); + } + else if(bias->type == kTfLiteInt32){ + return ConvReferenceEvalInt16(context, node); + } + else{ + MicroPrintf("Bias type %s (%d) not supported.", + TfLiteTypeGetName(bias->type), bias->type); + return kTfLiteError; + } +#else + return ConvReferenceEvalInt16(context, node); +#endif +} + +TfLiteStatus EvalFloat32(TfLiteContext* context, TfLiteNode* node) { +#if defined(INCLUDE_FLOAT_OPT) && (defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ)) + const auto& op_data = *(reinterpret_cast(node->user_data)); + const auto& params = + *(reinterpret_cast(node->builtin_data)); + + const TfLiteEvalTensor* input = + tflite::micro::GetEvalInput(context, node, kConvInputTensor); + TfLiteEvalTensor* output = + tflite::micro::GetEvalOutput(context, node, kConvOutputTensor); + const TfLiteEvalTensor* filter = + tflite::micro::GetEvalInput(context, node, kConvWeightsTensor); + const TfLiteEvalTensor* bias = + tflite::micro::GetEvalInput(context, node, kConvBiasTensor); + + return ConvEvalHifiFloat32(context, node, params, op_data, input, filter, bias, + output); +#else + return ConvReferenceEvalFloat32(context, node); +#endif +} + +} // namespace + +TFLMRegistration Register_CONV_2D_INT8() { + return tflite::micro::RegisterOp(ConvInitXtensa, ConvPrepareXtensa, EvalInt8); +} + +TFLMRegistration Register_CONV_2D_INT16() { + return tflite::micro::RegisterOp(ConvInitXtensa, ConvPrepareXtensa, + EvalInt16); +} + +TFLMRegistration Register_CONV_2D_FLOAT32() { + return tflite::micro::RegisterOp(ConvInitXtensa, ConvPrepareXtensa, + EvalFloat32); +} + +} // namespace tflite diff --git a/tensorflow/lite/micro/kernels/xtensa/xtensa_conv.h b/tensorflow/lite/micro/kernels/xtensa/xtensa_conv.h index f804a6d430c..a207056beae 100644 --- a/tensorflow/lite/micro/kernels/xtensa/xtensa_conv.h +++ b/tensorflow/lite/micro/kernels/xtensa/xtensa_conv.h @@ -25,9 +25,9 @@ namespace tflite { struct XtensaConvOpData { OpDataConv reference_op_data; -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) int scratch_tensor_index; -#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) #if defined(VISION_P6) int8_t* reorder_coefficient_bias; // buffers used to keep reordered coeff and @@ -40,9 +40,19 @@ struct XtensaConvOpData { #endif // VISION_P6 }; -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) TfLiteStatus ConvPrepareHifi(TfLiteContext* context, TfLiteNode* node); +#if (defined(HIFI5) && defined(NNLIB_HIFI5)) || defined(HIFI_IQ) +TfLiteStatus ConvPrepareHifiInt4(TfLiteContext* context, TfLiteNode* node); +TfLiteStatus ConvEvalHifiInt4(TfLiteContext* context, TfLiteNode* node, + const TfLiteConvParams& params, + const XtensaConvOpData& data, + const TfLiteEvalTensor* input, + const TfLiteEvalTensor* filter, + const TfLiteEvalTensor* bias, + TfLiteEvalTensor* output); +#endif TfLiteStatus ConvEvalHifiInt8(TfLiteContext* context, TfLiteNode* node, const TfLiteConvParams& params, const XtensaConvOpData& data, @@ -59,7 +69,17 @@ TfLiteStatus ConvEvalHifiInt16(TfLiteContext* context, TfLiteNode* node, const TfLiteEvalTensor* bias, TfLiteEvalTensor* output); -#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if defined(INCLUDE_FLOAT_OPT) +TfLiteStatus ConvEvalHifiFloat32(TfLiteContext* context, TfLiteNode* node, + const TfLiteConvParams& params, + const XtensaConvOpData& data, + const TfLiteEvalTensor* input, + const TfLiteEvalTensor* filter, + const TfLiteEvalTensor* bias, + TfLiteEvalTensor* output); +#endif + +#endif // defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) #if defined(VISION_P6) @@ -79,6 +99,8 @@ TfLiteStatus ConvReferenceEvalInt8(TfLiteContext* context, TfLiteNode* node); TfLiteStatus ConvReferenceEvalInt16(TfLiteContext* context, TfLiteNode* node); +TfLiteStatus ConvReferenceEvalFloat32(TfLiteContext* context, TfLiteNode* node); + void* ConvInitXtensa(TfLiteContext* context, const char* buffer, size_t length); TfLiteStatus ConvPrepareXtensa(TfLiteContext* context, TfLiteNode* node); diff --git a/tensorflow/lite/micro/tools/make/ext_libs/xtensa.inc b/tensorflow/lite/micro/tools/make/ext_libs/xtensa.inc index c3237cf8253..acac2b63ddc 100644 --- a/tensorflow/lite/micro/tools/make/ext_libs/xtensa.inc +++ b/tensorflow/lite/micro/tools/make/ext_libs/xtensa.inc @@ -1,5 +1,46 @@ +# Explicitly add kernel sources specific to the Xtensa optimized +# implementations. +MICROLITE_CC_KERNEL_SRCS += \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/add_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/conv_common_xtensa.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/conv_hifi.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/conv_int16_reference.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/conv_float32_reference.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/conv_int8_int16_float32.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/conv_int8_reference.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/conv_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/depthwise_conv_hifi.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/depthwise_conv_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_common_xtensa.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_int8.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/pad_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/pooling_int8.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/pooling_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/reduce_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/reshape_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/softmax_int8_int16.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/softmax_vision.cc + +# Temporarily disabled: source files not yet added on this branch. +# Re-enable each entry when its corresponding kernel PR lands. +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/depthwise_conv_common_xtensa.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/depthwise_conv_int8_int16_float32.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/depthwise_conv_int8_reference.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/depthwise_conv_int16_reference.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/depthwise_conv_float32_reference.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_int16.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_float32.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/pooling_int16.cc +# $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/reduce_hifi.cc + ifeq ($(TARGET_ARCH), hifimini) + # hifimini optimizations are implemented in the TFLM repository itself. + THIRD_PARTY_KERNEL_CC_SRCS += \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/hifimini/svdf.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/hifimini/fully_connected.cc + FFT_PATH := $(MAKEFILE_DIR)/downloads/hifi_fft INCLUDES += -I$(FFT_PATH)/ @@ -8,6 +49,36 @@ ifeq ($(TARGET_ARCH), hifimini) THIRD_PARTY_CC_HDRS += \ $(shell find $(FFT_PATH)/hifi2_fft -name "*.h") +else ifeq ($(TARGET_ARCH), hifi_iq) + + PLATFORM_FLAGS = \ + -Wno-shadow\ + + CCFLAGS += $(PLATFORM_FLAGS) + CXXFLAGS += $(PLATFORM_FLAGS) + + NNLIB_PATH := $(MAKEFILE_DIR)/downloads/xa_nnlib_hifi_iq + + THIRD_PARTY_KERNEL_CC_SRCS += \ + $(shell find $(NNLIB_PATH) -name "*.c") + + EXCLUDED_NNLIB_SRCS = \ + $(NNLIB_PATH)/algo/layers/cnn/src/xa_nn_cnn_api.c \ + $(NNLIB_PATH)/algo/layers/gru/src/xa_nn_gru_api.c \ + $(NNLIB_PATH)/algo/layers/lstm/src/xa_nn_lstm_api.c \ + + THIRD_PARTY_KERNEL_CC_SRCS := $(filter-out $(EXCLUDED_NNLIB_SRCS), $(THIRD_PARTY_KERNEL_CC_SRCS)) + + THIRD_PARTY_CC_HDRS += \ + $(shell find $(NNLIB_PATH) -name "*.h") \ + + INCLUDES += \ + -I$(NNLIB_PATH)/ \ + -I$(NNLIB_PATH)/algo/kernels/ \ + -I$(NNLIB_PATH)/include/nnlib/ \ + -I$(NNLIB_PATH)/include/ \ + -I$(NNLIB_PATH)/algo/common/include/ \ + else ifeq ($(TARGET_ARCH), hifi5) DOWNLOAD_RESULT := $(shell $(MAKEFILE_DIR)/ext_libs/xtensa_download.sh ${DOWNLOADS_DIR} hifi5 $(TENSORFLOW_ROOT)) ifneq ($(DOWNLOAD_RESULT), SUCCESS) @@ -23,8 +94,12 @@ else ifeq ($(TARGET_ARCH), hifi5) # not have separate cflags (or the concept of modular build targets) with the # Makefile, -Wno-shadow will be used for everything. + # TODO: Adding HIFI_SIMD_WIDTH here to avoid error in ndsp lib compilation, + # once ndsp headers are removed from NNLib, this won't be needed + PLATFORM_FLAGS = \ -DNNLIB_HIFI5 \ + -DHIFI_SIMD_WIDTH=16 \ -Wno-shadow CCFLAGS += $(PLATFORM_FLAGS) @@ -61,9 +136,11 @@ else ifeq ($(TARGET_ARCH), hifi5) -I$(NNLIB_PATH)/include/nnlib/ \ -I$(NNLIB_PATH)/include/ \ -I$(NNLIB_PATH)/algo/common/include/ \ + -I$(NNLIB_PATH)/algo/ndsp/hifi5/include/ \ -I$(NDSPLIB_PATH)/library/include/ \ -I$(NDSPLIB_PATH)/library/include_private/ -else ifeq ($(TARGET_ARCH), $(filter $(TARGET_ARCH), hifi3 hifi4)) + +else ifeq ($(TARGET_ARCH), $(filter $(TARGET_ARCH), hifi4 hifi4_internal hifi3 hifi3z fusion_f1 hifi1)) # NNLib hifi4 also supports hifi3 DOWNLOAD_RESULT := $(shell $(MAKEFILE_DIR)/ext_libs/xtensa_download.sh ${DOWNLOADS_DIR} hifi4 $(TENSORFLOW_ROOT)) ifneq ($(DOWNLOAD_RESULT), SUCCESS) @@ -78,9 +155,13 @@ else ifeq ($(TARGET_ARCH), $(filter $(TARGET_ARCH), hifi3 hifi4)) # TODO(b/161489252): -Wno-shadow is only needed for xannlib. But since we do # not have separate cflags (or the concept of modular build targets) with the # Makefile, -Wno-shadow will be used for everything. + + # TODO: Adding HIFI_SIMD_WIDTH here to avoid error in ndsp lib compilation, + # once ndsp headers are removed from NNLib, this won't be needed PLATFORM_FLAGS = \ -DNNLIB_V2 \ + -DHIFI_SIMD_WIDTH=8 \ -Wno-shadow CCFLAGS += $(PLATFORM_FLAGS) @@ -104,7 +185,7 @@ else ifeq ($(TARGET_ARCH), $(filter $(TARGET_ARCH), hifi3 hifi4)) EXCLUDED_NNLIB_SRCS = \ $(NNLIB_PATH)/algo/layers/cnn/src/xa_nn_cnn_api.c \ $(NNLIB_PATH)/algo/layers/gru/src/xa_nn_gru_api.c \ - $(NNLIB_PATH)/algo/layers/lstm/src/xa_nn_lstm_api.c + $(NNLIB_PATH)/algo/layers/lstm/src/xa_nn_lstm_api.c ifeq ($(TARGET_ARCH), hifi3) EXCLUDED_NNLIB_SRCS += \ @@ -116,7 +197,6 @@ else ifeq ($(TARGET_ARCH), $(filter $(TARGET_ARCH), hifi3 hifi4)) ifeq ($(TARGET_ARCH), hifi4) EXCLUDED_NNLIB_SRCS += \ - $(NNLIB_PATH)/algo/kernels/activations/hifi4/xa_nn_activations_asym8_asym8.c \ $(NNLIB_PATH)/algo/kernels/norm/hifi4/xa_nn_norm3D_16.c endif @@ -132,6 +212,7 @@ else ifeq ($(TARGET_ARCH), $(filter $(TARGET_ARCH), hifi3 hifi4)) -I$(NNLIB_PATH)/include/nnlib/ \ -I$(NNLIB_PATH)/include/ \ -I$(NNLIB_PATH)/algo/common/include/ \ + -I$(NNLIB_PATH)/algo/ndsp/hifi4/include/ \ -I$(NDSPLIB_PATH)/library/include/ \ -I$(NDSPLIB_PATH)/library/include_private/ @@ -168,3 +249,21 @@ else ifeq ($(TARGET_ARCH), vision_p6) else $(error Unsupported TARGET_ARCH=$(TARGET_ARCH)) endif + +FFT_PATH := $(MAKEFILE_DIR)/downloads/hifi_fft + +INCLUDES += -I$(FFT_PATH)/ + +ifeq ($(TARGET_ARCH), $(filter $(TARGET_ARCH), hifi3 hifi3z hifi4 hifi4_internal hifi5 hifi1)) +THIRD_PARTY_KERNEL_CC_SRCS += \ + $(shell find $(FFT_PATH)/hifi3_fft -name "*.c") + +THIRD_PARTY_CC_HDRS += \ + $(shell find $(FFT_PATH)/hifi3_fft -name "*.h") +else ifeq ($(TARGET_ARCH), hifimini) +THIRD_PARTY_KERNEL_CC_SRCS += \ + $(shell find $(FFT_PATH)/hifi2_fft -name "*.c") + +THIRD_PARTY_CC_HDRS += \ + $(shell find $(FFT_PATH)/hifi2_fft -name "*.h") +endif From 24580e8055352dd8e7294c2a90dad461e909ac50 Mon Sep 17 00:00:00 2001 From: unmeshna Date: Mon, 21 Sep 2026 02:36:09 -0700 Subject: [PATCH 2/2] Include NNLib headers for HIFI_IQ target in xtensa.h Add HIFI_IQ to the include guard so xa_nnlib_api.h and xa_nnlib_standards.h are pulled in for the hifi_iq build. --- tensorflow/lite/micro/kernels/xtensa/xtensa.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tensorflow/lite/micro/kernels/xtensa/xtensa.h b/tensorflow/lite/micro/kernels/xtensa/xtensa.h index 0e7e51b0cb6..cdcb1f75eb3 100644 --- a/tensorflow/lite/micro/kernels/xtensa/xtensa.h +++ b/tensorflow/lite/micro/kernels/xtensa/xtensa.h @@ -22,7 +22,7 @@ limitations under the License. #include "tensorflow/lite/micro/kernels/xtensa/fixedpoint_utils_hifimini.h" #endif // defined(HIFMINI) -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) #include "include/nnlib/xa_nnlib_api.h" #include "include/nnlib/xa_nnlib_standards.h"