From 33f2a9199e6800eb57130317aadf5ae010298687 Mon Sep 17 00:00:00 2001 From: unmeshna Date: Wed, 16 Sep 2026 05:16:00 -0700 Subject: [PATCH 1/2] Add HiFi-optimized Xtensa Logistic and Tanh kernels Optimize the Xtensa logistic and tanh kernels using the HiFi NNLib LUT ops (int8 + sym16s), adding logistic_common.cc (LUT init) and xtensa_logistic.h. Registers logistic_common.cc in xtensa_makefile.inc. Depends on the 09_03_2026 NNLib release (PR#3703) for the LUT APIs (e.g. xa_nn_init_lut_asym8s_sigmoid, xa_nn_init_lut_asym8s_tanh, xa_nn_vec_apply_lut_asym8s_asym8s). --- .../lite/micro/kernels/xtensa/logistic.cc | 39 ++- .../micro/kernels/xtensa/logistic_common.cc | 156 +++++++++++ tensorflow/lite/micro/kernels/xtensa/tanh.cc | 255 ++++++++++++++++++ .../micro/kernels/xtensa/xtensa_logistic.h | 31 +++ .../tools/make/targets/xtensa_makefile.inc | 1 + 5 files changed, 468 insertions(+), 14 deletions(-) create mode 100644 tensorflow/lite/micro/kernels/xtensa/logistic_common.cc create mode 100644 tensorflow/lite/micro/kernels/xtensa/tanh.cc create mode 100644 tensorflow/lite/micro/kernels/xtensa/xtensa_logistic.h diff --git a/tensorflow/lite/micro/kernels/xtensa/logistic.cc b/tensorflow/lite/micro/kernels/xtensa/logistic.cc index 2ddf82eff50..c64176b4d68 100644 --- a/tensorflow/lite/micro/kernels/xtensa/logistic.cc +++ b/tensorflow/lite/micro/kernels/xtensa/logistic.cc @@ -26,6 +26,7 @@ limitations under the License. #include "tensorflow/lite/micro/kernels/kernel_util.h" #include "tensorflow/lite/micro/kernels/logistic.h" #include "tensorflow/lite/micro/kernels/xtensa/xtensa.h" +#include "tensorflow/lite/micro/kernels/xtensa/xtensa_logistic.h" #include "tensorflow/lite/micro/micro_log.h" namespace tflite { @@ -33,7 +34,7 @@ namespace { void* LogisticInit(TfLiteContext* context, const char* buffer, size_t length) { TFLITE_DCHECK(context->AllocatePersistentBuffer != nullptr); - return context->AllocatePersistentBuffer(context, sizeof(OpDataLogistic)); + return context->AllocatePersistentBuffer(context, sizeof(OpDataLogisticXtensa)); } TfLiteStatus LogisticEval(TfLiteContext* context, TfLiteNode* node) { @@ -43,7 +44,9 @@ TfLiteStatus LogisticEval(TfLiteContext* context, TfLiteNode* node) { tflite::micro::GetEvalOutput(context, node, kLogisticOutputTensor); TFLITE_DCHECK(node->user_data != nullptr); - OpDataLogistic* data = static_cast(node->user_data); + OpDataLogisticXtensa* xtensa_data = + static_cast(node->user_data); + OpDataLogistic* data = &xtensa_data->reference_op_data; if (input->type != output->type) { MicroPrintf( @@ -54,7 +57,7 @@ TfLiteStatus LogisticEval(TfLiteContext* context, TfLiteNode* node) { switch (input->type) { case kTfLiteFloat32: { -#if HIFI_VFPU && (defined(HIFI3) || defined(HIFI4) || defined(HIFI5)) +#if defined(INCLUDE_FLOAT_OPT) && (defined(HIFI3) || defined(HIFI4) || defined(HIFI5)) const RuntimeShape& input_shape = tflite::micro::GetTensorShape(input); const RuntimeShape& output_shape = tflite::micro::GetTensorShape(output); const int flat_size = MatchingFlatSize(input_shape, output_shape); @@ -70,25 +73,24 @@ TfLiteStatus LogisticEval(TfLiteContext* context, TfLiteNode* node) { tflite::micro::GetTensorData(input), tflite::micro::GetTensorShape(output), tflite::micro::GetTensorData(output)); -#endif // HIFI_VFPU && (defined(HIFI3) || defined(HIFI4) || defined(HIFI5)) +#endif // defined(INCLUDE_FLOAT_OPT) && (defined(HIFI3) || defined(HIFI4) || defined(HIFI5)) break; } case kTfLiteInt8: { -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) const RuntimeShape& input_shape = tflite::micro::GetTensorShape(input); const RuntimeShape& output_shape = tflite::micro::GetTensorShape(output); const int flat_size = MatchingFlatSize(input_shape, output_shape); - const int8_t* input_data_ptr = - tflite::micro::GetTensorData(input); + const int8_t* input_data_ptr = tflite::micro::GetTensorData(input); int8_t* output_data_ptr = tflite::micro::GetTensorData(output); TF_LITE_ENSURE_EQ( context, - xa_nn_vec_sigmoid_asym8s_asym8s( - output_data_ptr, input_data_ptr, data->input_zero_point, - data->input_range_radius, data->input_multiplier, - data->input_left_shift, flat_size), + xa_nn_vec_apply_lut_asym8s_asym8s( + output_data_ptr, input_data_ptr, + static_cast(xtensa_data->sigmoid_lut), + 256, flat_size), 0); #else reference_integer_ops::Logistic( @@ -96,18 +98,27 @@ TfLiteStatus LogisticEval(TfLiteContext* context, TfLiteNode* node) { data->input_multiplier, data->input_left_shift, NumElements(input->dims), tflite::micro::GetTensorData(input), tflite::micro::GetTensorData(output)); -#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) break; } case kTfLiteInt16: { switch (output->type) { - case kTfLiteInt16: + case kTfLiteInt16 : { +#if defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + TF_LITE_ENSURE_EQ(context, xa_nn_vec_sigmoid_sym16s_sym16s(tflite::micro::GetTensorData(output), + tflite::micro::GetTensorData(input), + data->input_multiplier, + data->input_left_shift, + NumElements(input->dims)), 0); +#else reference_integer_ops::Logistic( data->input_multiplier, data->input_left_shift, NumElements(input->dims), tflite::micro::GetTensorData(input), tflite::micro::GetTensorData(output)); - break; +#endif + return kTfLiteOk; + } break; default: MicroPrintf("Input %s, output %s not supported.", TfLiteTypeGetName(input->type), diff --git a/tensorflow/lite/micro/kernels/xtensa/logistic_common.cc b/tensorflow/lite/micro/kernels/xtensa/logistic_common.cc new file mode 100644 index 00000000000..c800e98cbaa --- /dev/null +++ b/tensorflow/lite/micro/kernels/xtensa/logistic_common.cc @@ -0,0 +1,156 @@ +/* Copyright 2021 The TensorFlow Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#include "tensorflow/lite/c/builtin_op_data.h" +#include "tensorflow/lite/c/common.h" +#include "tensorflow/lite/kernels/internal/common.h" +#include "tensorflow/lite/kernels/internal/quantization_util.h" +#include "tensorflow/lite/kernels/internal/reference/integer_ops/logistic.h" +#include "tensorflow/lite/kernels/internal/reference/logistic.h" +#include "tensorflow/lite/kernels/internal/tensor_ctypes.h" +#include "tensorflow/lite/kernels/kernel_util.h" +#include "tensorflow/lite/kernels/op_macros.h" +#include "tensorflow/lite/micro/kernels/kernel_util.h" +#include "tensorflow/lite/micro/kernels/logistic.h" +#include "tensorflow/lite/micro/kernels/xtensa/xtensa.h" +#include "tensorflow/lite/micro/kernels/xtensa/xtensa_logistic.h" +#if defined(USE_HIFI_ACT_TIE) +#include +#endif + +namespace tflite { +const int kLogisticInputTensor = 0; +const int kLogisticOutputTensor = 0; + +TfLiteStatus CalculateArithmeticOpDataLogistic(TfLiteContext* context, + TfLiteNode* node, + OpDataLogistic* data) { + MicroContext* micro_context = GetMicroContext(context); + + TfLiteTensor* input = + micro_context->AllocateTempInputTensor(node, kLogisticInputTensor); + TF_LITE_ENSURE(context, input != nullptr); + TfLiteTensor* output = + micro_context->AllocateTempOutputTensor(node, kLogisticOutputTensor); + TF_LITE_ENSURE(context, output != nullptr); + + TF_LITE_ENSURE_TYPES_EQ(context, input->type, output->type); + if (input->type == kTfLiteInt8) { + TF_LITE_ENSURE_EQ(context, output->params.zero_point, + std::numeric_limits::min()); + + static constexpr int kInputIntegerBits = 4; + const double input_real_multiplier = + static_cast(input->params.scale) * + static_cast(1 << (31 - kInputIntegerBits)); + + data->input_zero_point = input->params.zero_point; + + const double q = std::frexp(input_real_multiplier, &data->input_left_shift); + data->input_multiplier = static_cast(TfLiteRound(q * (1ll << 31))); + + data->input_range_radius = + CalculateInputRadius(kInputIntegerBits, data->input_left_shift, 31); + } + + if (input->type == kTfLiteInt16) { + static constexpr int kInputIntegerBits = 3; + static constexpr int kOutputFractionalBits = 15; + + // See comments in TanhPrepare about requiring zero_point==0 + // and a power-of-two ("POT") scale. + + TF_LITE_ENSURE_EQ(context, input->params.zero_point, 0); + TF_LITE_ENSURE_EQ(context, output->params.zero_point, 0); + + int input_scale_log2_rounded; + bool param_scale_pot = + CheckedLog2(input->params.scale, &input_scale_log2_rounded); + + data->input_left_shift = + (15 - kInputIntegerBits) + input_scale_log2_rounded; + param_scale_pot &= (data->input_left_shift == 0); + + if (param_scale_pot) { + data->input_multiplier = 0; + } else { + // Calculate multiplier to change input scale to 1/(3*4096) + // as required by the table lookup. + // In this scaling +/-2^17 represents +/-10.7 +#if (defined(USE_HIFI_ACT_TIE) && (defined(AE_SIGMOID16X4) || defined(AE_SIGMOID16X4X2))) + double multiplier = + static_cast(input->params.scale) * 4096.0; +#else + double multiplier = + static_cast(input->params.scale) * 4096.0 * 3.0; +#endif + data->input_left_shift = 0; + + while (multiplier <= 32767.0 / 2.0 && data->input_left_shift <= 30) { + data->input_left_shift++; + multiplier = multiplier * 2.0; + } + + data->input_multiplier = static_cast(multiplier); + } + + int output_scale_log2_rounded; + TF_LITE_ENSURE( + context, CheckedLog2(output->params.scale, &output_scale_log2_rounded)); + TF_LITE_ENSURE_EQ(context, output_scale_log2_rounded, + -kOutputFractionalBits); + } + + micro_context->DeallocateTempTfLiteTensor(input); + micro_context->DeallocateTempTfLiteTensor(output); + return kTfLiteOk; +} + +TfLiteStatus LogisticPrepare(TfLiteContext* context, TfLiteNode* node) { + TFLITE_DCHECK(node->user_data != nullptr); + auto* xtensa_data = static_cast(node->user_data); + TF_LITE_ENSURE_OK(context, CalculateArithmeticOpDataLogistic( + context, node, &xtensa_data->reference_op_data)); + +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + MicroContext* micro_context = GetMicroContext(context); + TfLiteTensor* input = + micro_context->AllocateTempInputTensor(node, kLogisticInputTensor); + TF_LITE_ENSURE(context, input != nullptr); + if (input->type == kTfLiteInt8) + { + void* raw = context->AllocatePersistentBuffer( + context, sizeof(int8_t) * 256); + TF_LITE_ENSURE(context, raw != nullptr); + xtensa_data->sigmoid_lut = raw; + TF_LITE_ENSURE_EQ( + context, + xa_nn_init_lut_asym8s_sigmoid( + static_cast(xtensa_data->sigmoid_lut), + xtensa_data->reference_op_data.input_zero_point, + xtensa_data->reference_op_data.input_range_radius, + xtensa_data->reference_op_data.input_multiplier, + xtensa_data->reference_op_data.input_left_shift), + 0); + } else { + xtensa_data->sigmoid_lut = nullptr; + } + + micro_context->DeallocateTempTfLiteTensor(input); +#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + return kTfLiteOk; +} + +} // namespace tflite diff --git a/tensorflow/lite/micro/kernels/xtensa/tanh.cc b/tensorflow/lite/micro/kernels/xtensa/tanh.cc new file mode 100644 index 00000000000..cbf1ca4a5c3 --- /dev/null +++ b/tensorflow/lite/micro/kernels/xtensa/tanh.cc @@ -0,0 +1,255 @@ +/* Copyright 2020 The TensorFlow Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#include "tensorflow/lite/kernels/internal/reference/integer_ops/tanh.h" + +#include "tensorflow/lite/c/builtin_op_data.h" +#include "tensorflow/lite/c/common.h" +#include "tensorflow/lite/kernels/internal/common.h" +#include "tensorflow/lite/kernels/internal/quantization_util.h" +#include "tensorflow/lite/kernels/internal/reference/tanh.h" +#include "tensorflow/lite/kernels/internal/tensor_ctypes.h" +#include "tensorflow/lite/kernels/kernel_util.h" +#include "tensorflow/lite/kernels/op_macros.h" +#include "tensorflow/lite/micro/kernels/kernel_util.h" +#include "tensorflow/lite/micro/kernels/xtensa/xtensa.h" +#include "tensorflow/lite/micro/micro_utils.h" +#if defined(USE_HIFI_ACT_TIE) +#include +#endif + +namespace tflite { + +namespace { + +constexpr int kInputTensor = 0; +constexpr int kOutputTensor = 0; + +struct OpData { + int32_t input_zero_point; + int32_t input_range_radius; + int32_t input_multiplier; + int input_left_shift; + void* tanh_lut; +}; + +void* TanhInit(TfLiteContext* context, const char* buffer, size_t length) { + TFLITE_DCHECK(context->AllocatePersistentBuffer != nullptr); + return context->AllocatePersistentBuffer(context, sizeof(OpData)); +} + +TfLiteStatus CalculateArithmeticOpData(TfLiteContext* context, TfLiteNode* node, + OpData* data) { + MicroContext* micro_context = GetMicroContext(context); + TF_LITE_ENSURE_EQ(context, NumInputs(node), 1); + TF_LITE_ENSURE_EQ(context, NumOutputs(node), 1); + TfLiteTensor* input = + micro_context->AllocateTempInputTensor(node, kInputTensor); + TF_LITE_ENSURE(context, input != nullptr); + TfLiteTensor* output = + micro_context->AllocateTempOutputTensor(node, kOutputTensor); + TF_LITE_ENSURE(context, output != nullptr); + + TF_LITE_ENSURE_TYPES_EQ(context, input->type, output->type); + + if (input->type == kTfLiteInt8) { + static constexpr int kInputIntegerBits = 4; + const double input_real_multiplier = + static_cast(input->params.scale) * + static_cast(1 << (31 - kInputIntegerBits)); + + const double q = std::frexp(input_real_multiplier, &data->input_left_shift); + data->input_multiplier = static_cast(TfLiteRound(q * (1ll << 31))); + + data->input_range_radius = + CalculateInputRadius(kInputIntegerBits, data->input_left_shift, 31); + } + + if (input->type == kTfLiteInt16) { + static constexpr int kInputIntegerBits = 3; + static constexpr int kOutputFractionalBits = 15; + + // These operators are implemented in fixed-point arithmetic, + // which intrinsically wants symmetric ranges (zero_point==0) + // and power-of-two scales (power-of-two is abbreviated below as POT). + // While more general support would be possible by means of rescaling, + // that would add some overhead and some loss of accuracy and wouldn't + // be used at the moment as current quantized LSTM applications are + // happy with symmetric, power-of-two-scales quantization. So we just + // implement that narrow case only for now. + + TF_LITE_ENSURE_EQ(context, input->params.zero_point, 0); + TF_LITE_ENSURE_EQ(context, output->params.zero_point, 0); + + int input_scale_log2_rounded; + bool param_scale_pot = + CheckedLog2(input->params.scale, &input_scale_log2_rounded); + + data->input_left_shift = + (15 - kInputIntegerBits) + input_scale_log2_rounded; + param_scale_pot &= + (data->input_left_shift == 0 || data->input_left_shift == 1); + + if (param_scale_pot) { + data->input_multiplier = 0; + } else { + // Calculate multiplier to change input scale to 1/(3*4096) + // as required by the table lookup. + // The number 3.0 in the multiplier comes from here, + // because the interval is [-10.7, 10.7] instead of [-8, 8]. + // So, in this scaling +/-2^17 represents +/-10.7. +#if (defined(USE_HIFI_ACT_TIE) && (defined(AE_TANH16X4) || defined(AE_TANH16X4X2))) + double multiplier = + static_cast(input->params.scale) * 4096.0; +#else + double multiplier = + static_cast(input->params.scale) * 4096.0 * 3.0; +#endif + data->input_left_shift = 0; + + while (multiplier <= 32767.0 / 2.0 && data->input_left_shift <= 30) { + data->input_left_shift++; + multiplier = multiplier * 2.0; + } + + data->input_multiplier = static_cast(multiplier); + } + + int output_scale_log2_rounded; + TF_LITE_ENSURE( + context, CheckedLog2(output->params.scale, &output_scale_log2_rounded)); + TF_LITE_ENSURE_EQ(context, output_scale_log2_rounded, + -kOutputFractionalBits); + } + + micro_context->DeallocateTempTfLiteTensor(input); + micro_context->DeallocateTempTfLiteTensor(output); + return kTfLiteOk; +} + +TfLiteStatus TanhPrepare(TfLiteContext* context, TfLiteNode* node) { + TFLITE_DCHECK(node->user_data != nullptr); + + OpData* data = static_cast(node->user_data); + + MicroContext* micro_context = GetMicroContext(context); + TfLiteTensor* input = + micro_context->AllocateTempInputTensor(node, kInputTensor); + TF_LITE_ENSURE(context, input != nullptr); + data->input_zero_point = input->params.zero_point; + TF_LITE_ENSURE_OK(context, CalculateArithmeticOpData(context, node, data)); + +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + if (input->type == kTfLiteInt8) { + void* raw = + context->AllocatePersistentBuffer(context, 256 * sizeof(int8_t)); + TF_LITE_ENSURE(context, raw != nullptr); + data->tanh_lut = raw; + TF_LITE_ENSURE_EQ( + context, + xa_nn_init_lut_asym8s_tanh( + static_cast(data->tanh_lut), + data->input_zero_point, + data->input_range_radius, + data->input_multiplier, + data->input_left_shift), + 0); + } else { + data->tanh_lut = nullptr; + } +#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + + micro_context->DeallocateTempTfLiteTensor(input); + return kTfLiteOk; +} + +TfLiteStatus TanhEval(TfLiteContext* context, TfLiteNode* node) { + const TfLiteEvalTensor* input = + tflite::micro::GetEvalInput(context, node, kInputTensor); + TfLiteEvalTensor* output = + tflite::micro::GetEvalOutput(context, node, kOutputTensor); + + TFLITE_DCHECK(node->user_data != nullptr); + OpData* data = static_cast(node->user_data); + + switch (input->type) { + case kTfLiteFloat32: { + reference_ops::Tanh(tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output)); + return kTfLiteOk; + } break; + case kTfLiteInt16: { +#if defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + int32_t vec_len = MatchingFlatSize(tflite::micro::GetTensorShape(input), tflite::micro::GetTensorShape(output)); + TF_LITE_ENSURE_EQ(context, xa_nn_vec_tanh_sym16s_sym16s(tflite::micro::GetTensorData(output), + tflite::micro::GetTensorData(input), + data->input_multiplier, + data->input_left_shift, + vec_len), 0); +#else + reference_integer_ops::Tanh( + data->input_multiplier, data->input_left_shift, + tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output)); +#endif // defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + return kTfLiteOk; + } break; + case kTfLiteInt8: { +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + const int8_t *input_data_ptr; + int8_t *output_data_ptr; + const RuntimeShape& input_shape = tflite::micro::GetTensorShape(input); + const RuntimeShape& output_shape = tflite::micro::GetTensorShape(output); + const int flat_size = MatchingFlatSize(input_shape, output_shape); + + input_data_ptr = tflite::micro::GetTensorData(input); + output_data_ptr = tflite::micro::GetTensorData(output); + + TF_LITE_ENSURE_EQ( + context, + xa_nn_vec_apply_lut_asym8s_asym8s( + output_data_ptr, input_data_ptr, + static_cast(data->tanh_lut), + 256, flat_size), + 0); +#else + reference_integer_ops::Tanh( + data->input_zero_point, data->input_range_radius, data->input_multiplier, + data->input_left_shift, tflite::micro::GetTensorShape(input), + tflite::micro::GetTensorData(input), + tflite::micro::GetTensorShape(output), + tflite::micro::GetTensorData(output)); +#endif // defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) + return kTfLiteOk; + } break; + default: + TF_LITE_KERNEL_LOG(context, "Input %s, output %s not supported.", + TfLiteTypeGetName(input->type), + TfLiteTypeGetName(output->type)); + return kTfLiteError; + } +} + +} // namespace + +TFLMRegistration Register_TANH() { + return tflite::micro::RegisterOp(TanhInit, TanhPrepare, TanhEval); +} + +} // namespace tflite diff --git a/tensorflow/lite/micro/kernels/xtensa/xtensa_logistic.h b/tensorflow/lite/micro/kernels/xtensa/xtensa_logistic.h new file mode 100644 index 00000000000..991cd3f535f --- /dev/null +++ b/tensorflow/lite/micro/kernels/xtensa/xtensa_logistic.h @@ -0,0 +1,31 @@ +/* Copyright 2026 The TensorFlow Authors. All Rights Reserved. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +==============================================================================*/ + +#ifndef TENSORFLOW_LITE_MICRO_KERNELS_XTENSA_XTENSA_LOGISTIC_H_ +#define TENSORFLOW_LITE_MICRO_KERNELS_XTENSA_XTENSA_LOGISTIC_H_ + +#include "tensorflow/lite/micro/kernels/logistic.h" + +namespace tflite { + +struct OpDataLogisticXtensa { + OpDataLogistic reference_op_data; + // Points to a 256-entry int8_t sigmoid LUT for kTfLiteInt8 inputs; + void* sigmoid_lut; +}; + +} // namespace tflite + +#endif // TENSORFLOW_LITE_MICRO_KERNELS_XTENSA_XTENSA_LOGISTIC_H_ diff --git a/tensorflow/lite/micro/tools/make/targets/xtensa_makefile.inc b/tensorflow/lite/micro/tools/make/targets/xtensa_makefile.inc index 78d3dce16dc..4b9532b17c9 100644 --- a/tensorflow/lite/micro/tools/make/targets/xtensa_makefile.inc +++ b/tensorflow/lite/micro/tools/make/targets/xtensa_makefile.inc @@ -124,6 +124,7 @@ ifeq ($(OPTIMIZED_KERNEL_DIR), xtensa) $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_hifimini.cc \ $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_int8.cc \ $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/fully_connected_vision.cc \ + $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/logistic_common.cc \ $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/lstm_eval_hifi.cc \ $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/pad_vision.cc \ $(TENSORFLOW_ROOT)tensorflow/lite/micro/kernels/xtensa/pooling_int8.cc \ From 9cfbddd1cfcf8f5a328ee2ce925506f2ffd46a7e Mon Sep 17 00:00:00 2001 From: unmeshna Date: Mon, 21 Sep 2026 02:36:09 -0700 Subject: [PATCH 2/2] Include NNLib headers for HIFI_IQ target in xtensa.h Add HIFI_IQ to the include guard so xa_nnlib_api.h and xa_nnlib_standards.h are pulled in for the hifi_iq build. --- tensorflow/lite/micro/kernels/xtensa/xtensa.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tensorflow/lite/micro/kernels/xtensa/xtensa.h b/tensorflow/lite/micro/kernels/xtensa/xtensa.h index 0e7e51b0cb6..cdcb1f75eb3 100644 --- a/tensorflow/lite/micro/kernels/xtensa/xtensa.h +++ b/tensorflow/lite/micro/kernels/xtensa/xtensa.h @@ -22,7 +22,7 @@ limitations under the License. #include "tensorflow/lite/micro/kernels/xtensa/fixedpoint_utils_hifimini.h" #endif // defined(HIFMINI) -#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) +#if defined(HIFI3) || defined(HIFI4) || defined(HIFI5) || defined(HIFI_IQ) #include "include/nnlib/xa_nnlib_api.h" #include "include/nnlib/xa_nnlib_standards.h"