tflite: mtk_neuron: add neuron delegate self abortion test

This is the initial CL of testing the neuron delegate
preemption/scheduling behavior.

This CL includes:
* A general model structure that we could control its depth, width, and
  inference speed.
* A test to ensure the model will be aborted when
  inference_abort_time_ms is given.

BUG=b:374245241
TEST=neuron_delegate_test on navi

Change-Id: Ibeec61b3c5b88e1af8f2c64f8bafd9573079a241
Reviewed-on: https://chromium-review.googlesource.com/c/chromiumos/platform/tflite/+/6176631
Tested-by: Tommy Chiang <ototot@google.com>
Reviewed-by: Shik Chen <shik@chromium.org>
Commit-Queue: Tommy Chiang <ototot@google.com>
diff --git a/delegate/mtk_neuron/BUILD.bazel b/delegate/mtk_neuron/BUILD.bazel
index 66f5e9d..65dfc10 100644
--- a/delegate/mtk_neuron/BUILD.bazel
+++ b/delegate/mtk_neuron/BUILD.bazel
@@ -4,6 +4,7 @@
 
 load("@org_tensorflow//tensorflow/lite:build_def.bzl", "tflite_cc_shared_object")
 load("@org_tensorflow//tensorflow/lite/core/shims:cc_library_with_tflite.bzl", "cc_library_with_tflite")
+load("@rules_cc//cc:defs.bzl", "cc_binary")
 
 exports_files(["settings.json"])
 
@@ -105,3 +106,24 @@
         "@org_tensorflow//tensorflow/lite/delegates/utils/experimental/stable_delegate:version_script.lds",
     ],
 )
+
+cc_binary(
+    name = "neuron_delegate_test",
+    testonly = True,
+    srcs = [
+        "neuron_delegate_test.cc",
+    ],
+    deps = [
+        "//common:simple_model_builder",
+        "//common:sync_driver",
+        "//common/test_util:aligned_tflite_buffer_allocator",
+        "//common/test_util:fp16_compare",
+        "@com_google_absl//absl/random",
+        "@com_google_absl//absl/strings",
+        "@com_google_absl//absl/time",
+        "@com_google_googletest//:gtest",
+        "@org_tensorflow//tensorflow/lite:framework",
+        "@org_tensorflow//tensorflow/lite/delegates/utils/experimental/stable_delegate:delegate_loader",
+        "@org_tensorflow//tensorflow/lite/kernels:builtin_ops",
+    ],
+)
diff --git a/delegate/mtk_neuron/neuron_delegate_test.cc b/delegate/mtk_neuron/neuron_delegate_test.cc
new file mode 100644
index 0000000..9dab59e
--- /dev/null
+++ b/delegate/mtk_neuron/neuron_delegate_test.cc
@@ -0,0 +1,308 @@
+/*
+ * Copyright 2025 The ChromiumOS Authors
+ * Use of this source code is governed by a BSD-style license that can be
+ * found in the LICENSE file.
+ */
+
+#include <gmock/gmock.h>
+#include <gtest/gtest.h>
+
+#include <string>
+#include <vector>
+
+#include "absl/random/random.h"
+#include "absl/strings/str_format.h"
+#include "absl/time/clock.h"
+#include "absl/time/time.h"
+#include "common/simple_model_builder.h"
+#include "common/sync_driver.h"
+#include "common/test_util/aligned_tflite_buffer_allocator.h"
+#include "common/test_util/fp16_compare.h"
+#include "tensorflow/lite/acceleration/configuration/configuration_generated.h"
+#include "tensorflow/lite/core/c/builtin_op_data.h"
+#include "tensorflow/lite/delegates/utils/experimental/stable_delegate/delegate_loader.h"
+
+namespace tflite::cros::tests {
+
+using TfLiteDelegatePtr = Interpreter::TfLiteDelegatePtr;
+
+TFLiteSettingsT GetDefaultSettings() {
+  TFLiteSettingsT settings;
+  settings.stable_delegate_loader_settings =
+      std::make_unique<StableDelegateLoaderSettingsT>();
+  settings.stable_delegate_loader_settings->delegate_name =
+      "mtk_neuron_delegate";
+  settings.stable_delegate_loader_settings->delegate_path =
+      "/usr/lib64/libtensorflowlite_mtk_neuron_delegate.so";
+  settings.mtk_neuron_settings = std::make_unique<tflite::MtkNeuronSettingsT>();
+  settings.mtk_neuron_settings->allow_fp16_precision_for_fp32 = true;
+  return settings;
+}
+
+TfLiteDelegatePtr GetDelegateFromSettings(TFLiteSettingsT const* settings) {
+  auto delegate = tflite::delegates::utils::LoadDelegateFromSharedLibrary(
+      settings->stable_delegate_loader_settings->delegate_path);
+
+  flatbuffers::FlatBufferBuilder fbb;
+  const flatbuffers::Offset<TFLiteSettings> settings_buffer =
+      TFLiteSettings::Pack(fbb, settings);
+  fbb.Finish(settings_buffer);
+
+  const TFLiteSettings* settings_ptr =
+      flatbuffers::GetRoot<TFLiteSettings>(fbb.GetBufferPointer());
+  return std::unique_ptr<TfLiteOpaqueDelegate, void (*)(TfLiteOpaqueDelegate*)>(
+      delegate->delegate_plugin->create(settings_ptr),
+      delegate->delegate_plugin->destroy);
+}
+
+void SumTo(SimpleModelBuilder& mb, std::vector<int> shape, std::string prefix,
+           std::vector<int> inputs, int result, int depth = 1) {
+  if (inputs.size() == 2) {
+    mb.AddOperator<TfLiteAddParams>({
+        .op = kTfLiteBuiltinAdd,
+        .inputs = inputs,
+        .outputs = {result},
+    });
+  } else {
+    std::vector<int> sum_results;
+    for (int i = 1; i < std::ssize(inputs); i += 2) {
+      sum_results.push_back(mb.AddInternalTensor(SimpleModelBuilder::TensorArgs{
+          .name = prefix + absl::StrFormat("sum_results_%d_%d", i, depth),
+          .type = kTfLiteFloat32,
+          .shape = shape,
+      }));
+      mb.AddOperator<TfLiteAddParams>({
+          .op = kTfLiteBuiltinAdd,
+          .inputs = {inputs[i - 1], inputs[i]},
+          .outputs = {sum_results.back()},
+      });
+    }
+    if (inputs.size() % 2 == 1) {
+      sum_results.push_back(inputs.back());
+    }
+    SumTo(mb, shape, prefix, sum_results, result, depth + 1);
+  }
+}
+
+/*
+ * The model constructed looks like:
+ *                  +-----------+
+ *                  |   input   |
+ *                  +-----+-----+
+ *                        |
+ *      +---------+-------+------+-----------+
+ *      |         |       |      |           |
+ * +----v---+ +---v----+  v  +---v----+ +----v---+
+ * | Conv2d | | Conv2d | ... | Conv2d | | Conv2d |
+ * +----+---+ +---+----+  |  +---+----+ +--+-----+
+ *      |         |       |      |         |
+ *    +-v---------v-+     |    +-v---------v-+
+ *    |     Add     |     |    |     Add     |
+ *    +-------------+     |    +------+------+
+ *           |            |           |
+ *           +-------+    |    +------+
+ *                 +-v----v----v-+
+ *                 |     Add     |
+ *                 +------+------+
+ *                        |
+ *      +---------+-------+------+-----------+
+ *      |         |       |      |           |
+ * +----v---+ +---v----+  v  +---v----+ +----v---+
+ * | Conv2d | | Conv2d | ... | Conv2d | | Conv2d |
+ * +----+---+ +---+----+  |  +---+----+ +--+-----+
+ *      |         |       |      |         |
+ *    +-v---------v-+     |    +-v---------v-+
+ *    |     Add     |     |    |     Add     |
+ *    +------+------+     |    +------+------+
+ *           |            |           |
+ *           +-------+    |    +------+
+ *                   |    |    |
+ *                 +-v----v----v-+
+ *                 |     Add     |
+ *                 +-------------+
+ *                        .
+ *                        .
+ *                        .
+ *      +---------+-------+------+-----------+
+ *      |         |       |      |           |
+ * +----v---+ +---v----+  v  +---v----+ +----v---+
+ * | Conv2d | | Conv2d | ... | Conv2d | | Conv2d |
+ * +----+---+ +---+----+  |  +---+----+ +--+-----+
+ *      |         |       |      |         |
+ *    +-v---------v-+     |    +-v---------v-+
+ *    |     Add     |     |    |     Add     |
+ *    +------+------+     |    +------+------+
+ *           |            |           |
+ *           +-------+    |    +------+
+ *                   |    |    |
+ *                 +-v----v----v-+
+ *                 |    Output   |
+ *                 +-------------+
+ */
+
+template <int Batch, int Size, int Channels, int Kernel, int Layers, int Width>
+std::unique_ptr<FlatBufferModel> GetModel() {
+  absl::BitGen gen;
+  SimpleModelBuilder mb;
+
+  // Batch, Height, Width, Channels
+  const std::vector<int> content_tensor_shapes = {Batch, Size, Size, Channels};
+  auto get_content_tensor_args = [&](std::string name) {
+    return SimpleModelBuilder::TensorArgs{
+        .name = name,
+        .type = kTfLiteFloat32,
+        .shape = content_tensor_shapes,
+    };
+  };
+
+  int input = mb.AddInput(get_content_tensor_args("input"));
+
+  std::vector<std::vector<int>> weights(Width);
+  for (int i = 0; i < Width; ++i) {
+    std::vector<float> weight_values(Channels * Kernel * Kernel * Channels);
+    for (auto& weight_value : weight_values) {
+      weight_value = absl::Uniform(gen, 0.0, 1.0);
+    }
+    std::vector<uint8_t> weight_data(
+        reinterpret_cast<const uint8_t*>(weight_values.data()),
+        reinterpret_cast<const uint8_t*>(weight_values.data()) +
+            weight_values.size() * sizeof(float));
+    weights[i].push_back(mb.AddInternalTensor(SimpleModelBuilder::TensorArgs{
+        .name = absl::StrFormat("weight_%d", i),
+        .type = kTfLiteFloat32,
+        // Output Channels, Kernel Height, Kernel Width, Input Channels
+        .shape = {Channels, Kernel, Kernel, Channels},
+        .buffer = mb.AddBuffer(weight_data),
+    }));
+  }
+
+  std::vector<uint8_t> bias_data(Channels * sizeof(float));
+  int bias = mb.AddInternalTensor(SimpleModelBuilder::TensorArgs{
+      .name = "bias",
+      .type = kTfLiteFloat32,
+      // Output Channels
+      .shape = {Channels},
+      .buffer = mb.AddBuffer(bias_data),
+  });
+
+  int output = mb.AddOutput(get_content_tensor_args("output"));
+
+  for (int i = 1; i <= Layers; ++i) {
+    const TfLiteConvParams conv_params = {
+        .padding = kTfLitePaddingSame,
+        .stride_width = 1,
+        .stride_height = 1,
+        .activation = kTfLiteActNone,
+        .dilation_width_factor = 1,
+        .dilation_height_factor = 1,
+    };
+    int current_output = (i == Layers)
+                             ? output
+                             : mb.AddInternalTensor(get_content_tensor_args(
+                                   absl::StrFormat("intermediate_%d", i)));
+    if (Width == 1) {
+      mb.AddOperator<TfLiteConvParams>({
+          .op = kTfLiteBuiltinConv2d,
+          .inputs = {input, weights[0].back(), bias},
+          .outputs = {current_output},
+          .params = conv_params,
+      });
+    } else {
+      std::vector<int> intermediates;
+      for (int j = 0; j < Width; ++j) {
+        intermediates.push_back(mb.AddInternalTensor(get_content_tensor_args(
+            absl::StrFormat("intermediate_%d_%d", i, j))));
+        mb.AddOperator<TfLiteConvParams>({
+            .op = kTfLiteBuiltinConv2d,
+            .inputs = {input, weights[j].back(), bias},
+            .outputs = {intermediates.back()},
+            .params = conv_params,
+        });
+      }
+      SumTo(mb, content_tensor_shapes, absl::StrFormat("intermediate_%d_", i),
+            intermediates, current_output);
+    }
+    input = current_output;
+  }
+  return mb.Build();
+}
+
+std::pair<std::unique_ptr<FlatBufferModel>, int>
+GetTimeConsumingModelAndBufferSize() {
+  // The model with these parameters runs >500ms on Navi.
+  constexpr int kBatch = 8;
+  constexpr int kSize = 512;
+  constexpr int kChannels = 6;
+  constexpr int kKernel = 15;
+  constexpr int kLayers = 32;
+  constexpr int kWidth = 1;
+  return {GetModel<kBatch, kSize, kChannels, kKernel, kLayers, kWidth>(),
+          kBatch * kSize * kSize * kChannels};
+}
+
+std::vector<float> GetRandomizedFloatData(int size) {
+  absl::BitGen gen;
+  std::vector<float> buffer(size);
+  for (auto& buffer_value : buffer) {
+    buffer_value = absl::Uniform(gen, 0.0, 1.0);
+  }
+  return buffer;
+}
+
+TEST(NeuronDelegateTest, InferenceSelfAborted) {
+  TFLiteSettingsT settings = GetDefaultSettings();
+
+  auto [model, buffer_size] = GetTimeConsumingModelAndBufferSize();
+  ASSERT_NE(model, nullptr);
+  std::vector<float> input = GetRandomizedFloatData(buffer_size);
+
+  // Get the expected output with the given input.
+  TfLiteDelegatePtr delegate = GetDelegateFromSettings(&settings);
+  ASSERT_NE(delegate, nullptr);
+  auto driver = SyncDriver::Create(
+      std::move(delegate), FlatBufferModel::BuildFromModel(model->GetModel()));
+  ASSERT_NE(driver, nullptr);
+  driver->SetInputTensorBuffer(
+      "input", AllocateTfLiteAlignedTensorBuffer(buffer_size * sizeof(float)));
+  ASSERT_EQ(driver->SetInputTensorData("input", input), kTfLiteOk);
+  ASSERT_EQ(driver->AllocateBuffers(), kTfLiteOk);
+  ASSERT_EQ(driver->Invoke(), kTfLiteOk);
+  std::vector<float> expected_output =
+      driver->GetOutputTensorData<float>("output");
+
+  // Set the inference abort time to 1ms.
+  settings.mtk_neuron_settings->inference_abort_time_ms = 1;
+
+  // Try inference the model with an abort time.
+  delegate = GetDelegateFromSettings(&settings);
+  ASSERT_NE(delegate, nullptr);
+  driver = SyncDriver::Create(
+      std::move(delegate), FlatBufferModel::BuildFromModel(model->GetModel()));
+  ASSERT_NE(driver, nullptr);
+  driver->SetInputTensorBuffer(
+      "input", AllocateTfLiteAlignedTensorBuffer(buffer_size * sizeof(float)));
+  ASSERT_EQ(driver->SetInputTensorData("input", input), kTfLiteOk);
+  ASSERT_EQ(driver->AllocateBuffers(), kTfLiteOk);
+
+  // Check the model inference is properly aborted.
+  absl::Time before_invoke = absl::Now();
+  ASSERT_EQ(driver->Invoke(), kTfLiteError);
+  absl::Time after_invoke = absl::Now();
+  // Use a larger time duration to accommodate inaccuracy and overheads.
+  EXPECT_LT(after_invoke - before_invoke, absl::Milliseconds(60));
+
+  // The model inference should be aborted, so the output should be different.
+  std::vector<float> aborted_output =
+      driver->GetOutputTensorData<float>("output");
+  ASSERT_EQ(aborted_output.size(), expected_output.size());
+  EXPECT_THAT(
+      aborted_output,
+      testing::Not(testing::ElementsAreArray(ArrayFp16Eq(expected_output))));
+}
+
+}  // namespace tflite::cros::tests
+
+int main(int argc, char** argv) {
+  testing::InitGoogleTest(&argc, argv);
+  return RUN_ALL_TESTS();
+}